<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>and Ca´ssia Tro-
jahn dos Santos. Ontology alignment evaluation initiative: six years of experience. Journal
on Data Semantics</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <title-group>
        <article-title>Results of the Ontology Alignment Evaluation Initiative 2016?</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Manel Achichi</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Michelle Cheatham</string-name>
          <email>michelle.cheatham@wright.edu</email>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Zlatan Dragisic</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Je´roˆme Euzenat</string-name>
          <email>Jerome.Euzenat@inria.fr</email>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Faria</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alfio Ferrara</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Giorgos Flouris</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Irini Fundulaki</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ian Harrow</string-name>
          <email>ian.harrow@pistoiaalliance.org</email>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Valentina Ivanova</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Jime´nez-Ruiz</string-name>
          <email>ernestoj@ifi.uio.no</email>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Elena Kuss</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Lambrix</string-name>
          <email>patrick.lambrixg@liu.se</email>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Henrik Leopold</string-name>
          <email>h.leopold@vu.nl</email>
          <xref ref-type="aff" rid="aff16">16</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Huanyu Li</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Christian Meilicke</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Stefano Montanelli</string-name>
          <email>stefano.montanellig@unimi.it</email>
          <xref ref-type="aff" rid="aff13">13</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Catia Pesquita</string-name>
          <email>cpesquita@di.fc.ul.pt</email>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Tzanina Saveta</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pavel Shvaiko</string-name>
          <email>pavel.shvaiko@infotn.it</email>
          <xref ref-type="aff" rid="aff12">12</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andrea Splendiani</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Heiner Stuckenschmidt</string-name>
          <email>heinerg@informatik.uni-mannheim.de</email>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Konstantin Todorov</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ca´ssia Trojahn</string-name>
          <email>fcassia.trojahng@irit.fr</email>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ondrˇej Zamazal</string-name>
          <email>ondrej.zamazal@vse.cz</email>
          <xref ref-type="aff" rid="aff14">14</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Data Semantics (DaSe) Laboratory, Wright State University</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Department of Computer Science, University of Oxford</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Department of Informatics, University of Oslo</institution>
          ,
          <country country="NO">Norway</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>INRIA &amp; Univ. Grenoble Alpes</institution>
          ,
          <addr-line>Grenoble</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>IRIT &amp; Universite ́ Toulouse II</institution>
          ,
          <addr-line>Toulouse</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Institute of Computer Science-FORTH</institution>
          ,
          <addr-line>Heraklion</addr-line>
          ,
          <country country="GR">Greece</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>Instituto Gulbenkian de Cieˆncia</institution>
          ,
          <addr-line>Lisbon</addr-line>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>LASIGE, Faculdade de Cieˆncias, Universidade de Lisboa</institution>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>LIRMM/University of Montpellier</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>Linko ̈ping University &amp; Swedish e-Science Research Center</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>Novartis Institutes for Biomedical Research</institution>
          ,
          <addr-line>Basel</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>Pistoia Alliance Inc.</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff12">
          <label>12</label>
          <institution>TasLab</institution>
          ,
          <addr-line>Informatica Trentina, Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff13">
          <label>13</label>
          <institution>Universita` degli studi di Milano</institution>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff14">
          <label>14</label>
          <institution>University of Economics</institution>
          ,
          <addr-line>Prague</addr-line>
          ,
          <country country="CZ">Czech Republic</country>
        </aff>
        <aff id="aff15">
          <label>15</label>
          <institution>University of Mannheim</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff16">
          <label>16</label>
          <institution>Vrije Universiteit Amsterdam</institution>
          ,
          <country country="NL">The Netherlands</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2016</year>
      </pub-date>
      <volume>8797</volume>
      <fpage>61</fpage>
      <lpage>104</lpage>
      <abstract>
        <p>Ontology matching consists of finding correspondences between semantically related entities of two ontologies. OAEI campaigns aim at comparing ? The official results of the campaign are on the OAEI web site.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>-</title>
      <p>ontology matching systems on precisely defined test cases. These test cases can
use ontologies of different nature (from simple thesauri to expressive OWL
ontologies) and use different modalities, e.g., blind evaluation, open evaluation, or
consensus. OAEI 2016 offered 9 tracks with 22 test cases, and was attended by
21 participants. This paper is an overall presentation of the OAEI 2016 campaign.</p>
    </sec>
    <sec id="sec-2">
      <title>1 Introduction</title>
      <p>The Ontology Alignment Evaluation Initiative1 (OAEI) is a coordinated international
initiative, which organises the evaluation of an increasing number of ontology matching
systems [18,21]. Its main goal is to compare systems and algorithms openly and on
the same basis, in order to allow anyone to draw conclusions about the best matching
strategies. Furthermore, our ambition is that, from such evaluations, tool developers can
improve their systems.</p>
      <p>
        Two first events were organised in 2004: (i) the Information Interpretation and
Integration Conference (I3CON) held at the NIST Performance Metrics for
Intelligent Systems (PerMIS) workshop and (ii) the Ontology Alignment Contest held at
the Evaluation of Ontology-based Tools (EON) workshop of the annual International
Semantic Web Conference (ISWC) [41]. Then, a unique OAEI campaign occurred in
2005 at the workshop on Integrating Ontologies held in conjunction with the
International Conference on Knowledge Capture (K-Cap) [
        <xref ref-type="bibr" rid="ref4">4</xref>
        ]. From 2006 until now, the
OAEI campaigns were held at the Ontology Matching workshop, collocated with ISWC
[
        <xref ref-type="bibr" rid="ref12 ref2 ref6 ref8 ref9">19,17,6,14,15,16,2,9,12,8</xref>
        ], which this year took place in Kobe, JP2.
      </p>
      <p>Since 2011, we have been using an environment for automatically processing
evaluations (x2.2), which has been developed within the SEALS (Semantic Evaluation At
Large Scale) project3. SEALS provided a software infrastructure, for automatically
executing evaluations, and evaluation campaigns for typical semantic web tools, including
ontology matching. In the OAEI 2016, all systems were executed under the SEALS
client in all tracks, and evaluated with the SEALS client in all tracks. This year we
welcomed two new tracks: the Disease and Phenotype track, sponsored by the Pistoia
Alliance Ontologies Mapping project, and the Process Model Matching track.
Additionally, the Instance Matching track featured a total of 7 matching tasks based on all
new data sets. On the other hand, the OA4QA track was discontinued this year.</p>
      <p>This paper synthesises the 2016 evaluation campaign. The remainder of the paper
is organised as follows: in Section 2, we present the overall evaluation methodology
that has been used; Sections 3-11 discuss the settings and the results of each of the test
cases; Section 12 overviews lessons learned from the campaign; and finally, Section 13
concludes the paper.
2</p>
    </sec>
    <sec id="sec-3">
      <title>General methodology</title>
      <p>We first present the test cases proposed this year to the OAEI participants (x2.1). Then,
we discuss the resources used by participants to test their systems and the execution</p>
      <sec id="sec-3-1">
        <title>1 http://oaei.ontologymatching.org 2 http://om2016.ontologymatching.org 3 http://www.development.seals-project.eu</title>
        <p>environment used for running the tools (x2.2). Finally, we describe the steps of the
OAEI campaign (x2.3-2.5) and report on the general execution of the campaign (x2.6).
2.1</p>
        <sec id="sec-3-1-1">
          <title>Tracks and test cases</title>
          <p>This year’s OAEI campaign consisted of 9 tracks gathering 22 test cases, and different
evaluation modalities:
The benchmark track (x3): Like in previous campaigns, a systematic benchmark
series has been proposed. The goal of this benchmark series is to identify the areas
in which each matching algorithm is strong or weak by systematically altering an
ontology. This year, we generated a new benchmark based on the original
bibliographic ontology and another benchmark using a film ontology.</p>
          <p>The expressive ontology track offers alignments between real world ontologies
expressed in OWL:
Anatomy (x4): The anatomy test case is about matching the Adult Mouse
Anatomy (2744 classes) and a small fragment of the NCI Thesaurus (3304
classes) describing the human anatomy.</p>
          <p>Conference (x5): The goal of the conference test case is to find all correct
correspondences within a collection of ontologies describing the domain of
organising conferences. Results were evaluated automatically against reference
alignments and by using logical reasoning techniques.</p>
          <p>Large biomedical ontologies (x6): The largebio test case aims at finding
alignments between large and semantically rich biomedical ontologies such as
FMA, SNOMED-CT, and NCI. The UMLS Metathesaurus has been used as
the basis for reference alignments.</p>
          <p>Disease &amp; Phenotype (x7): The disease &amp; phenotype test case aims at finding
alignments between two disease ontologies (DOID and ORDO) as well as
between human (HPO) and mammalian (MP) phenotype ontologies. The
evaluation was semi-automatic: consensus alignments were generated based on those
produced by the participating systems, and the unique mappings found by each
system were evaluated manually.</p>
        </sec>
        <sec id="sec-3-1-2">
          <title>Multilingual</title>
          <p>Multifarm (x8): This test case is based on a subset of the Conference data set,
translated into ten different languages (Arabic, Chinese, Czech, Dutch, French,
German, Italian, Portuguese, Russian, and Spanish) and the corresponding
alignments between these ontologies. Results are evaluated against these
alignments.</p>
        </sec>
        <sec id="sec-3-1-3">
          <title>Interactive matching</title>
          <p>Interactive (x9): This test case offers the possibility to compare different
matching tools which can benefit from user interaction. Its goal is to show if user
interaction can improve matching results, which methods are most promising
and how many interactions are necessary. Participating systems are evaluated
on the conference data set using an oracle based on the reference alignment,
which can generate erroneous responses to simulate user errors.</p>
          <p>Instance matching (x10). The track aims at evaluating the performance of
matching tools when the goal is to detect the degree of similarity between pairs of
test formalism relations confidence modalities
language
benchmark</p>
          <p>anatomy
conference</p>
          <p>largebio
phenotype
multifarm
interactive</p>
          <p>instance
process model</p>
          <p>OWL
OWL
OWL
OWL
OWL
OWL
OWL
OWL
OWL
=
=
=, &lt;=
=
=
=
=, &lt;=
=
&lt;=</p>
          <p>items/instances expressed in the form of OWL Aboxes. Three independent tasks
are defined:
SABINE: The task is articulated in two sub-tasks called inter-lingual mapping and
data linking. Both sub-tasks are based on OWL ontologies containing topics as
instances of the class “Topic”. In inter-lingual mapping, two ontologies are
given, one containing topics in the English language and one containing topics
in the Italian language. The goal is to discover mappings between English and
Italian topics. In data linking, the goal is to discover the DBpedia entity which
better corresponds to each topic belonging to a source ontology.</p>
          <p>SYNTHETIC: The task is articulated in two sub-tasks called UOBM and
SPIMBENCH. In UOBM, the goal is to recognize when two OWL instances
belonging to different data sets, i.e., ontologies, describe the same individual. In
SPIMBENCH, the goal is to determine when two OWL instances describe the
same Creative Work. Data Sets are produced by altering a set of original data.
DOREMUS: The DOREMUS task contains real world data coming from the
French National Library (BnF) and the Philharmonie de Paris (PP). Data are
about classical music work and follow the DOREMUS model (one single
vocabulary for both datasets). Three sub-tasks are defined called nine
heterogeneities, four heterogeneities, and false-positive trap characterized by
different degrees of heterogeneity in work descriptions.</p>
          <p>
            Process Model Matching (x11): The track is concerned with the application of
ontology matching techniques to the problem of matching process models. It is based
on a data set used in the Process Model Matching Campaign 2015 [
            <xref ref-type="bibr" rid="ref3">3</xref>
            ], which has
been converted to an ontological representation. The data set contains nine process
models which represent the application process for a master program of German
universities as well as reference alignments between all pairs of models.
Since 2011, tool developers had to implement a simple interface and to wrap their tools
in a predefined way including all required libraries and resources. A tutorial for tool
wrapping was provided to the participants, describing how to wrap a tool and how to
use the SEALS client to run a full evaluation locally. This client is then executed by
the track organisers to run the evaluation. This approach ensures the reproducibility and
comparability of the results of all systems.
2.3
          </p>
        </sec>
        <sec id="sec-3-1-4">
          <title>Preparatory phase</title>
          <p>Ontologies to be matched and (where applicable) reference alignments have been
provided in advance during the period between June 1st and June 30th, 2016. This gave
potential participants the occasion to send observations, bug corrections, remarks and
other test cases to the organisers. The goal of this preparatory period is to ensure that
the delivered tests make sense to the participants. The final test base was released on
July 15th, 2016. The (open) data sets did not evolve after that.
2.4</p>
        </sec>
        <sec id="sec-3-1-5">
          <title>Execution phase</title>
          <p>
            During the execution phase, participants used their systems to automatically match the
test case ontologies. In most cases, ontologies are described in OWL-DL and serialised
in the RDF/XML format [
            <xref ref-type="bibr" rid="ref11">11</xref>
            ]. Participants can self-evaluate their results either by
comparing their output with reference alignments or by using the SEALS client to compute
precision and recall. They can tune their systems with respect to the non blind
evaluation as long as the rules published on the OAEI web site are satisfied. This phase has
been conducted between July 15th and August 31st, 2016. Unlike previous years, we
requested a mandatory registration of systems and a preliminary evaluation of wrapped
systems by July 31st. This reduced the cost of debugging systems with respect to issues
with the SEALS client during the Evaluation phase as it happened in the past.
2.5
          </p>
        </sec>
        <sec id="sec-3-1-6">
          <title>Evaluation phase</title>
          <p>Participants were required to submit their wrapped tools by August 31st, 2016. Tools
were then tested by the organisers and minor problems were reported to some tool
developers, who were given the opportunity to fix their tools and resubmit them.</p>
          <p>Initial results were provided directly to the participants between September 23rd and
October 15th, 2016. The final results for most tracks were published on the respective
pages of the OAEI website by October 15th, although some tracks were delayed.</p>
          <p>The standard evaluation measures are usually precision and recall computed against
the reference alignments. More details on the evaluation are given in the sections for
the test cases.
2.6</p>
        </sec>
        <sec id="sec-3-1-7">
          <title>Comments on the execution</title>
          <p>Following the recent trend, the number of participating systems has remained
approximately constant at slightly over 20 (see Figure 1). This year was no exception, as we
counted 21 participating systems (out of 30 registered systems). Remarkably,
participating systems have changed considerably between editions, and new systems keep
emerging. For example, this year 10 systems had not participated in any of the
previous OAEI campaigns. The list of participants is summarised in Table 2. Note that some
systems were also evaluated with different versions and configurations as requested by
developers (see test case sections for details).
The goal of the benchmark data set is to provide a stable and detailed picture of each
algorithm. For that purpose, algorithms are run on systematically generated test cases.
The systematic benchmark test set is built around a seed ontology and many variations
of it. Variations are artificially generated by discarding and modifying features from a
seed ontology. Considered features are names of entities, comments, the specialisation
hierarchy, instances, properties and classes. This test focuses on the characterisation of
the behaviour of the tools rather than having them compete on real-life problems. Full
description of the systematic benchmark test set can be found on the OAEI web site.</p>
          <p>Since OAEI 2011.5, the test sets are generated automatically from different seed
ontologies [20]. This year, we used two ontologies:
biblio The bibliography ontology used in the previous years which concerns
bibliographic references and is inspired freely from BibTeX;
film A movie ontology developed in the MELODI team at IRIT (FilmographieV14). It
uses fragments in French and labels in French and English.</p>
          <p>The characteristics of these ontologies are described in Table 3.</p>
          <p>The film data set was not available to participants when they submitted their
systems. The tests were also blind for the organisers since we did not look into them before
running the systems.</p>
          <p>The reference alignments are still restricted to named classes and properties and use
the “=” relation with confidence of 1.
4 https://www.irit.fr/recherches/MELODI/ontologies/</p>
          <p>FilmographieV1.owl
In order to avoid the discrepancy of last year, all systems were run in the most simple
homogeneous setting. So, this year, we can write anew: All tests have been run entirely
in the same conditions with the same strict protocol.</p>
          <p>Evaluations were run on a Debian Linux virtual machine configured with four
processors and 8GB of RAM running under a Dell PowerEdge T610 with 2*Intel Xeon
Quad Core 2.26GHz E5607 processors and 32GB of RAM, under Linux ProxMox 2
(Debian). All matchers where run under the SEALS client using Java 1.8 and a
maximum heap size of 8GB.</p>
          <p>As a result, many systems were not able to properly match the benchmark.
Evaluators availability is not unbounded and it was not possible to pay attention to each system
as much as necessary.</p>
          <p>Participation From the 21 systems participating to OAEI this year, only 10 systems
were providing results for this track. Several of these systems encountered problems:</p>
          <p>However we encountered problems with one very slow matcher (LogMapBio) that
has been run anyway. RiMOM did not terminate, but was able to provide (empty)
alignments for biblio, not for film. No timeout was explicitly set.</p>
          <p>Reported figures are the average of 5 runs. As has already been shown in [20], there
is not much variance in compliance measures across runs.</p>
          <p>Compliance Table 4 synthesises the results obtained by matchers.</p>
          <p>biblio</p>
          <p>F-m.</p>
          <p>Matcher</p>
          <p>Prec.</p>
          <p>Rec.</p>
          <p>Prec.</p>
          <p>film
F-m.</p>
          <p>Rec.</p>
          <p>Systems that participated previously (AML, CroMatcher, Lily, LogMap, LogMapLite,
XMap) still obtain the best results with Lily and CroMatcher still achieving an impressive
.89 F-measure (against .90 and .88 last year). They combine very high precision (.96
and .97) with high recall (.83). The PhenoXX suite of systems return huge but poor
alignments. It is surprising that some of the systems (AML, LogMapLite) do not clearly
outperform edna (our edit distance baseline).</p>
          <p>On the film data set (which was not known from the participants when submitting
their systems, and actually have been generated afterwards), the results of biblio are
fully confirmed: (1) those system able to return results were still able to do it besides
CroMatcher and those unable, were still not able; (2) the order between these systems
and their performances are commensurate. Point (1) shows that these are robust
systems. Point (2) shows that the performances of these system are consistent across data
sets, hence we are indeed measuring something. However, (2) has for exception LogMap
and LogMapBio whose precision is roughly preserved but whose recall dramatically
drops. A tentative explanation is that film contains many labels in French and these two
systems rely too much on WordNet. Anyway, these and CroMatcher seem to show some
overfit to biblio.</p>
          <p>Polarity Besides LogMapLite, all systems have higher precision than recall as usual
and usually very high precision as shown on the triangle graph for biblio (Figure 2).
This can be compared with last year.</p>
          <p>R=.9
R=.8 F=0.9</p>
          <p>R=1. Pre=1fa.lign (b/f)</p>
          <p>P=.9</p>
          <p>Lily (b) P=.8</p>
          <p>CroMatcher (b)
R=.7</p>
          <p>F=0.8</p>
          <p>F=0.7
R=.6</p>
          <p>F=0.6
R=.5</p>
          <p>F=0.5</p>
          <p>R=.4</p>
          <p>R=.3
R=.2
R=.1
recall</p>
          <p>Lily (f)
XMap (f)</p>
          <p>P=.7</p>
          <p>P=.6</p>
          <p>P=.5</p>
          <p>P=.4</p>
          <p>XMap (b)</p>
          <p>LogMap (b) P=.3
LogMapLite (f)</p>
          <p>AML (b)
AML (f) P=.2</p>
          <p>P=.1
precision</p>
          <p>The precision/recall graph (Figure 3) confirms that, as usual, there are a level of
recall unreachable by any system and this is where some of them go to catch their good
F-measure.</p>
          <p>Concerning confidence-weighted measures, there are two types of systems: those
(CroMatcher, Lily) which obviously threshold their results but keep low confidence
values and those (LogMap, XMap, LogMapBio) which provide relatively faithful measures.
The former shows a strong degradation of the measured values while the latters resist</p>
          <p>recall</p>
          <p>refalign
1.00
AML
0.24
LogMap</p>
          <p>0.38
PhenoMF
0.01
edna
0.50
XMap
0.40
LogMapLite</p>
          <p>0.50
PhenoMM
0.01</p>
          <p>Lily
0.82
CroMatcher</p>
          <p>0.81
LogMapBio</p>
          <p>0.19
PhenoMP
0.01
very well with XMap even improving its score. This measure which is supposed to
reward systems able to provide accurate confidence values is beneficial to these faithful
systems.</p>
          <p>Speed Beside LogMapBio which uses alignment repositories on the web to find
matches, all matchers do the task in less than 40 min (for biblio and 12h for film).
There is still a large discrepancy between matchers concerning the time spent from less
than two minutes for LogMapLite, AML and XMap to nearly two hours for LogMapBio
(on biblio).</p>
          <p>time
film</p>
          <p>stdev F-m./s.</p>
          <p>Table 5 provides the average time, time standard deviation and 1/100e F-measure
point provided per second by matchers. The F-measure point provided per second shows
that efficient matchers are, like two years ago, LogMapLite and XMap followed by AML
and LogMap. The correlation between time and F-measure only holds for these systems.</p>
          <p>Time taken by systems is, for most of them, far larger on film than biblio and the
deviation from average increased as well.
This year, there is no increase or decrease of the performance of the best matchers which
are roughly the same as previous years. Precision is still preferred to recall by the best
systems. It seems difficult to other matchers to catch up both in terms of robustness and
performances. This confirms the trend observed last year.
4</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>Anatomy</title>
      <p>The anatomy test case confronts matchers with a specific type of ontologies from the
biomedical domain. We focus on two fragments of biomedical ontologies which
describe the human anatomy5 and the anatomy of the mouse6. This data set has been used
since 2007 with some improvements over the years.
4.1</p>
      <sec id="sec-4-1">
        <title>Experimental Setting</title>
        <p>We conducted experiments by executing each system in its standard setting and we
compare precision, recall, F-measure and recall+. The measure recall+ indicates the
amount of detected non-trivial correspondences. The matched entities in a non-trivial
correspondence do not have the same normalised label. The approach that generates
only trivial correspondences is depicted as baseline StringEquiv in the following section.</p>
        <p>We ran the systems on a server with 3.46 GHz (6 cores) and 8GB RAM allocated
to each matching system. Further, we used the SEALS client to execute our evaluation.
However, we slightly changed the way precision and recall are computed, i.e., the results
generated by the SEALS client vary in some cases by 0.5% compared to the results
presented below. In particular, we removed trivial correspondences in the oboInOwl
namespace like:</p>
        <p>http://...oboInOwl#Synonym = http://...oboInOwl#Synonym
as well as correspondences expressing relations different from equivalence. Using the
Pellet reasoner we also checked whether the generated alignment is coherent, i.e., that
there are no unsatisfiable classes when the ontologies are merged with the alignment.
4.2</p>
      </sec>
      <sec id="sec-4-2">
        <title>Results</title>
        <p>5 http://www.cancer.gov/cancertopics/cancerlibrary/</p>
        <p>terminologyresources/
6 http://www.informatics.jax.org/searches/AMA_form.shtml
AML and XMap joined the track in 2013. DKP-AOM, Lily and CroMatcher participate
for the second year in a row in this track. Lily participated in the track back in 2011.
CroMatcher participated in 2013 but did not produce an alignment within the given
time frame. Thus, this year we have 10 different systems (not counting different
versions) which generated an alignment. For more details, we refer the reader to the papers
presenting the systems.</p>
        <p>Matcher
AML
CroMatcher
XMap
LogMapBio
FCA Map
LogMap
LYAM
Lily
LogMapLite
StringEquiv
LPHOM
Alin
DKP-AOM-Lite
DKP-AOM</p>
        <p>Runtime</p>
        <p>Size Precision</p>
        <p>F-measure</p>
        <p>Recall Recall+</p>
        <p>Unlike the last two editions of the track when 6 systems generated an alignment in
less than 100 seconds, this year only 4 of them were able to complete the alignment
task in this time frame. These are AML, XMap, LogMap and LogMapLite. Similarly to
the last 4 years LogMapLite has the shortest runtime, followed by LogMap, XMap and
AML. Depending on the specific version of the systems, they require between 20 and 50
seconds to match the ontologies. The table shows that there is no correlation between
quality of the generated alignment in terms of precision and recall and required runtime.
This result has also been observed in previous OAEI campaigns.</p>
        <p>The table also shows the results for precision, recall and F-measure. In terms of
F-measure, the top 5 ranked systems are AML, CroMatcher, XMap, LogMapBio and
FCA Map. LogMap is sixth with a F-measure very close to FCA Map. All the long-term
participants in the track showed comparable results (in term or F-measure) to their last
year’s results and at least as good as the results of the best systems in OAEI 2007-2010.
LogMap and XMap generated the same number of correspondences in their alignment
(XMap generated one correspondence more). AML and LogMapBio generated a slightly
different number—16 correspondences more for AML and 18 less for LogMapBio.</p>
        <p>The results for the DKP-AOM systems are identical this year; by contrast, last year
the lite version performed significantly better in terms of the observed measures. While
Lily had improved its 2015 results in comparison to 2011 (precision: from 0.814 to
0.870, recall: from 0.734 to 0.793, and F-measure: from 0.772 to 0.830), this year
it performed similarly to last year. CroMatcher improved its results in comparison to
last year. Out of all systems participating in the anatomy track CroMatcher showed the
largest improvement in the observed measures in comparison to its values from the
previous edition of the track.</p>
        <p>Comparing the F-measures of the new systems, FCA Map (0.882) scored very close
to one of the tracks’ long-term participants LogMap. Another of the new systems—
LYAM—also achieved a good F-measure (0.869) which ranked sixth. As for the other
two systems, LPHOM achieved a slightly lower F-measure than the baseline
(StringEquiv) whereas Alin was considerably below the baseline.</p>
        <p>This year, 9 out of 13 systems achieved an F-measure higher than the baseline which
is based on (normalised) string equivalence (StringEquiv in the table). This is a slightly
better result (percentage-wise) than last year’s (9 out of 15) and similar to 2014’s (7 out
of 10). Two of the new participants in the track and the two DKP-AOM systems achieved
an F-measure lower than the baseline. LPHOM scored under the StringEquiv baseline but
at the same time it is the system that produced the highest number of correspondences.
Its precision is significantly lower than the other three systems which scored under the
baseline and generated only trivial correspondences.</p>
        <p>This year seven systems produced coherent alignments which is comparable to the
last two years, when 7 out of 15 and 5 out of 10 systems achieved this. From the five
best systems only FCA Map produced an incoherent alignment.
4.3
Like for OAEI in general, the number of participating systems in the anatomy track
this year was lower than in 2015 and 2013 but higher than in 2014, and there was a
combination of newly-joined systems and long-term participants.</p>
        <p>The systems that participated in the previous edition scored similarly to their
previous results, indicating that no substantial developments were made with regard to
this track. Of the newly-joined systems, (FCA Map and LYAM) ranked 4th and 6th with
respect to the F-measure.
5</p>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>Conference</title>
      <p>5.1</p>
      <sec id="sec-5-1">
        <title>Test data</title>
        <p>The conference test case requires matching several moderately expressive ontologies
from the conference organisation domain.</p>
        <p>The data set consists of 16 ontologies in the domain of organising conferences. These
ontologies have been developed within the OntoFarm project7.</p>
        <p>The main features of this test case are:
– Generally understandable domain. Most ontology engineers are familiar with
organising conferences. Therefore, they can create their own ontologies as well as
evaluate the alignments among their concepts with enough erudition.
– Independence of ontologies. Ontologies were developed independently and based
on different resources, they thus capture the issues in organising conferences from
different points of view and with different terminologies.</p>
        <sec id="sec-5-1-1">
          <title>7 http://owl.vse.cz:8080/ontofarm/</title>
          <p>– Relative richness in axioms. Most ontologies were equipped with OWL DL axioms
of various kinds; this opens a way to use semantic matchers.</p>
          <p>Ontologies differ in their numbers of classes and properties, in expressivity, but also
in underlying resources.
5.2</p>
        </sec>
      </sec>
      <sec id="sec-5-2">
        <title>Results</title>
        <p>We provide results in terms of F-measure, comparison with baseline matchers and
results from previous OAEI editions and precision/recall triangular graph based on sharp
reference alignments. This year we can provide comparison between OAEI editions
of results based on the uncertain version of reference alignment and on violations of
consistency and conservativity principles.</p>
      </sec>
      <sec id="sec-5-3">
        <title>Evaluation based on sharp reference alignments We evaluated the results of partic</title>
        <p>ipants against blind reference alignments (labelled as rar2). This includes all pairwise
combinations between 7 different ontologies, i.e., 21 alignments.</p>
        <p>These reference alignments have been made in two steps. First, we have generated
them as a transitive closure computed on the original reference alignments. In order to
obtain a coherent result, conflicting correspondences, i.e., those causing
unsatisfiability, have been manually inspected and removed by evaluators. The resulting reference
alignments are labelled as ra2. Second, we detected violations of conservativity
using the approach from [39] and resolved them by an evaluator. The resulting reference
alignments are labelled as rar2. As a result, the degree of correctness and completeness
of the new reference alignments is probably slightly better than for the old one.
However, the differences are relatively limited. Whereas the new reference alignments are
not open, the old reference alignments (labeled as ra1 on the conference web page) are
available. These represent close approximations of the new ones.</p>
        <p>Table 7 shows the results of all participants with regard to the reference alignment
rar2. F0:5-measure, F1-measure and F2-measure are computed for the threshold that
provides the highest average F1-measure. F1 is the harmonic mean of precision and
recall where both are equally weighted; F2 weights recall higher than precision and
F0:5 weights precision higher than recall. The matchers shown in the table are ordered
according to their highest average F1-measure. We employed two baseline matchers.
edna (string edit distance matcher) is used within the benchmark test case and with
regard to performance it is very similar as the previously used baseline2 in the
conference track; StringEquiv is used within the anatomy test case. This year these
baselines divide matchers into two performance groups. The first group consists of
matchers (CroMatcher, AML, LogMap, XMap, LogMapBio, FCA Map, DKP-AOM, NAISC and
LogMapLite) having better (or the same) results than both baselines in terms of highest
average F1-measure. Other matchers (Lily, LPHOM, Alin and LYAM) performed worse
than both baselines. The performance of all matchers (except LYAM) regarding their
precision, recall and F1-measure is visualised in Figure 4. Matchers are represented as
squares or triangles. Baselines are represented as circles.</p>
        <p>Further, we evaluated the performance of matchers separately on classes and
properties. We compared the position of tools within overall performance groups and within
only classes and only properties performance groups. We observed that while the
position of matchers changed slightly in overall performance groups in comparison with
CroMatcher</p>
        <p>AML
LogMap</p>
        <p>XMap
LogMapBio
FCA Map
DKP-AOM</p>
        <p>NAISC
edna
LogMapLite
StringEquiv</p>
        <p>Lily
LPHOM</p>
        <p>Alin
LYAM</p>
        <p>Prec. F0:5-m. F1-m. F2-m. Rec. Inc.Align. Conser.V. Consist.V.
only classes performance groups, a couple of matchers (DKP-AOM and FCA Map)
worsen their position from overall performance groups with regard to their position
in only properties performance groups due to the fact that they do not match
properties at all (Alin and Lily also fall into this category). More details about these evaluation
modalities are on the conference web page.</p>
        <p>Comparison with previous years with regard to rar2 Seven matchers also participated
in this test case in OAEI 2015. The largest improvement was achieved by CroMatcher
(precision increased from .57 to .74 and recall increased from .47 to .65).</p>
      </sec>
      <sec id="sec-5-4">
        <title>Evaluation based on uncertain version of reference alignments The confidence val</title>
        <p>ues of all matches in the sharp reference alignments for the conference track are all 1.0.
For the uncertain version of this track, the confidence value of a match has been set
equal to the percentage of a group of people who agreed with the match in question
(this uncertain version is based on the reference alignment labeled ra1). One key thing
to note is that the group was only asked to validate matches that were already present in
the existing reference alignments – so some matches had their confidence value reduced
from 1.0 to a number near 0, but no new match was added.</p>
        <p>There are two ways that we can evaluate matchers according to these “uncertain”
reference alignments, which we refer to as discrete and continuous. The discrete
evaluation considers any match in the reference alignment with a confidence value of 0.5 or
greater to be fully correct and those with a confidence less than 0.5 to be fully incorrect.
Similarly, a matcher’s match is considered a “yes” if the confidence value is greater than
or equal to the matcher’s threshold and a “no” otherwise. In essence, this is the same as
the “sharp” evaluation approach, except that some matches have been removed because
less than half of the crowdsourcing group agreed with them. The continuous evaluation</p>
        <p>F1-measure=0.7</p>
        <p>F1-measure=0.6
F1-measure=0.5
AML
CroMatcher
DKP-AOM
FCA-Map
Lily
LogMap
LogMapBio
LogMapLite
LPHOM
strategy penalises a matcher more if it misses a match on which most people agree than
if it misses a more controversial match. For instance, if A B with a confidence of
0.85 in the reference alignment and a matcher gives that correspondence a confidence of
0.40, then that is counted as 0:85 0:40 = 0:34 of a true positive and 0:85{0:40 = 0:45
of a false negative.</p>
        <p>
          Out of the 13 matchers, three (DKP-AOM, FCA-Map and LogMapLite) use 1.0 as the
confidence values for all matches they identify. Two of the remaining ten (Alin and
CroMatcher) have some variation in confidence values, though the majority are 1.0. The rest
of the systems have a fairly wide variation of confidence values. Last year, the majority
of these values were near the upper end of the [
          <xref ref-type="bibr" rid="ref1">0,1</xref>
          ] range. This year we see much more
variation in the average confidence values. For example, LopMap’s confidence values
range from 0.29 to 1.0 and average 0.78 whereas Lily’s range from 0.22 to 0.41 with an
average of 0.33.
        </p>
        <p>Discussion When comparing the performance of the matchers on the uncertain
reference alignments versus that on the sharp version, we see that in the discrete case all
matchers performed slightly better. Improvement in F-measure ranged from 1 to 8
percentage points over the sharp reference alignment. This was driven by increased recall,
which is a result of the presence of fewer “controversial” matches in the uncertain
version of the reference alignment.</p>
        <p>The performance of most matchers is similar regardless of whether a discrete or
continuous evaluation methodology is used (provided that the threshold is optimised to
achieve the highest possible F-measure in the discrete case). The primary exceptions</p>
        <p>Alin</p>
        <p>AML
CroMatcher
DKP-AOM
FCA-Map</p>
        <p>Lily</p>
        <p>LogMap
LogMapBio
LogMapLite</p>
        <p>LPHOM
Light YAM++</p>
        <p>NAISC
XMap</p>
        <p>Sharp
Prec. F1-m. Rec.</p>
        <p>Discrete
Prec. F1-m. Rec.</p>
        <p>Continuous</p>
        <p>Prec. F1-m. Rec.
to this are Lily and NAISC. These matchers perform significantly worse when evaluated
using the continuous version of the metrics. In Lily’s case, this is because it assigns very
low confidence values to some matches in which the labels are equivalent strings, which
many crowdsourcers agreed with unless there was a compelling technical reason not to.
This hurts recall, but using a low threshold value in the discrete version of the evaluation
metrics “hides” this problem. NAISC has the opposite issue: it assigns relatively high
confidence values to some matches that most people disagree with, such as “Assistant”
and “Listener” (confidence value of 0.89). This hurts precision in the continuous case,
but is taken care of by using a high threshold value (1.0) in the discrete case.</p>
        <p>Seven matchers from this year also participated last year, and thus we are able to
make some comparisons over time. The F-measures of all matchers either held constant
or improved when evaluated against the uncertain reference alignments. Most matchers
made modest gains (in the neighborhood of 1 to 6 percentage points). CroMatcher made
the largest improvement, and it is now the second-best matcher when evaluated in this
way. AgreementMakerLight remains the top performer.</p>
        <p>Perhaps more importantly, the difference in the performance of most matchers
between the discrete and continuous evaluation has shrunk between this year and last year.
This is an indication that more matchers are providing confidence values that reflect the
disagreement of humans on various matches.</p>
      </sec>
      <sec id="sec-5-5">
        <title>Evaluation based on violations of consistency and conservativity principles We</title>
        <p>performed evaluation based on detection of conservativity and consistency violations
[39,40]. The consistency principle states that correspondences should not lead to
unsatisfiable classes in the merged ontology; the conservativity principle states that
correspondences should not introduce new semantic relationships between concepts from
one of the input ontologies.</p>
        <p>Table 7 summarises statistics per matcher. The table shows the number of
unsatisfiable TBoxes after the ontologies are merged (Inc. Align.), the total number of all
conservativity principle violations within all alignments (Conser.V.) and the total
number of all consistency principle violations (Consist.V.).</p>
        <p>Seven tools (Alin, AML, DKP-AOM, LogMap, LogMapBio, LPHOM and XMap) have
no consistency principle violations (in comparison to five last year) and one tool (LYAM)
generated only one incoherent alignment. There are two tools (Alin, LPHOM) that have
no conservativity principle violations, and four more that have an average of only
one conservativity principle violation (XMap, LogMap, LogMapBio and DKP-AOM). We
should note that these conservativity principle violations can be “false positives” since
the entailment in the aligned ontology can be correct although it was not derivable in
the single input ontologies.</p>
        <p>In conclusion, this year eight matchers performed better than both baselines on
reference alignments which is not only consistent but also conservative. Further, this year
seven matchers generated coherent alignments (against five matchers last year and four
matchers the year before). This confirms the trend that increasingly matchers
generate coherent alignments. Based on the uncertain reference alignments, more matchers
are providing confidence values that reflect the disagreement of humans on various
matches.
6</p>
      </sec>
    </sec>
    <sec id="sec-6">
      <title>Large biomedical ontologies (largebio)</title>
      <p>The largebio test case requires to match the large and semantically rich biomedical
ontologies FMA, SNOMED-CT, and NCI, which contain 78,989, 306,591 and 66,724
classes, respectively.
The test case has been split into three matching problems: FMA-NCI, FMA-SNOMED
and SNOMED-NCI. Each matching problem has been further divided in 2 tasks
involving differently sized fragments of the input ontologies: small overlapping fragments
versus whole ontologies (FMA and NCI) or large fragments (SNOMED-CT).</p>
      <p>
        The UMLS Metathesaurus [
        <xref ref-type="bibr" rid="ref5">5</xref>
        ] has been selected as the basis for reference
alignments. UMLS is currently the most comprehensive effort for integrating
independentlydeveloped medical thesauri and ontologies, including FMA, SNOMED-CT, and NCI.
      </p>
      <p>Although the standard UMLS distribution does not directly provide alignments (in
the sense of [21]) between the integrated ontologies, it is relatively straightforward to
extract them from the information provided in the distribution files (see [25] for details).</p>
      <p>It has been noticed, however, that although the creation of UMLS alignments
combines expert assessment and auditing protocols they lead to a significant number of
logical inconsistencies when integrated with the corresponding source ontologies [25].</p>
      <p>Since alignment coherence is an aspect of ontology matching that we aim to
promote, in previous editions we provided coherent reference alignments by refining the
UMLS mappings using the Alcomo (alignment) debugging system [31], LogMap’s
(alignment) repair facility [24], or both [26].</p>
      <p>However, concerns were raised about the validity and fairness of applying
automated alignment repair techniques to make reference alignments coherent [35]. It is
clear that using the original (incoherent) UMLS alignments would be penalising to
ontology matching systems that perform alignment repair. However, using automatically
repaired alignments would penalise systems that do not perform alignment repair and
also systems that employ a repair strategy that differs from that used on the reference
alignments [35].</p>
      <p>Thus, as of the 2014 edition, we arrived at a compromising solution that should be
fair to all ontology matching systems. Instead of repairing the reference alignments as
normal, by removing correspondences, we flagged the incoherence-causing
correspondences in the alignments by setting the relation to “?” (unknown). These “?”
correspondences will neither be considered as positive nor as negative when evaluating the
participating ontology matching systems, but will simply be ignored. This way, systems
that do not perform alignment repair are not penalised for finding correspondences that
(despite causing incoherences) may or may not be correct, and systems that do perform
alignment repair are not penalised for removing such correspondences.</p>
      <p>To ensure that this solution was as fair as possible to all alignment repair strategies,
we flagged as unknown all correspondences suppressed by any of Alcomo, LogMap or
AML [?], as well as all correspondences suppressed from the reference alignments of
last year’s edition (using Alcomo and LogMap combined). Note that, we have used the
(incomplete) repair modules of the above mentioned systems.</p>
      <p>The flagged UMLS-based reference alignment for the OAEI 2016 campaign is
summarised in Table 9.</p>
      <p>Reference alignment “=” corresp. “?” corresp.</p>
      <p>FMA-NCI
FMA-SNOMED
SNOMED-NCI</p>
      <sec id="sec-6-1">
        <title>Evaluation setting, participation and success</title>
        <p>We have run the evaluation on a Ubuntu Laptop with an Intel Core i7-4600U CPU @
2.10GHz x 4 and allocating 15Gb of RAM. Precision, recall and F-measure have been
computed with respect to the UMLS-based reference alignment. Systems have been
ordered in terms of F-measure.</p>
        <p>This year, out of the 21 systems participating in OAEI 2016, 13 were registered to
participate in the largebio track, and 11 of these were able to cope with at least one
of the largebio tasks within a 2 hour time frame. However, only 6 systems were able
to complete more than one task, and only 4 systems completed all 6 tasks in this time
frame.
6.3</p>
      </sec>
      <sec id="sec-6-2">
        <title>Background knowledge</title>
        <p>Regarding the use of background knowledge, LogMap-Bio uses BioPortal as a mediating
ontology provider, that is, it retrieves from BioPortal the most suitable top-10 ontologies
for the matching task.</p>
        <p>LogMap uses normalisations and spelling variants from the general (biomedical)
purpose UMLS Lexicon.</p>
        <p>AML has three sources of background knowledge which can be used as mediators
between the input ontologies: the Uber Anatomy Ontology (Uberon), the Human
Disease Ontology (DOID) and the Medical Subject Headings (MeSH).</p>
        <p>System
LogMapLite
AML
LogMap
LogMapBio
XMap
FCA-Map
Lily
LYAM
DKP-AOM
DKP-AOM-Lite
Alin
# Systems</p>
        <p>FMA-NCI
Task 1 Task 2</p>
        <p>FMA-SNOMED
Task 3 Task 4</p>
        <p>SNOMED-NCI</p>
        <p>Task 5 Task 6</p>
        <p>XMap uses synonyms provided by the UMLS Metathesaurus. Note that matching
systems using UMLS Metathesaurus as background knowledge will have a notable
advantage since the largebio reference alignment is also based on the UMLS
Metathesaurus.
Together with precision, recall, F-measure and run times we have also evaluated the
coherence of alignments. We report (1) the number of unsatisfiabilities when reasoning
with the input ontologies together with the computed alignments, and (2) the ratio of
unsatisfiable classes with respect to the size of the union of the input ontologies.</p>
        <p>We have used the OWL 2 reasoner HermiT [33] to compute the number of
unsatisfiable classes. For the cases in which HermiT could not cope with the input ontologies
and the alignments (in less than 2 hours) we have provided a lower bound on the number
of unsatisfiable classes (indicated by ) using the OWL 2 EL reasoner ELK [27].</p>
        <p>In this OAEI edition, only three distinct systems have shown alignment repair
facilities: AML, LogMap and its LogMap-Bio variant, and XMap (which reuses the repair
techniques from Alcomo [31]). Tables 11-12 (see last two columns) show that even the
most precise alignment sets may lead to a huge number of unsatisfiable classes. This
proves the importance of using techniques to assess the coherence of the generated
alignments if they are to be used in tasks involving reasoning. We encourage ontology
matching system developers to develop their own repair techniques or to use
state-ofthe-art techniques such as Alcomo [31], the repair module of LogMap (LogMap-Repair)
[24] or the repair module of AML [?], which have worked well in practice [26,22].</p>
      </sec>
      <sec id="sec-6-3">
        <title>Runtimes and task completion</title>
        <p>Table 10 shows which systems were able to complete each of the matching tasks in
less than 24 hours and the required computation times. Systems have been ordered with
respect to the number of completed tasks and the average time required to complete
them. Times are reported in seconds.</p>
        <p>The last column reports the number of tasks that a system could complete. For
example, 8 system were able to complete all six tasks. The last row shows the number
.</p>
        <p>.</p>
        <p>O</p>
        <p>O</p>
        <p>O
0 0 0</p>
        <p>0 0 0
R</p>
        <p>R</p>
        <p>R
D</p>
        <p>D</p>
        <p>D
O</p>
        <p>O</p>
        <p>O
0 0 0
0</p>
        <p>2
0 0 0 P</p>
        <p>0 0 0 P
0 0 0</p>
        <p>0 0 0</p>
        <p>L
X o A T
M g</p>
        <p>M M o
a o
p a L l</p>
        <p>p
2
,2 42 54 iTm
6 7 0</p>
        <p>6 2 T
7 8 0 N</p>
        <p>1 4
1 299 93 FP
2 225 28 FN
.7 .
O</p>
        <p>L
X o A T
M g</p>
        <p>M M o</p>
        <p>o
a
p a L l</p>
        <p>p
2
,3 44 52 iTm
5 0 5
2</p>
        <p>e
0 0 0 P
.9 .9 .9 r
3 9 2 ec
3 4 9 .
0 0 0 R
.7 .
0 0 0 P</p>
        <p>Fig. 6. Average time between requests per task in the Conference data set (whiskers: Q1-1,5IQR,
Q3+1,5IQR, IQR=Q3-Q1). The labels under the system names show the average number of
requests and the mean time between the requests (calculated by taking the average of the average
request intervals per task) for the ten runs and all tasks.
tial set. LogMap and AML both request feedback on only selected mapping candidates
(based on their similarity patterns or their involvement in unsatisfiabilities) and only
present one mapping at a time to the user. XMap also presents one mapping at a time
and asks mainly for true negatives. Only Alin employs the new feature in this year’s
evaluation: analysing several conflicting mappings simultaneously, whereby a system
can present up to three mappings together to the oracle, provided that each mapping
presented has a mapped entity, i.e., class or property, in common with at least one other
mapping presented.</p>
        <p>The performance of the systems improves when interacting with a perfect oracle
compared to no interaction. Although systems’ performance deteriorates when moving
towards larger error rates there are still benefits from the user interaction—some of the
systems’ measures stay above their non-interactive values even for the larger error rates.
For the Anatomy track Alin detects only trivial correspondences in the non-interactive
version while user interactions led to detecting some non-trivial correspondences.</p>
        <p>The impact of the oracle’s errors is linear for Alin, AML and XMap and supra-linear
for LogMap for all data sets. The ”Positive Precision” value affects the true positives
and false positives, and the ”Negative Precision” value affects the true negatives and
Fig. 7. Time between requests per task in the HP-MP data set (whiskers: Q1-1,5IQR, Q3+1,5IQR,
IQR=Q3-Q1). The labels under the system names show the number of requests and the mean time
between the requests.
false negatives. The more a system relies on the oracle, the more sensitive it will be to
its errors.</p>
        <p>In general, XMap performs very few requests to the oracle compared to the other
systems.</p>
        <p>
          Two models for system response times are frequently used in the literature [
          <xref ref-type="bibr" rid="ref10">10</xref>
          ]:
Shneiderman and Seow take different approaches to categorise the response times.
Shneiderman takes a task-centred view and sorts the response times in four categories
according to task complexity: typing, mouse movement (50-150 ms), simple frequent
tasks (1 s), common tasks (2-4 s) and complex tasks (8-12 s). He suggests that the user
is more tolerable to delays with the growing complexity of the task at hand.
Unfortunately, no clear definition is given for how to define the task complexity. Seow’s model
looks at the problem from a user-centred perspective by considering the user
expectations towards the execution of a task: instantaneous (100-200 ms), immediate (0.5-1
s), continuous (2-5 s), captive (7-10 s); Ontology alignment is a cognitively demanding
task and can fall into the third or fourth categories in both models. In this regard the
response times (request intervals as we call them above) observed in all data sets fall
into the tolerable and acceptable response times, and even into the first categories, in
both models. The request intervals for both AML and LogMap stay under 3 ms for all
data sets. Alin’s request intervals are higher, but still in the tenth of second range. It
could be the case however that the user could not take advantage of very low response
times because the task complexity may result in higher user response time (analogically
it measures the time the user needs to respond to the system after the system is ready).
        </p>
        <p>Regarding the number of unsatisfiable classes resulting from the alignments we
observe some expected variations as the error increases. We note that, with interaction,
the alignments produced by the systems are typically larger than without interaction,
which makes the repair process harder. The introduction of oracle errors complicates
the process further, and may make an alignment irreparable if the system follows the
oracle’s feedback blindly.
10</p>
      </sec>
    </sec>
    <sec id="sec-7">
      <title>Instance matching</title>
      <p>The instance matching track aims at evaluating the performance of matching tools
when the goal is to detect the degree of similarity between pairs of items/instances
expressed in the form of RDF data. The track is organized in three independent tasks
called SABINE, SYNTHETIC and DOREMUS. Each test is based on two data sets called
source and target and the goal is to discover the matching pairs, i.e., mappings, among
the instances in the source data set and the instances in the target data set.
Fig. 9. Time between requests per task in the FMA-NCI data set (whiskers: Q1-1,5IQR,
Q3+1,5IQR, IQR=Q3-Q1). The labels under the system names show the number of requests
and the mean time between the requests.</p>
      <p>For the sake of clarity, we split the presentation of task results in three different
sections as follows.</p>
      <sec id="sec-7-1">
        <title>Results of the SABINE task</title>
        <p>SABINE is a modular benchmark in the domain of European politics for Social
Business Intelligence (SBI) and it includes an ontology with 500 topics, both in English and
Italian languages. The task is articulated in two sub-tasks called inter-lingual mapping
and data linking.</p>
        <p>In inter-lingual mapping, source and target datasets are OWL ontologies containing
topics as instances of the class “Topic”. The source ontology contains topics in the
English language; the target ontology contains other topics in the Italian language. The
goal is to discover mappings between English and Italian topics by also defining the
kind of relation which is most suitable for describing the discovered mapping between
two matching topics.</p>
        <p>In data linking, just the source dataset is defined and it is given to the participants
as an OWL ontology containing topics as instances of the class “Topic”. The goal is to
discover the best corresponding DBpedia entity for each topic in the source ontology.</p>
        <p>
          The SABINE sub-tasks are defined as open tests, meaning that the set of expected
mappings, i.e., reference alignment, is given in advance to the participants and it
constitutes the gold standard for result evaluation. The task size is around 23K ontology
instances to consider. The gold standard has been defined through crowdsourcing
validation though the Argo system14. For creating the gold standard, workers are called to
recognize and confirm the mapping between instance topics of source and target
ontologies. In particular, a task is represented as a choice question in which a topic of the
source ontology is specified and a number of instance topics of the target ontology are
provided as possible mappings. A worker receiving a task to execute has to consider
a source topic and to choose the most appropriate mapping with a target topic among
those provided as possible options. Multi-worker task assignment and consensus
evaluation techniques are defined in Argo for quality assessment of the task result. A task is
assigned to a group G of 6 different workers. A group member autonomously executes
a task and independently produces the answer according to her/his personal feeling and
judgement. Given a task, its result is defined as an answer agreement, i.e., consensus,
among the members of the group that executed the task. Two workers agree on a task
result when they selected the same target topic as mapping with the given source topic.
14 http://island.ricerca.di.unimi.it/projects/argo/ (in Italian).
The mapping between the source and the target topics is confirmed and inserted in the
gold standard when the task answer having the highest degree of consensus within the
group G is supported by a qualified majority larger than 50%. Conversely, when a
qualified majority of workers is not found in G, the task is uncommitted and it is scheduled
for re-execution by a different group of workers with higher reliability. Further details
on the Argo techniques for task management are provided in [
          <xref ref-type="bibr" rid="ref7">7</xref>
          ]. The gold standard of
the SABINE task contains 249 crowd-validated mappings for the inter-lingual sub-task
and 338 crowd-validated mappings for the data linking sub-task.
        </p>
        <p>Participants to the SABINE sub-tasks are LogMapIm, AML, LogMapLite, and
RiMOM. Results are shown in Table 43. For each test, the tool performances are expressed
in terms of precision, recall, and F-measure.</p>
        <p>Inter-lingual mapping Data linkinging</p>
        <p>Precision F-measure Recall Precision F-measure Recall
LogMapIm
AML
LogMapLite
RiMOM</p>
        <p>We focus our considerations on AML and RiMOM that provided high-value results
for precision, recall, and F-measure on both inter-lingual mapping and data linking
subtasks. In particular, RiMOM outperforms AML on the inter-lingual mapping sub-task, in
that both precision and recall values of RiMOM are higher than the corresponding values
of AML. However, both tools are over 90% for precision and recall values, meaning
that mapping corresponding instances of different languages is a successfully-addressed
task by RiMOM and AML. In the data linking sub-task, AML outperforms RiMOM on
precision and the difference between the tools on this result value is significant (i.e.,
AML &gt;90% and RiMOM &lt;50% on the precision value). On the opposite, for recall, we
note that the RiMOM value is higher than the AML value, and the result of both tools
is very positive (i.e., &gt;85%). We argue that these results on the data linking sub-task
are due to the problem of selecting the most appropriate mapping when a number of
possible alternatives are available. Both AML and RiMOM are successful in providing a
set of candidate DBpedia entities as target mapping with a given OWL instance (i.e.,
high recall value). On the opposite, the capability to choose/select the most appropriate
mapping among the set of available options is still challenging and only AML succeeds
in providing high-quality results on this task (i.e., high precision value).</p>
      </sec>
      <sec id="sec-7-2">
        <title>Results of the SYNTHETIC task</title>
        <p>UOBM and SPIMBENCH tasks are two of the evaluation tasks of instance matching
tools where the goal is to determine when two OWL instances describe the same real
world object. For the first task, the data sets have been produced by altering a set of
source data and generated by SPIMBENCH [37] with the aim to generate descriptions
of the same entity where value-based, structure-based and semantics-aware
transformations are employed in order to create the target data. While, for the latter task the data
sets have been generated with the University Ontology Benchmark (UOBM) [30] and
transformed with the LANCE benchmark generator [36].</p>
        <p>For both tasks, the transformations applied were a combination of value-based,
structure-based, and semantics-aware test cases. The value-based transformations
consider mainly typographical errors and different data formats, the structure-based
transformations consider transformations applied on the structure of object and datatype
properties and the semantics-aware transformations are transformations at the instance
level considering the TBox information. The latter are used to examine if the matching
systems take into account RDFS and OWL semantics in order to discover
correspondences between instances that can be found only by considering information found in
the TBox.</p>
        <p>We stress that an instance in the source data set can have none or one matching
counterpart in the target data set. A data set is composed of a TBox and a corresponding
ABox. Source and target data sets share almost the same TBox (differences in the
properties, due to the structure-based transformations). For SPIMBENCH, the sandbox scale
is 10K triples 380 instances while the mainbox scale is 50K triples 1800 instances.
We asked the participants to match the Creative Works instances (NewsItem, BlogPost
and Programme) in the source data set against the instances of the corresponding class
in the target data set. For UOBM, the sandbox scale is 14K triples 2.5K instances
while the mainbox scale is 60K triples 10K instances. We asked the participants to
match all the instances that are not common to the two data sets. For both tasks, we
expected to receive a set of links denoting the pairs of matching instances that they found
to refer to the same entity.</p>
        <p>The participants to these tasks are LogMap, AML and RiMOM. For evaluation, we
built a ground truth containing the set of expected links where an instance i1 in the
source data set is associated with an instance in the target data set that has been
generated as an altered description of i1.</p>
        <p>The way that the transformations were done, was to apply value-based,
structurebased and semantics-aware transformations, on different triples pertaining to one class
instance.</p>
        <p>The systems were judged on the basis of precision, recall and F-measure results that
are shown in Tables 44 and 45.</p>
        <p>Sandbox task
Precision F-measure Recall</p>
        <p>LogMap responds well regarding the SPIMBENCH task, while the performance
drops when matching the data sets of the UOBM task. LogMap is automatic and does
not require the definition of a configuration file in contrast to AML and RiMOM.
LogMap
AML
RiMOM</p>
        <p>AML responds well regarding the SPIMBENCH task, while the performance drops
when matching the data sets of the UOBM task. AML had to turn off the reasoner in
order to handle missing information about the domain and range of TBox properties.</p>
        <p>LogMap and AML produce links that are quite often correct (resulting in a good
precision) but fail in capturing a large number of the expected links (resulting in a
lower recall).</p>
        <p>RiMOM performs better than any other system for most of the tasks; it performs
excellent in the case of SPIMBENCH but, although it exhibits the best results for the
Sandbox track of UOBM, its performance drops for the Mainbox track. For RiMOM,
the probability of capturing a correct link is high, but the probability of a retrieved link
to be correct is lower, resulting in a high recall but not a high precision.</p>
        <p>The main comments for the SPIMBENCH and UOBM tasks are:
– LogMap and AML have consistent behaviour regarding Sandbox and Mainbox.
– RiMOM has a consistent behaviour for the SPIMBENCH task and an inconsistent
behaviour for the UOBM task.
– All systems performed well for the SPIMBENCH task.
– The UOBM data sets seem to be more “difficult” for both IM systems, and this
difficulty stems from the data set itself, rather than from the transformations imposed
by LANCE.
– The UOBM data sets seem to be more difficult for both IM systems, and this
difficulty stems from the data set itself, rather than from the transformations imposed
by LANCE. In particular, an important source of difficulty for the systems is that
the URIs of the instances in the data set look very similar to each other, so even the
change of a URI can lead to false positives or false negatives.</p>
      </sec>
      <sec id="sec-7-3">
        <title>Results of the DOREMUS task</title>
        <p>
          The DOREMUS task, having its premier at OAEI, contains real world data sets coming
from two major French cultural institutions—The BnF (French National Library) and
the PP (Philharmonie de Paris). The data are about classical music works and follow the
DOREMUS model (one single vocabulary for both data sets) issued from the
DOREMUS project15. Each data entry, or instance, is a bibliographical record about a musical
piece, containing properties such as the composer, the title(s) of the work, the year of
creation, the key, the genre, the instruments, to name a few. These data have been
converted to RDF from their original UNI- and INTER-MARC format and anchored to the
DOREMUS ontology and a set of domain controlled vocabularies by the help of the
marc2rdf converter16, developed for this purpose within the DOREMUS Project (for
15 http://www.doremus.org
16 https://github.com/DOREMUS-ANR/marc2rdf
more details on the conversion method and on the ontology we refer to [
          <xref ref-type="bibr" rid="ref1">1</xref>
          ] and [29]).
Note that these data are highly heterogeneous. We have selected works described both
at the BnF and at the PP with different degrees of heterogeneity in their descriptions.
The data sets have been selected in three sub-tasks.
        </p>
        <p>Nine heterogeneities. This task consists in aligning two small data sets, BnF-1 and
PP1, containing about 40 instances each, by discovering 1:1 equivalence relations between
their instances. There are 9 types of heterogeneities that these data manifest, that have
been identified by the music library experts, such as multilingualism, differences in
catalogues, differences in spelling, different degrees of description (number of properties).
Four heterogeneities. This task consists in aligning two larger data sets, BnF-2 and
PP-2, containing about 200 instances each, by discovering 1:1 equivalence relations
between the instances that they contain. There are 4 types of heterogeneities that these
data manifest, that we have selected from the nine in Task 1 and that appear to be
the most problematic: 1) Orthographical differences, 2) Multilingual titles, 3) Missing
properties, 4) Missing titles.</p>
        <p>The False Positives Trap. This task consists in correctly disambiguating the instances
contained in two data sets, BnF-3 and PP-3, by discovering 1:1 equivalence relations
between the instances that they contain. We have selected several groups of pairs of
works with highly similar descriptions where there exists only one correct match in
each group. The goal is to challenge the linking tools capacity to avoid the generation
of false positives and match correctly instances in the presence of highly similar but
still distinct candidates.</p>
        <p>AML (th=0.2)
AML (th=0.6)
RiMOM</p>
        <p>
          9 heterogeneities
Prec. F-m. Rec.
Results Only two systems returned results on the track: AML and RiMOM. Note that
AML has been configured with two different thresholds. The results of their
performances, evaluated by using precision, recall and F-measure, on each of the three tasks
can be seen in Table 46. The best performance in terms of F-measure is provided by the
AML tool with a threshold of 0:2 on all tasks.
11
In 2013 and in 2015 the community interested in business process modelling conducted
an evaluation campaign similar to OAEI [
          <xref ref-type="bibr" rid="ref3">3</xref>
          ]. Instead of matching ontologies, the task
was to match process models described in different formalisms like BPMN and Petri
Nets. Within this track we offer a subset of the tasks from the Process Model Matching
Contest as OAEI track by converting the process models to an ontological
representation. By offering this track, we hope to gain insights in how far ontology matching
systems are capable of solving the more specific problem of matching process
models. This track is also motivated by the discussions at the end of the 2015 Ontology
Matching workshop, where many participants showed their interest in such a track.
We were using the first data set from the 2015 Process Matching Contest. This data set
deals with processing applications to a university. It consists of nine different process
models where each describes the concrete process of a specific German university. The
models are encoded as BPMN process models. We converted the BPMN
representation of the process models to a set of assertions (ABox) using the vocabulary defined
in the BPMN 2.0 ontology (TBox). For that reason the resulting matching task is an
instance matching task where each ABox is described by the same TBox. For each
pair of processes manually generated reference alignments are available. Typical
activities within that domain are Sending acceptance, Invite student for interview, or Wait
for response. These examples illustrate one of the main differences from the ontology
matching task. The labels are usually verb-object phrases that are sometimes extended
with more words. Another important difference is related to the existence of an
execution order, i.e., the model is a complex sequence of activities, which can be understood
as the counterpart to a type hierarchy.
        </p>
        <p>Only few systems have been marked as capable of generating alignments for the
Process Model Matching track. We have tried to execute all these systems, however,
some of them generated only trivial TBox mappings instead of mappings between
activities. After contacting the developer of the systems, we received the feedback that
the systems have been marked mistakenly and are designed for terminological
matching only. We have excluded them from the evaluation. Moreover, we tried to run all
systems that were marked as instance matching tools, which have been submitted as
executable SEALS bundles. One of these tools (LogMap), generated meaningful results
and was also added to the set of systems that we evaluated. Finally we evaluated three
systems (AML, LogMap, and DKP), one of these systems was configured in two different
settings related to the treatment of events-to-activity mappings. This was the tool DKP.
Thus we distinguish between DKP and DKP*.</p>
        <p>
          In our evaluation, we computed standard precision and recall, as well as the
harmonic mean known as F-measure. The data set we used consists of several test cases.
We aggregated the results and present the micro average results. The gold standard we
used for our first set of evaluation experiments is based on the gold standard that has
also been used at the Process Model Matching Contest in 2015 [
          <xref ref-type="bibr" rid="ref3">3</xref>
          ]. We modified only
some minor mistakes (resulting in changes less than 0.5 percentage points). In order to
compare the results to the results obtained by the process model matching community,
we present also the recomputed values of the submissions to the 2015 contest.
        </p>
        <p>Moreover, we extended our evaluation (“Standard” in Table 47) by a new
evaluation measure that makes use of a probabilistic reference alignment (“Probabilistic” in
Table 47). This probabilistic measure is based on a gold standard which is manually
and independently generated by several domain experts. The number of votes of these
annotators are applied as support values in the probabilistic evaluation. For a detailed
discussion, please refer to [28].
11.2
Table 47 summarises the results of our evaluation. “P” abbreviates precision, “R” is
recall, “FM” stands for F-measure and “Rk” means rank. The prefix “Pro” indicates the
probabilistic versions of the precision, recall, F-measure and the associated rank. These
metrics are explained below. Participants of the Process Model Matching Contest in
2015 (PMMC 2015) are depicted in grey font, while OAEI 2016 participants are shown
in black font. The OAEI participants are ranked on position 1, 8, 9 and 11 with an overall
number of 16 systems listed in the table (when using the standard metrics). Note that
AML-PM at the PMMC 2015 was a matching system that was based on a predecessor
of AML participating at OAEI 2016. The good results of AML are surprising, since we
expected that matching systems specifically developed for the purpose of process model
matching would outperform ontology matching systems applied to the special case of
process model matching. While AML contains also components that are specifically
designed for the process matching task (a flooding-like structural matching algorithm),
its relevant main components are components developed for ontology matching and the
sub-problem of instance matching.</p>
      </sec>
      <sec id="sec-7-4">
        <title>Matcher</title>
      </sec>
      <sec id="sec-7-5">
        <title>Participants</title>
      </sec>
      <sec id="sec-7-6">
        <title>Contest</title>
      </sec>
      <sec id="sec-7-7">
        <title>Size P</title>
      </sec>
      <sec id="sec-7-8">
        <title>Standard</title>
        <p>R FM</p>
      </sec>
      <sec id="sec-7-9">
        <title>Probabilistic</title>
      </sec>
      <sec id="sec-7-10">
        <title>Rk ProP ProR ProFM Rk</title>
        <p>AML OAEI-16 221
AML-PM PMMC-15 579
BPLangMatch PMMC-15 277
DKP OAEI-16 177
DKP* OAEI-16 150
KnoMa-Proc PMMC-15 326
KMatch-SSS PMMC-15 261
LogMap OAEI-16 267
Match-SSS PMMC-15 140
OPBOT PMMC-15 234
pPalm-DS PMMC-15 828
RMM-NHCM PMMC-15 220
RMM-NLM PMMC-15 164
RMM-SMSL PMMC-15 262
RMM-VM2 PMMC-15 505
TripleS PMMC-15 230
0,719 0,685 0,702 1 0,742 0,283
0,269 0,672 0,385 14 0,377 0,398
0,368 0,440 0,401 12 0,532 0,272
0,621 0,474 0,538 8 0,686 0,219
0,680 0,440 0,534 9 0,772 0,211
0,337 0,474 0,394 13 0,506 0,302
0,513 0,578 0,544 6 0,563 0,274
0,449 0,517 0,481 11 0,594 0,291
0,807 0,487 0,608 4 0,761 0,192
0,603 0,608 0,605 5 0,648 0,258
0,162 0,578 0,253 16 0,210 0,335
0,691 0,655 0,673 2 0,783 0,297
0,768 0,543 0,636 3 0,681 0,197
0,511 0,578 0,543 7 0,516 0,242
0,216 0,470 0,296 15 0,309 0,294
0,487 0,483 0,485 10 0,486 0,210
0,410
0,387
0,360
0,333
0,331
0,378
0,368
0,390
0,307
0,369
0,258
0,431
0,306
0,329
0,301
0,293</p>
        <p>In the probabilistic evaluation, however, the OAEI participants gain position 2, 3, 9
and 10, respectively. LogMap rises from position 11 to 3. The (probabilistic) precision
improves over-proportionally for this matcher, because LogMap generates many
correspondences which are not included in the binary gold standard but are included in the
probabilistic one. The ranking of LogMap demonstrates that a strength of the
probabilistic metric lies in the broadened definition of the gold standard where weak mappings
are included but softened (via the support values).</p>
      </sec>
    </sec>
    <sec id="sec-8">
      <title>Lesson learned and suggestions</title>
      <p>The lessons learned from running OAEI 2016 were the following:
A) This year, as suggested in previous campaigns, we requested tool registration in
June and preliminary submission of wrapped systems by the end of July. This
measure was successful in reducing the number of systems with errors and
incompatibilities with the SEALS client during the evaluation phase as had happened in the
past. However, not all systems complied with the deadlines, and some did have
problems, which still delayed the evaluation. In future editions, we must be more
strict in enforcing the participation protocol.</p>
      <p>B) Thanks in part to the new submission schedule, this marked the first OAEI edition
where all participants and all tracks were evaluated using the SEALS client.
Nevertheless, some system developers still struggled to get their systems working with
the client, mostly due to incompatible versions of libraries. This recurring problem,
plus the effort required to update the SEALS client’s libraries, lead to the
consideration of whether it would not be better to develop a simpler, more streamlined
evaluation solution.
(c) Probabilistic F-measure</p>
      <p>C) The continued absence of the SEALS web portal did not seem to affect
participation, as the Google drive solution for submission was well received by the
participants. OAEI may move towards a cloud-based solution.</p>
      <p>D) While the number of participants this year was similar to that of recent years, their
distribution through the tracks was uneven. Long-standing tracks had no shortage
of participants, but alas the same was not true for the Interactive, Process Model
(new) or Instance (new data sets) tracks. One reason for this is that the OAEI data
sets have been released too close to the submission deadline to allow system
developers to develop their systems to tackle them all—the timing is barely sufficient to
allow serious development focusing on one new data set. Thus, with prize money
on offer on one of the new tracks, it is no surprise that system developers were
polarised towards that track and eschewed the other new ones. We should consider
anticipating the deadline for initial release of OAEI data sets, particular for those
that are new, in order to give system developers more time to tackle them, thereby
increasing participation.</p>
      <p>E) The increasing variety of OAEI tracks also poses difficulties to system developers
in configuring their systems to handle different types of tasks. It is noteworthy that
only two systems, both of which are long-term OAEI participants, have tackled all
tracks—and one of them did so using external configuration files specifying the
type of task. One solution to facilitate participation in multiple tracks would be
to have the evaluation client transmit to the system the specifications of the task,
e.g., whether classes, properties, and/or individuals are to be matched, and whether
only a specific subset of them are to be matched. This would also make the tasks
more realistic, in the sense that in normal use, a user would provide to the ontology
matching system this type of information.</p>
      <p>F) With regard to the low participation in the Process Model and Instance tracks, it
merits considering whether enforcing adherence to the SEALS client and
ontologybased data sets were not deterrent factors. It should be noted that the Process Model
Matching Contest (PMMC) received a much larger number of participants in 2015
than did the Process Model track, and that there is a considerable number of
publications on data interlinking systems, but only one of these participated in the
Instance track.</p>
      <p>G) In previous years we identified the need for considering non-binary forms of
evaluation, namely in cases where there is uncertainty about some of the reference
mappings. A first non-binary evaluation type was implemented in last year’s
Conference track, but this year two new tracks followed suit: Disease and Phenotype
where the evaluation was semantic, and Process Model, where it was probabilistic.
These new strategies should provide a fairer evaluation of the systems in complex
test cases.</p>
      <p>The lessons learned in the various OAEI 2016 track were the following:
largebio: While the current reference alignments, with incoherence-causing mappings
flagged as uncertain, make the evaluation fair to all systems, they are only a
compromise solution, not an ideal one. Thus, we should aim for manually repairing and
validating the reference alignments for future editions.
phenotype: The prize offered in this track, thanks to the kind sponsorship of the Pistoia
Alliance Ontologies Mapping project, was positively accepted by the community
and helped attract new participants. However, it also had a polarising effect, with
some systems focusing exclusively in this track. In future editions, we will consider
including a prize across OAEI tracks in order to motivate developers to successfully
participate in more than one track.
interactive: The new functionality of the Oracle allowing systems to submit a set of up
to three conflicting mappings, rather than a mapping at a time, was successfully
exploited by one new participating system. Nevertheless, this track’s participation has
remained low, as most systems participating in OAEI focussed exclusively on fully
automatic matching. We hope to draw more participants to this track in the future
and will continue to expand it so as to better approximate real user interactions.
process model: The results of the new Process Model track have shown that the
participating ontology matching systems are capable of generating very good results for
the specific problem of process model matching. This shows that the basic
components of an ontology matching system can also be successfully applied to other
kind of matching problems.
instance: In order to attract more instance matching systems to participate in value
semantics (val-sem), value structure (val-struct), and value structure semantics
(valstruct-sem) tasks, we need to produce benchmarks that have fewer instances (in the
order of 10000), of the same type (in our benchmark we asked systems to compare
instances of different types). To balance those aspects, we must then produce data
sets with more complex transformations.
13
OAEI 2016 saw the same number (21) of participants as in recent years, with a healthy
mix of new and returning systems. While some new participants were mainly drawn
by the allure of prize money in the new Disease and Phenotype track, the very fact
that there was prize money on offer shows that interest in ontology matching is not
waning, which bodes well for the future of OAEI. All the test cases were performed
on the SEALS client, including those in the instance matching track, which is good
news regarding the interoperability of matching systems. Furthermore, the fact that the
SEALS client can be used for such a variety of tasks is a good sign of its relevance.</p>
      <p>Unlike previous years, this year there was no noticeable improvement with regard
to system run times—for instance, the distribution of run times in Anatomy and Large
Biomedical Ontologies was approximately the same as last year. There was also no
progress with regard to the ability to handle large ontologies and data sets, as the number
of systems able to cope with the Large Biomedical Ontologies data set in full was the
same as last year, and all systems able to cope with the Instance Synthetic data set were
established systems already known for their ability to handle large data sets. Finally,
there was no progress with regard to alignment repair systems, with only a few returning
systems employing them. As a consequence, incoherent alignments are common.</p>
      <p>With regard to F-measure, some returning systems showed substantial
improvements, but overall, the improvements in F-measure were subtle in Anatomy and Large
Biomedical Ontologies, and non-existent in Conference. As has been the trend, most
systems favour precision over recall.</p>
      <p>Most of the participants have provided a description of their systems and their
experience in the evaluation. These OAEI papers, like the present one, have not been peer
reviewed. However, they are full contributions to this evaluation exercise and reflect
the hard work and clever insight people put into the development of participating
systems. Reading the papers of the participants should help people involved in ontology
matching find out what makes these algorithms work and what could be improved.</p>
      <p>The Ontology Alignment Evaluation Initiative will strive to continue to be a
reference to the ontology matching community by improving both the test cases and the
testing methodology to better reflect the actual needs of the community. Evaluating
ontology matching systems remains a challenging but critical topic, which is essential to
enable the progress of this field [38]. More information can be found at:</p>
    </sec>
    <sec id="sec-9">
      <title>Acknowledgements</title>
      <p>We warmly thank the participants of this campaign. We know that they have worked hard to have
their matching tools executable in time and they provided useful reports on their experience. The
best way to learn about the results remains to read the papers that follow.</p>
      <p>We would also like to thank the Pistoia Alliance9 which sponsored the Disease and Phenotype
track and funded the prize for the winners.</p>
      <p>We are very grateful to the Universidad Polite´cnica de Madrid (UPM), especially to
Nandana Mihindukulasooriya and Asuncio´n Go´mez Pe´rez, for moving, setting up and providing the
necessary infrastructure to run the SEALS repositories.</p>
      <p>We are also grateful to Martin Ringwald and Terry Hayamizu for providing the reference
alignment for the anatomy ontologies and thank Elena Beisswanger for her thorough support on
improving the quality of the data set.</p>
      <p>We thank Khiat Abderrahmane for his support in the Arabic data set and Catherine Comparot
for her feedback and support in the MultiFarm test case.</p>
      <p>We also thank for their support the other members of the Ontology Alignment Evaluation
Initiative steering committee: Yannis Kalfoglou (Ricoh laboratories, UK), Miklos Nagy (The Open
University (UK), Natasha Noy (Stanford University, USA), Yuzhong Qu (Southeast University,
CN), York Sure (Leibniz Gemeinschaft, DE), Jie Tang (Tsinghua University, CN), George Vouros
(University of the Aegean, GR).</p>
      <p>Michelle Cheatham has been supported by the National Science Foundation award
ICER1440202 “EarthCube Building Blocks: Collaborative Proposal: GeoLink”.</p>
      <p>Je´roˆme Euzenat, Ernesto Jimenez-Ruiz, Christian Meilicke, Heiner Stuckenschmidt and
Ca´ssia Trojahn dos Santos have been partially supported by the SEALS (IST-2009-238975)
European project in previous years.</p>
      <p>Daniel Faria was supported by the ELIXIR-EXCELERATE project (INFRADEV-3-2015).</p>
      <p>Ernesto Jimenez-Ruiz has also been partially supported by the Seventh Framework Program
(FP7) of the European Commission under Grant Agreement 318338, “Optique”, the EPSRC
projects DBOnto and ED3, the Research Council of Norway project BigMed, and the Centre
for Scalable Data Access (SIRIUS).</p>
      <p>Catia Pesquita was supported by the FCT through the LASIGE Strategic Project
(UID/CEC/00408/2013) and the research grant PTDC/EEI-ESS/4633/2014.</p>
      <p>Ondrˇej Zamazal has been supported by the CSF grant no. 14-14076P.
26. Ernesto Jime´nez-Ruiz, Christian Meilicke, Bernardo Cuenca Grau, and Ian Horrocks.
Evaluating mapping repair systems with large biomedical ontologies. In Proc. 26th Description
Logics Workshop, 2013.
27. Yevgeny Kazakov, Markus Kro¨tzsch, and Frantisek Simancik. Concurrent classification of
EL ontologies. In Proc. 10th International Semantic Web Conference (ISWC), Bonn (DE),
pages 305–320, 2011.
28. Elena Kuss, Henrik Leopold, Han Van der Aa, Heiner Stuckenschmidt, and Hajo A. Reijers.</p>
      <p>Probabilistic evaluation of process model matching techniques. In Lecture notes in
computer science. Conceptual modeling: 35th international conference, ER 2016, Gifu, Japan,
November 14-17, 2016, pages 279–292, 2016.
29. Pasquale Lisena, Manel Achichi, Eva Ferna´ndez, Konstantin Todorov, and Raphae¨l Troncy.</p>
      <p>Exploring linked classical music catalogs with overture. In ISWC PD: International
Semantic Web Conference Posters and Demos, 2016.
30. L. Ma, Y. Yang, Z. Qiu, G, Xie, Y. Pan, and S. Liu. Towards a Complete OWL Ontology</p>
      <p>Benchmark. In ESWC, 2006.
31. Christian Meilicke. Alignment Incoherence in Ontology Matching. PhD thesis, University</p>
      <p>Mannheim, 2011.
32. Christian Meilicke, Rau´l Garc´ıa Castro, Frederico Freitas, Willem Robert van Hage, Elena
Montiel-Ponsoda, Ryan Ribeiro de Azevedo, Heiner Stuckenschmidt, Ondrej Sva´b-Zamazal,
Vojtech Sva´tek, Andrei Tamilin, Ca´ssia Trojahn, and Shenghui Wang. MultiFarm: A
benchmark for multilingual ontology matching. Journal of web semantics, 15(3):62–68, 2012.
33. Boris Motik, Rob Shearer, and Ian Horrocks. Hypertableau reasoning for description logics.</p>
      <p>Journal of Artificial Intelligence Research, 36:165–228, 2009.
34. Heiko Paulheim, Sven Hertling, and Dominique Ritze. Towards evaluating interactive
ontology matching tools. In Proc. 10th Extended Semantic Web Conference (ESWC), Montpellier
(FR), pages 31–45, 2013.
35. Catia Pesquita, Daniel Faria, Emanuel Santos, and Francisco Couto. To repair or not to
repair: reconciling correctness and coherence in ontology reference alignments. In Proc. 8th
ISWC ontology matching workshop (OM), Sydney (AU), pages 13–24, 2013.
36. Tzanina Saveta, Evangelia Daskalaki, Giorgos Flouris, Irini Fundulaki, Melanie Herschel,
and Axel-Cyrille Ngonga Ngomo. Lance: Piercing to the heart of instance matching tools.</p>
      <p>In International Semantic Web Conference, pages 375–391. Springer, 2015.
37. Tzanina Saveta, Evangelia Daskalaki, Giorgos Flouris, Irini Fundulaki, Melanie Herschel,
and Axel-Cyrille Ngonga Ngomo. Pushing the limits of instance matching systems: A
semantics-aware benchmark for linked data. In WWW, Companion Volume, 2015.
38. Pavel Shvaiko and Je´roˆme Euzenat. Ontology matching: state of the art and future challenges.</p>
      <p>IEEE Transactions on Knowledge and Data Engineering, 25(1):158–176, 2013.
39. Alessandro Solimando, Ernesto Jime´nez-Ruiz, and Giovanna Guerrini. Detecting and
correcting conservativity principle violations in ontology-to-ontology mappings. In The
Semantic Web–ISWC 2014, pages 1–16. Springer, 2014.
40. Alessandro Solimando, Ernesto Jimenez-Ruiz, and Giovanna Guerrini. Minimizing
conservativity violations in ontology alignments: Algorithms and evaluation. Knowledge and
Information Systems, 2016.
41. York Sure, Oscar Corcho, Je´roˆme Euzenat, and Todd Hughes, editors. Proc. ISWC Workshop
on Evaluation of Ontology-based Tools (EON), Hiroshima (JP), 2004.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          1.
          <string-name>
            <given-names>Manel</given-names>
            <surname>Achichi</surname>
          </string-name>
          , Rodolphe Bailly, Ce´cile Cecconi, Marie Destandau, Konstantin Todorov, and Raphae¨l Troncy.
          <article-title>Doremus: Doing reusable musical data</article-title>
          .
          <source>In ISWC PD: International Semantic Web Conference Posters and Demos</source>
          ,
          <year>2015</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          2. Jose´ Luis Aguirre, Bernardo Cuenca Grau, Kai Eckert, Je´roˆme Euzenat, Alfio Ferrara, Robert Willem van Hague,
          <string-name>
            <surname>Laura Hollink</surname>
          </string-name>
          , Ernesto Jime´
          <article-title>nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>Christian</given-names>
          </string-name>
          <string-name>
            <surname>Meilicke</surname>
          </string-name>
          , Andriy Nikolov, Dominique Ritze, Franc¸ois Scharffe, Pavel Shvaiko, Ondrej Sva´
          <fpage>b</fpage>
          -Zamazal, Ca´ssia Trojahn, and
          <string-name>
            <given-names>Benjamin</given-names>
            <surname>Zapilko</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2012</article-title>
          .
          <source>In Proc. 7th ISWC ontology matching workshop (OM)</source>
          , Boston (MA US), pages
          <fpage>73</fpage>
          -
          <lpage>115</lpage>
          ,
          <year>2012</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          3.
          <string-name>
            <given-names>Goncalo</given-names>
            <surname>Antunes</surname>
          </string-name>
          , Marzieh Bakhshandeh, Jose Borbinha, Joao Cardoso, Sharam Dadashnia, Chiara Di Francescomarino, Mauro Dragoni,
          <string-name>
            <given-names>Peter</given-names>
            <surname>Fettke</surname>
          </string-name>
          , Avigdor Gal, Chiara Ghidini, Philip Hake, Abderrahmane Khiat, Christopher Klinkmu¨ller, Elena Kuss, Henrik Leopold,
          <string-name>
            <given-names>Peter</given-names>
            <surname>Loos</surname>
          </string-name>
          , Christian Meilicke, Tim Niesen, Catia Pesquita, Timo Pe´us, Andreas Schoknecht, Eitam Sheetrit, Andreas Sonntag, Heiner Stuckenschmidt, Tom Thaler,
          <string-name>
            <given-names>Ingo</given-names>
            <surname>Weber</surname>
          </string-name>
          , and
          <string-name>
            <given-names>Matthias</given-names>
            <surname>Weidlich</surname>
          </string-name>
          .
          <article-title>The process model matching contest 2015</article-title>
          .
          <source>In 6th International Workshop on Enterprise Modelling and Information Systems Architectures, September 3-4</source>
          , 2015 Innsbruck, Austria, pages
          <fpage>127</fpage>
          -
          <lpage>155</lpage>
          ,
          <year>2015</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          4.
          <string-name>
            <given-names>Benjamin</given-names>
            <surname>Ashpole</surname>
          </string-name>
          , Marc Ehrig, Je´roˆme Euzenat, and Heiner Stuckenschmidt, editors.
          <source>Proc. K-Cap Workshop on Integrating Ontologies</source>
          ,
          <source>Banff (Canada)</source>
          ,
          <year>2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          5.
          <string-name>
            <given-names>Olivier</given-names>
            <surname>Bodenreider</surname>
          </string-name>
          .
          <article-title>The unified medical language system (UMLS): integrating biomedical terminology</article-title>
          .
          <source>Nucleic Acids Research</source>
          ,
          <volume>32</volume>
          :
          <fpage>267</fpage>
          -
          <lpage>270</lpage>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          6.
          <string-name>
            <given-names>Caterina</given-names>
            <surname>Caracciolo</surname>
          </string-name>
          , Je´roˆme Euzenat, Laura Hollink, Ryutaro Ichise, Antoine Isaac, Ve´ronique Malaise´,
          <string-name>
            <surname>Christian</surname>
            <given-names>Meilicke</given-names>
          </string-name>
          , Juan Pane, Pavel Shvaiko, Heiner Stuckenschmidt, Ondrej Sva´
          <article-title>b-</article-title>
          <string-name>
            <surname>Zamazal</surname>
          </string-name>
          ,
          <article-title>and Vojtech Sva´tek. Results of the ontology alignment evaluation initiative 2008</article-title>
          .
          <source>In Proc. 3rd ISWC ontology matching workshop (OM)</source>
          ,
          <source>Karlsruhe (DE)</source>
          , pages
          <fpage>73</fpage>
          -
          <lpage>120</lpage>
          ,
          <year>2008</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          7.
          <string-name>
            <given-names>Silvana</given-names>
            <surname>Castano</surname>
          </string-name>
          , Alfio Ferrara, Lorenzo Genta, and
          <string-name>
            <given-names>Stefano</given-names>
            <surname>Montanelli</surname>
          </string-name>
          .
          <article-title>Combining Crowd Consensus and User Trustworthiness for Managing Collective Tasks</article-title>
          .
          <source>Future Generation Computer Systems</source>
          ,
          <volume>54</volume>
          ,
          <year>2016</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          8.
          <string-name>
            <given-names>Michelle</given-names>
            <surname>Cheatham</surname>
          </string-name>
          , Zlatan Dragisic, Je´roˆme Euzenat, Daniel Faria, Alfio Ferrara, Giorgos Flouris, Irini Fundulaki, Roger Granada, Valentina Ivanova, Ernesto Jime´
          <article-title>nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
          </string-name>
          , et al.
          <article-title>Results of the ontology alignment evaluation initiative 2015</article-title>
          .
          <source>In 10th ISWC workshop on ontology matching (OM)</source>
          , pages
          <fpage>60</fpage>
          -
          <lpage>115</lpage>
          ,
          <year>2015</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          9.
          <string-name>
            <given-names>Bernardo</given-names>
            <surname>Cuenca</surname>
          </string-name>
          <string-name>
            <surname>Grau</surname>
          </string-name>
          , Zlatan Dragisic, Kai Eckert, Je´roˆme Euzenat, Alfio Ferrara, Roger Granada, Valentina Ivanova, Ernesto Jime´
          <fpage>nez</fpage>
          -Ruiz, Andreas Oskar Kempf, Patrick Lambrix, Andriy Nikolov, Heiko Paulheim, Dominique Ritze, Franc¸ois Scharffe, Pavel Shvaiko, Ca´ssia Trojahn dos Santos, and
          <string-name>
            <given-names>Ondrej</given-names>
            <surname>Zamazal</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2013</article-title>
          . In Pavel Shvaiko, Je´roˆme Euzenat, Kavitha Srinivas, Ming Mao, and Ernesto Jime´
          <fpage>nez</fpage>
          -Ruiz, editors,
          <source>Proc. 8th ISWC workshop on ontology matching (OM)</source>
          ,
          <source>Sydney (NSW AU)</source>
          , pages
          <fpage>61</fpage>
          -
          <lpage>100</lpage>
          ,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          10.
          <string-name>
            <given-names>Jim</given-names>
            <surname>Dabrowski</surname>
          </string-name>
          and
          <string-name>
            <given-names>Ethan V.</given-names>
            <surname>Munson</surname>
          </string-name>
          .
          <article-title>40 years of searching for the best computer system response time</article-title>
          .
          <source>Interacting with Computers</source>
          ,
          <volume>23</volume>
          (
          <issue>5</issue>
          ):
          <fpage>555</fpage>
          -
          <lpage>564</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          11. Je´roˆme David, Je´roˆme Euzenat, Franc¸ois Scharffe, and Ca´ssia Trojahn dos Santos.
          <source>The alignment API 4.0. Semantic web journal</source>
          ,
          <volume>2</volume>
          (
          <issue>1</issue>
          ):
          <fpage>3</fpage>
          -
          <lpage>10</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          12.
          <string-name>
            <surname>Zlatan</surname>
            <given-names>Dragisic</given-names>
          </string-name>
          , Kai Eckert, Je´roˆme Euzenat, Daniel Faria, Alfio Ferrara, Roger Granada, Valentina Ivanova, Ernesto Jime´
          <fpage>nez</fpage>
          -Ruiz, Andreas Oskar Kempf, Patrick Lambrix, Stefano Montanelli, Heiko Paulheim, Dominique Ritze, Pavel Shvaiko, Alessandro Solimando, Ca´ssia Trojahn dos Santos, Ondrej Zamazal, and Bernardo Cuenca Grau. Results of the
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>