<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>Results of the Ontology Alignment Evaluation Initiative 2013?</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Bernardo Cuenca Grau</string-name>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Zlatan Dragisic</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Kai Eckert</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Je´roˆme Euzenat</string-name>
          <email>Jerome.Euzenat@inria.fr</email>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alfio Ferrara</string-name>
          <email>alfio.ferrara@unimi.it</email>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Roger Granada</string-name>
          <email>roger.granada@acad.pucrs.br</email>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Valentina Ivanova</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Jime´nez-Ruiz</string-name>
          <email>ernestog@cs.ox.ac.uk</email>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andreas Oskar Kempf</string-name>
          <email>andreas.kempf@gesis.org</email>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Lambrix</string-name>
          <email>patrick.lambrixg@liu.se</email>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andriy Nikolov</string-name>
          <email>andriy.nikolov@fluidops.com</email>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Heiko Paulheim</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Dominique Ritze</string-name>
          <email>dominiqueg@informatik.uni-mannheim.de</email>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Franc¸ois Scharffe</string-name>
          <email>francois.scharffe@lirmm.fr</email>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pavel Shvaiko</string-name>
          <email>pavel.shvaiko@infotn.it</email>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ca´ssia Trojahn</string-name>
          <email>cassia.trojahn@irit.fr</email>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ondrˇej Zamazal</string-name>
          <email>ondrej.zamazal@vse.cz</email>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Fluid operations</institution>
          ,
          <addr-line>Walldorf</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>GESIS - Leibniz Institute for the Social Sciences</institution>
          ,
          <addr-line>Cologne</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>INRIA &amp; LIG</institution>
          ,
          <addr-line>Montbonnot</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>IRIT &amp; Universite ́ Toulouse II</institution>
          ,
          <addr-line>Toulouse</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>LIRMM</institution>
          ,
          <addr-line>Montpellier</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Linko ̈ping University &amp; Swedish e-Science Research Center</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>Pontif ́ıcia Universidade Cato ́lica do Rio Grande do Sul</institution>
          ,
          <addr-line>Porto Alegre</addr-line>
          ,
          <country country="BR">Brazil</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>TasLab</institution>
          ,
          <addr-line>Informatica Trentina, Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>Universita` degli studi di Milano</institution>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>University of Economics</institution>
          ,
          <addr-line>Prague</addr-line>
          ,
          <country country="CZ">Czech Republic</country>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>University of Mannheim</institution>
          ,
          <addr-line>Mannheim</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>University of Oxford</institution>
          ,
          <country country="UK">UK</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2013</year>
      </pub-date>
      <abstract>
        <p>Ontology matching consists of finding correspondences between semantically related entities of two ontologies. OAEI campaigns aim at comparing ontology matching systems on precisely defined test cases. These test cases can use ontologies of different nature (from simple thesauri to expressive OWL ontologies) and use different modalities, e.g., blind evaluation, open evaluation and consensus. OAEI 2013 offered 6 tracks with 8 test cases followed by 23 participants. Since 2010, the campaign has been using a new evaluation modality which provides more automation to the evaluation. This paper is an overall presentation of the OAEI 2013 campaign.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>Introduction</title>
      <p>The Ontology Alignment Evaluation Initiative1 (OAEI) is a coordinated international
initiative, which organizes the evaluation of the increasing number of ontology
matching systems [11, 14]. The main goal of OAEI is to compare systems and algorithms on
the same basis and to allow anyone for drawing conclusions about the best matching
strategies. Our ambition is that, from such evaluations, tool developers can improve
their systems.</p>
      <p>
        Two first events were organized in 2004: (i) the Information Interpretation and
Integration Conference (I3CON) held at the NIST Performance Metrics for Intelligent
Systems (PerMIS) workshop and (ii) the Ontology Alignment Contest held at the
Evaluation of Ontology-based Tools (EON) workshop of the annual International Semantic
Web Conference (ISWC) [
        <xref ref-type="bibr" rid="ref13">28</xref>
        ]. Then, a unique OAEI campaign occurred in 2005 at the
workshop on Integrating Ontologies held in conjunction with the International
Conference on Knowledge Capture (K-Cap) [3]. Starting from 2006 through 2012 the OAEI
campaigns were held at the Ontology Matching workshops collocated with ISWC [12,
10, 5, 7–9, 1]. In 2013, the OAEI results were presented again at the Ontology Matching
workshop2 collocated with ISWC, in Sydney, Australia.
      </p>
      <p>Since 2011, we have been promoting an environment for automatically processing
evaluations (x2.2), which has been developed within the SEALS (Semantic Evaluation
At Large Scale) project3. SEALS provided a software infrastructure, for automatically
executing evaluations, and evaluation campaigns for typical semantic web tools,
including ontology matching. For OAEI 2013, almost all of the OAEI data sets were evaluated
under the SEALS modality, providing a more uniform evaluation setting.</p>
      <p>This paper synthetizes the 2013 evaluation campaign and introduces the results
provided in the papers of the participants. The remainder of the paper is organised as
follows. In Section 2, we present the overall evaluation methodology that has been used.
Sections 3-10 discuss the settings and the results of each of the test cases. Section 11
overviews lessons learned from the campaign. Finally, Section 12 concludes the paper.
2</p>
    </sec>
    <sec id="sec-2">
      <title>General methodology</title>
      <p>We first present the test cases proposed this year to the OAEI participants (x2.1). Then,
we discuss the resources used by participants to test their systems and the execution
environment used for running the tools (x2.2). Next, we describe the steps of the OAEI
campaign (x2.3-2.5) and report on the general execution of the campaign (x2.6).
2.1</p>
      <sec id="sec-2-1">
        <title>Tracks and test cases</title>
        <p>This year’s campaign consisted of 6 tracks gathering 8 test cases and different
evaluation modalities:
1 http://oaei.ontologymatching.org
2 http://om2013.ontologymatching.org
3 http://www.seals-project.eu
The benchmark track (x3): Like in previous campaigns, a systematic benchmark
series has been proposed. The goal of this benchmark series is to identify the areas
in which each matching algorithm is strong or weak by systematically altering an
ontology. This year, we generated a new benchmark based on the original
bibliographic ontology.</p>
        <p>The expressive ontology track offers real world ontologies using OWL modelling
capabilities:
Anatomy (x4): The anatomy real world test case is about matching the Adult
Mouse Anatomy (2744 classes) and a small fragment of the NCI Thesaurus
(3304 classes) describing the human anatomy.</p>
        <p>Conference (x5): The goal of the conference test case is to find all correct
correspondences within a collection of ontologies describing the domain of
organizing conferences. Results were evaluated automatically against reference
alignments and by using logical reasoning techniques.</p>
        <p>Large biomedical ontologies (x6): The Largebio test case aims at finding
alignments between large and semantically rich biomedical ontologies such as
FMA, SNOMED-CT, and NCI. The UMLS Metathesaurus has been used as
the basis for reference alignments.</p>
      </sec>
      <sec id="sec-2-2">
        <title>Multilingual</title>
        <p>Multifarm(x7): This test case is based on a subset of the Conference data set,
translated into eight different languages (Chinese, Czech, Dutch, French,
German, Portuguese, Russian, and Spanish) and the corresponding alignments
between these ontologies. Results are evaluated against these alignments.</p>
      </sec>
      <sec id="sec-2-3">
        <title>Directories and thesauri</title>
        <p>Library(x8): The library test case is a real-word task to match two thesauri. The
goal of this test case is to find whether the matchers can handle such lightweight
ontologies including a huge amount of concepts and additional descriptions.
Results are evaluated both against a reference alignment and through manual
scrutiny.</p>
      </sec>
      <sec id="sec-2-4">
        <title>Interactive matching</title>
        <p>Interactive(x9): This test case offers the possibility to compare different
interactive matching tools which require user interaction. Its goal is to show if user
interaction can improve matching results, which methods are most promising
and how many interactions are necessary. All participating systems are
evaluated on the conference data set using an oracle based on the reference
alignment.</p>
        <p>
          Instance matching (x10): The goal of the instance matching track is to evaluate the
performance of different tools on the task of matching RDF individuals which
originate from different sources but describe the same real-world entity. Both the
training data set and the evaluation data set were generated by exploiting the same
configuration of the RDFT transformation tool. It performs controlled alterations
of an initial data source generating data sets and reference links (i.e. alignments).
Reference links were provided for the training set but not for the evaluation set, so
the evaluation is blind.
In 2010, participants of the Benchmark, Anatomy and Conference test cases were asked
for the first time to use the SEALS evaluation services: they had to wrap their tools as
web services and the tools were executed on the machines of the developers [
          <xref ref-type="bibr" rid="ref14">29</xref>
          ]. Since
2011, tool developers had to implement a simple interface and to wrap their tools in a
predefined way including all required libraries and resources. A tutorial for tool
wrapping was provided to the participants. It describes how to wrap a tool and how to use
a simple client to run a full evaluation locally. After local tests are passed successfully,
the wrapped tool had to be uploaded on the SEALS portal4. Consequently, the
evaluation was executed by the organizers with the help of the SEALS infrastructure. This
approach allowed to measure runtime and ensured the reproducibility of the results. As
a side effect, this approach also ensures that a tool is executed with the same settings
for all of the test cases that were executed in the SEALS mode.
2.3
        </p>
      </sec>
      <sec id="sec-2-5">
        <title>Preparatory phase</title>
        <p>Ontologies to be matched and (where applicable) reference alignments have been
provided in advance during the period between June 15th and July 3rd, 2013. This gave
potential participants the occasion to send observations, bug corrections, remarks and
other test cases to the organizers. The goal of this preparatory period is to ensure that
the delivered tests make sense to the participants. The final test base was released on
July 3rd, 2013. The data sets did not evolve after that.
2.4</p>
      </sec>
      <sec id="sec-2-6">
        <title>Execution phase</title>
        <p>During the execution phase, participants used their systems to automatically match the
test case ontologies. In most cases, ontologies are described in OWL-DL and serialized
in the RDF/XML format [6]. Participants can self-evaluate their results either by
comparing their output with reference alignments or by using the SEALS client to compute
4 http://www.seals-project.eu/join-the-community/
precision and recall. They can tune their systems with respect to the non blind
evaluation as long as the rules published on the OAEI web site are satisfied. This phase has
been conducted between July 3rd and September 1st, 2013.
2.5</p>
      </sec>
      <sec id="sec-2-7">
        <title>Evaluation phase</title>
        <p>Participants have been encouraged to provide (preliminary) results or to upload their
wrapped tools on the SEALS portal by September 1st, 2013. For the SEALS modality,
a full-fledged test including all submitted tools has been conducted by the organizers
and minor problems were reported to some tool developers, who had the occasion to fix
their tools and resubmit them.</p>
        <p>First results were available by September 23rd, 2013. The organizers provided these
results individually to the participants. The results were published on the respective
web pages by the organizers by October 1st. The standard evaluation measures are
usually precision and recall computed against the reference alignments. More details
on evaluation measures are given in each test case section.
2.6</p>
      </sec>
      <sec id="sec-2-8">
        <title>Comments on the execution</title>
        <p>The number of participating systems has regularly increased over the years: 4
participants in 2004, 7 in 2005, 10 in 2006, 17 in 2007, 13 in 2008, 16 in 2009, 15 in 2010,
18 in 2011, 21 in 2012, 23 in 2013. However, participating systems are now constantly
changing. In 2013, 11 (7 in 2012) systems have not participated in any of the previous
campaigns. The list of participants is summarized in Table 2. Note that some systems
were also evaluated with different versions and configurations as requested by
developers (see test case sections for details).
3</p>
        <sec id="sec-2-8-1">
          <title>System</title>
          <p>Confidence p p p
benchmarks p p p p p p</p>
          <p>anatomy p p p p p
conference p p p p p p
multifarm p p p p p</p>
          <p>library p p p p
interactive p p
large bio p p p p
im-rdft
Table 2. Participants and the state of their submissions. Confidence stands for the type of results
returned by a system: it is ticked when the confidence is a non boolean value.</p>
          <p>Only four systems participated in the instance matching track, where two of them
(LogMap and RiMOM2013) also participated in the SEALS tracks. The interactive
track also had the same participation, since there are not yet many tools supporting user
intervention within the matching process. Finally, some systems were not able to pass
some test cases as indicated in Table 2. SPHeRe is an exception since it only participated
in the largebio test case. It is a special system based on cloud computing which did not
use the SEALS interface this year.</p>
          <p>The result summary per test case is presented in the following sections.
3</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-3">
      <title>Benchmark</title>
      <p>The goal of the benchmark data set is to provide a stable and detailed picture of each
algorithm. For that purpose, algorithms are run on systematically generated test cases.
3.1</p>
      <sec id="sec-3-1">
        <title>Test data</title>
        <p>The systematic benchmark test set is built around a seed ontology and many variations
of it. Variations are artificially generated, and focus on the characterization of the
behavior of the tools rather than having them compete on real-life problems.</p>
        <p>Since OAEI 2011.5, they are obtained by discarding and modifying features from a
seed ontology. Considered features are names of entities, comments, the specialization
hierarchy, instances, properties and classes. Full description of the systematic
benchmark test set can be found on the OAEI web site.</p>
        <p>This year, we used a version of the benchmark test suite generated by the test
generator described in [13] from the usual bibliography ontology. The biblio seed
ontology concerns bibliographic references and is inspired freely from BibTeX. It contains
33 named classes, 24 object properties, 40 data properties, 56 named individuals and
20 anonymous individuals. The test case was not available to participants: participants
could test their systems with respect to last year data sets, but they have been evaluated
against a newly generated test. The tests were also blind for the organizers since we did
not looked into them before running the systems.</p>
        <p>We also generated and run another test suite from a different seed ontology, but
we decided to cancel the evaluation because, due to the particular nature of the seed
ontology, the generator was not able to properly discard important information. We did
not run scalability tests this year.</p>
        <p>The reference alignments are still restricted to named classes and properties and use
the “=” relation with confidence of 1.
3.2</p>
      </sec>
      <sec id="sec-3-2">
        <title>Results</title>
        <p>We run the experiments on a Debian Linux virtual machine configured with four
processors and 8GB of RAM running under a Dell PowerEdge T610 with 2*Intel Xeon
Quad Core 2.26GHz E5607 processors and 32GB of RAM, under Linux ProxMox 2
(Debian). All matchers where run under the SEALS client using Java 1.7 and a
maximum heap size of 6GB. No timeout was explicitly set.</p>
        <p>Reported figures are the average of 5 runs. As has already been shown in [13], there
is not much variance in compliance measures across runs. This is not necessarily the
case for time measurements so we report standard deviations with time measurements.</p>
        <p>From the 23 systems listed in Table 2, 20 systems participated in this test case.
Three systems were only participating in the instance matching or largebio test cases.
XMap had two different system versions.</p>
        <p>A few of these systems encountered problems (marked * in the results table):
LogMap and OntoK had quite random problems and did not return results for some
tests sometimes; ServOMap did not return results for tests past #261-4; MaasMatch did
not return results for tests past #254; MapSSS and StringsAuto did not return results
for tests past #202 or #247. Besides the two last systems, the problems where persistent
across all 5 runs. MapSSS and StringsAuto alternated between the two failure patterns.
We suspect that some of the random problems are due to internal or network timeouts.
Compliance Concerning F-measure results, YAM++ (.89) and CroMatcher (.88) are
far ahead before Cider-CL (.75), IAMA (.73) and ODGOMS (.71). Without surprise,
such systems have all the same profile: their precision is higher than their recall.</p>
        <p>With respect to 2012, some systems maintained their performances or slightly
improved them (YAM++, MaasMatch, Hertuda, HotMatch, WikiMatch) while other
showed severe degradations. Some of these are explained by failures (MapSSS,
ServOMap, LogMap) some others are not explained (LogMapLite, WeSeE). Matchers with
lower performance than the baseline are those mentioned before as encountering
problems when running tests. This is a problem that such matchers are not robust to these
classical tests. It is noteworthy, and surprising, that most of the systems which did not
complete all the tests were systems which completed them in 2012!
Confidence accuracy Confidence-weighted measures reward systems able to provide
accurate confidence values. Using confidence-weighted F-measures does not increase
the evaluation of systems (beside edna which does not perform any filtering). In
principle, the weighted recall cannot be higher, but the weighted precision can. In fact, only
edna, OntoK and XMapSig have an increased precision. The order given above does
not change much with the weighted measures: IAMA and ODGOMS pass CroMatcher
and Cider-CL. The only system to suffer a dramatic decrease is RiMOM, owing to the
very low confidence measures that it provides.</p>
        <p>For those systems which have provided their results with confidence measures
different from 1 or 0, it is possible to draw precision/recall graphs in order to compare
them; these graphs are given in Figure 1. The graphs show the real precision at n%
recall and they stop when no more correspondences are available; then the end point
corresponds to the precision and recall reported in the Table 3.</p>
        <p>The precision-recall curves confirm the good performances of YAM++ and
CroMatcher. CroMatcher achieves the same level of recall as YAM++ but with consistently
lower precision. The curves show the large variability across systems. This year,
systems seem to be less focussed on precision and make progress at the expense of
precision. However, this may be an artifact due to systems facing problems.</p>
        <p>refalign</p>
        <p>1,00
CIDER-CL</p>
        <p>0,65
HotMatch</p>
        <p>0,50
LogMapLite</p>
        <p>0,50
ODGOMS</p>
        <p>0,55
ServOMap</p>
        <p>0,22
WeSeE</p>
        <p>0,39
XMapSig
0,49
recall
edna
0,50
CroMatcher
0,80
IAMA
0,57
MaasMatch
0,52
OntoK
0,40
StringsAuto</p>
        <p>0,02
WikiMatch</p>
        <p>0,53
YAM++
0,82</p>
        <p>AML
0,40
HerTUDA</p>
        <p>0,54
LogMap</p>
        <p>0,36
MapSSS</p>
        <p>0,02
RiMOM2013</p>
        <p>0,44
Synthesis</p>
        <p>0,60
XMapGen
0,40</p>
        <p>Runtime There is a large discrepancy between matchers concerning the time spent
to match one test run, i.e., 94 matching tests. It ranges from less than a minute for
LogMapLite and AML (we do not count StringsAuto which failed to perform many
tests) to nearly three hours for OntoK. In fact, OntoK takes as much time as all the other
matchers together. Beside these large differences, we also observed large deviations
across runs.</p>
        <p>We provide (Table 3) the average F-measure point provided per second by matchers.
This makes a different ordering of matchers: AML (1.04) comes first before Hertuda
(0.94) and LogMapLite (0.81). None of the matchers with the best performances come
first. This means that, for achieving good results, considerable time should be spent
(however, YAM++ still performs the 94 matching operations in less than 12 minutes).
Regarding compliance, we observed that, with very few exceptions, the systems
performed always better than the baseline. Most of the systems are focussing on precision.
This year there was a significant number of systems unable to pass the tests correctly.
On the one hand, this is good news: this means that systems are focussing on other test
cases than benchmarks. On the other hand, it exhibits system brittleness.</p>
        <p>Except for a very few exception, system run time performance is acceptable on tests
of that size, but we did not perform scalability tests like last year.</p>
        <p>Matching system
edna</p>
        <p>AML
CIDER-CL</p>
        <sec id="sec-3-2-1">
          <title>CroMatcher</title>
        </sec>
        <sec id="sec-3-2-2">
          <title>Hertuda Hotmatch IAMA LogMap</title>
        </sec>
        <sec id="sec-3-2-3">
          <title>LogMapLt MaasMtch (*)</title>
        </sec>
        <sec id="sec-3-2-4">
          <title>MapSSS (*) ODGOMS</title>
        </sec>
        <sec id="sec-3-2-5">
          <title>OntoK</title>
        </sec>
        <sec id="sec-3-2-6">
          <title>RiMOM2013</title>
        </sec>
        <sec id="sec-3-2-7">
          <title>ServOMap (*) StringsAuto (*) Synthesis WeSeE</title>
          <p>Wikimatch
XMapGen</p>
        </sec>
        <sec id="sec-3-2-8">
          <title>XMapSig</title>
          <p>YAM++
0.50 0.35 0.41 0.50
(0.58) (0.54)
1.00 0.57 0.40
0.85 0.75 0.67
(0.84) (0.66) (0.55)
0.95 0.88 0.82
(0.75) (0.68) (0.63)
0.53 0.90 0.68 0.54
0.52 0.96 0.68 0.50
0.99 0.73 0.57
0.47 0.72 0.53 0.42
(0.42) (0.51) (0.39)
0.50 0.43 0.46 0.50
0.6 0.84 0.69 0.59
(0.5) (0.66) (0.50) (0.41)
0.75 0.84 0.14 0.08
0.99 0.71 0.55
(0.98) (0.70) (0.54)
0.63 0.51 0.43
(0.69) (0.40)
0.59 0.58 0.58
(0.49) (0.19) (0.12)
0.5 0.53 0.33 0.22
0.84 0.14 0.08
0.60 0.60 0.60
0.96 0.55 0.39
0.99 0.69 0.53
0.66 0.54 0.46</p>
          <p>(0.52) (0.44)
0.70 0.58 0.50
(0.71) (0.59)
0.82 0.97 0.89 0.82
(0.56) (0.84) (0.77) (0.70)
55
844
1114
72
103
102
123
57
173
81
100
105
409
56
659
4933
1845
594
612
702
6
19
21
6
6
10
7
7
6
44
6
34
33
38
11
40
39
5
11
46</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>Anatomy</title>
      <p>The anatomy test case confronts matchers with a specific type of ontologies from the
biomedical domain. We focus on two fragments of biomedical ontologies which
describe the human anatomy5 and the anatomy of the mouse6. This data set has been used
since 2007 with some improvements over the years.
We conducted experiments by executing each system in its standard setting and we
compare precision, recall, F-measure and recall+. The measure recall+ indicates the
amount of detected non-trivial correspondences. The matched entities in a non-trivial
correspondence do not have the same normalized label. The approach that generates
only trivial correspondences is depicted as baseline StringEquiv in the following
section.</p>
      <p>This year we run the systems on a server with 3.46 GHz (6 cores) and 8GB RAM
allocated to the matching systems. This is a different setting compared to previous years,
so, runtime results are not fully comparable across years. The evaluation was performed
with the SEALS client. However, we slightly changed the way how precision and recall
are computed, i.e., the results generated by the SEALS client vary in some cases by
0.5% compared to the results presented below. In particular, we removed trivial
correspondences in the oboInOwl namespace like</p>
      <p>http://...oboInOwl#Synonym = http://...oboInOwl#Synonym
as well as correspondences expressing relations different from equivalence. Using the
Pellet reasoner we also checked whether the generated alignment is coherent, i.e., there
are no unsatisfiable concepts when the ontologies are merged with the alignment.
4.2</p>
      <sec id="sec-4-1">
        <title>Results</title>
        <p>In Table 4, we analyze all participating systems that could generate an alignment in less
than ten hours. The listing comprises of 20 entries sorted by F-measure. Four systems
participated each with two different versions. These are AML and GOMMA with
versions which use background knowledge (indicated with suffix “-bk”), LogMap with a
lightweight version LogMapLite that uses only some core components and XMap with
versions XMapSig and XMapGen which use two different parameters. For comparison
purposes, we run again last year version of GOMMA. GOMMA and HerTUDA
participated with the same system as last year (indicated by * in the table). In addition to
these two tools we have eight more systems which participated in 2012 and now
participated with new versions (HotMatch, LogMap, MaasMatch, MapSSS, ServOMap,
WeSeE, WikiMatch and YAM++). Due to some software and hardware incompatibilities,
5 http://www.cancer.gov/cancertopics/cancerlibrary/</p>
        <p>terminologyresources/
6 http://www.informatics.jax.org/searches/AMA_form.shtml
YAM++ had to be run on a different machine and therefore its runtime (indicated by **)
is not fully comparable to that of other systems. Thus, 20 different systems generated an
alignment within the given time frame. Four participants (CroMatcher, RiMOM2013,
OntoK and Synthesis) did not finish in time or threw an exception.</p>
        <sec id="sec-4-1-1">
          <title>Precision</title>
        </sec>
        <sec id="sec-4-1-2">
          <title>F-measure</title>
        </sec>
        <sec id="sec-4-1-3">
          <title>Recall Recall+ Matcher AML-bk GOMMA-bk*</title>
          <p>YAM++
AML
LogMap
GOMMA*
StringsAuto
LogMapLite
MapSSS
ODGOMS
WikiMatch
HotMatch
StringEquiv
XMapSig
ServOMap
XMapGen
IAMA
CIDER-CL
HerTUDA*
WeSeE
MaasMatch</p>
          <p>Runtime</p>
          <p>Nine systems finished in less than 100 seconds, compared to 8 systems in OAEI
2012 and 2 systems in OAEI 2011. This year, 20 out of 24 systems generated results
compared to last year when 14 out of 18 systems generated results within the given
time frame. The top systems in terms of runtimes are LogMap, GOMMA, IAMA and
AML. Depending on the specific version of the systems, they require between 7 and 15
seconds to match the ontologies. The table shows that there is no correlation between
quality of the generated alignment in terms of precision and recall and required runtime.
This result has also been observed in previous OAEI campaigns.</p>
          <p>Table 4 also shows the results for precision, recall and F-measure. In terms of
Fmeasure, the two top ranked systems are AML-bk and GOMMA-bk. These systems
use specialised background knowledge, i.e., they are based on mapping composition
techniques and the reuse of mappings between UMLS, Uberon and FMA. AML-bk and
GOMMA-bk are followed by a group of matching systems (YAM++, AML, LogMap,
GOMMA) generating alignments that are very similar with respect to precision, recall
and F-measure (between 0.87 and 0.91 F-measure). LogMap uses the general
(biomedical) purpose UMLS Lexicon, while the other systems either use Wordnet or no
background knowledge. The results of these systems are at least as good as the results of the
best system in OAEI 2007-2010. Only AgreementMaker using additional background
knowledge could generate better results than these systems in 2011.</p>
          <p>This year, 8 out of 20 systems achieved an F-measure that is lower than the baseline
which is based on (normalized) string equivalence (StringEquiv in the table).</p>
          <p>Moreover, nearly all systems find many non-trivial correspondences. An exception
are IAMA and WeSeE which generated an alignment that is quite similar to the
alignment generated by the baseline approach.</p>
          <p>From the systems which participated last year WikiMatch showed a considerable
improvement. It increased precision from 0.86 to 0.99 and F-measure from 0.76 to
0.80. The other systems produced very similar results compared to the previous year.
One exception is WeSeE which achieved a much lower F-measure than in 2012.</p>
          <p>Three systems have produced an alignment which is coherent. Last year two systems
produced such alignments.
This year 24 systems (or system variants) participated in the anatomy test case out
of which 20 produced results within 10 hours. This is so far the highest number of
participating systems as well as the highest number of systems which produce results
given time constraints for the anatomy test case.</p>
          <p>As last year, we have witnessed a positive trend in runtimes as the majority of
systems finish execution in less than one hour (16 out of 20). The AML-bk system improves
the best result in terms of F-measure set by a previous version of the system in 2010
and makes it also the top result for the anatomy test case.
5</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>Conference</title>
      <p>5.1</p>
      <sec id="sec-5-1">
        <title>Test data</title>
        <p>The conference test case introduces matching several moderately expressive ontologies.
Within this test case, participant results were evaluated against reference alignments
(containing merely equivalence correspondences) and by using logical reasoning. The
evaluation has been performed with the SEALS infrastructure.</p>
        <p>The data set consists of 16 ontologies in the domain of organizing conferences. These
ontologies have been developed within the OntoFarm project7.</p>
        <p>The main features of this test case are:
– Generally understandable domain. Most ontology engineers are familiar with
organizing conferences. Therefore, they can create their own ontologies as well as
evaluate the alignments among their concepts with enough erudition.
7 http://nb.vse.cz/˜svatek/ontofarm.html
– Independence of ontologies. Ontologies were developed independently and based
on different resources, they thus capture the issues in organizing conferences from
different points of view and with different terminologies.
– Relative richness in axioms. Most ontologies were equipped with OWL DL axioms
of various kinds; this opens a way to use semantic matchers.</p>
        <p>Ontologies differ in their numbers of classes, of properties, in expressivity, but also
in underlying resources.
5.2</p>
      </sec>
      <sec id="sec-5-2">
        <title>Results</title>
        <p>We provide results in terms of F0:5-measure, F1-measure and F2-measure, comparison
with baseline matchers, precision/recall triangular graph and coherency evaluation.</p>
      </sec>
      <sec id="sec-5-3">
        <title>Evaluation based on reference alignments We evaluated the results of participants</title>
        <p>against blind reference alignments (labelled as ra2 on the conference web-page). This
includes all pairwise combinations between 7 different ontologies, i.e. 21 alignments.</p>
        <p>These reference alignments have been generated as a transitive closure computed
on the original reference alignments. In order to obtain a coherent result, conflicting
correspondences, i.e., those causing unsatisfiability, have been manually inspected and
removed by evaluators. As a result, the degree of correctness and completeness of the
new reference alignment is probably slightly better than for the old one. However, the
differences are relatively limited. Whereas the new reference alignments are not open,
the old reference alignments (labeled as ra1 on the conference web-page) are available.
These represent close approximations of the new ones.</p>
        <p>Table 5 shows the results of all participants with regard to the new reference
alignment. F0:5-measure, F1-measure and F2-measure are computed for the threshold that
provides the highest average F1-measure. F1 is the harmonic mean of precision and
recall where both are equally weighted; F2 weights recall higher than precision and
F0:5 weights precision higher than recall. The matchers shown in the table are
ordered according to their highest average F1-measure. This year we employed two
baselines matcher. edna (string edit distance matcher) is used within the benchmark
test case and with regard to performance it is very similar as previously used
baseline2; StringEquiv is used within the anatomy test case. These baselines divide
matchers into three groups. Group 1 consists of matchers (YAM++, AML-bk –AML
standing for AgreementMakerLight–, LogMap, AML, ODGOMS, StringsAuto, ServOMap,
MapSSS, HerTUDA, WikiMatch, WeSeE-Match, IAMA, HotMatch, CIDER-CL)
having better (or the same) results than both baselines in terms of highest average
F1measure. Group 2 consists of matchers (OntoK, LogMapLite, XMapSigG, XMapGen
and SYNTHESIS) performing better than baseline StringEquiv but worse than edna.
Other matchers (RIMOM2013, CroMatcher and MaasMatch) performed worse than
both baselines. CroMatcher was unable to process any ontology pair where
conference.owl ontology was included. Therefore, the evaluation was run only on 15 test
cases. Thus, its results are just an approximation.</p>
        <p>Performance of matchers from Group 1 regarding F1-measure is visualized in
Figure 2.
YAM++
AML-bk
LogMap</p>
        <p>AML
ODGOMS1 2</p>
        <p>StringsAuto
ServOMap v104</p>
        <p>MapSSS
ODGOMS1 1</p>
        <p>HerTUDA</p>
        <p>WikiMatch
WeSeE-Match</p>
        <p>IAMA
HotMatch
CIDER-CL
edna</p>
        <p>OntoK
LogMapLite
XMapSiG1 3
XMapGen1 4
SYNTHESIS</p>
        <p>StringEquiv
RIMOM2013*
XMapSiG1 4
CroMatcher
XMapGen
MaasMatch
0.65
0.53
0.54
0.51
0.55
0.50
0.50
0.46
0.47
0.46
0.45
0.42
0.44
0.47
0.44
0.44
0.43
0.45
0.44
0.45
0.41
0.39
0.47
0.37
0.43
0.36
0.53</p>
        <p>Size
7
7
0
0
9
20
1
1
AML-bk
LogMap
AML</p>
        <sec id="sec-5-3-1">
          <title>ODGOMS</title>
        </sec>
        <sec id="sec-5-3-2">
          <title>StringsAuto</title>
        </sec>
        <sec id="sec-5-3-3">
          <title>ServOMap</title>
        </sec>
        <sec id="sec-5-3-4">
          <title>MapSSS</title>
        </sec>
        <sec id="sec-5-3-5">
          <title>Hertuda</title>
        </sec>
        <sec id="sec-5-3-6">
          <title>WikiMatch</title>
        </sec>
        <sec id="sec-5-3-7">
          <title>WeSeE-Match</title>
        </sec>
        <sec id="sec-5-3-8">
          <title>IAMA</title>
        </sec>
        <sec id="sec-5-3-9">
          <title>HotMatch</title>
          <p>CIDER-CL
edna
rec=1.0
rec=.8
rec=.6
pre=.6
pre=.8
pre=1.0
Fig. 2. Precision/recall triangular graph for the conference test case. Dotted lines depict level
of precision/recall while values of F1-measure are depicted by areas bordered by corresponding
lines F1-measure=0.[5j6j7].</p>
          <p>Comparison with previous years Ten matchers also participated in this test case in
OAEI 2012. The largest improvement was achieved by MapSSS (precision from .47 to
.77, whereas recall remains the same, .46) and ServOMap (precision from .68 to .69
and recall from .41 to .50).</p>
          <p>Runtimes We measured the total time of generating 21 alignments. It was executed on a
laptop under Ubuntu running on Intel Core i5, 2.67GHz and 8GB RAM. In all, there are
eleven matchers which finished all 21 tests within 1 minute or around 1 minute
(AMLbk: 16s, ODGOMS: 19s, LogMapLite: 21s, AML, HerTUDA, StringsAuto, HotMatch,
LogMap, IAMA, RIMOM2013: 53s and MaasMatch: 76s). Next, four systems needed
less than 10 minutes (ServOMap, MapSSS, SYNTHESIS, CIDER-CL). 10 minutes are
enough for the next three matchers (YAM++, XMapGen, XMapSiG). Finally, three
matchers needed up to 40 minutes to finish all 21 test cases (WeSeE-Match: 19 min,
WikiMatch: 26 min, OntoK: 40 min).</p>
          <p>In conclusion, regarding performance we can see (clearly from Figure 2) that
YAM++ is on the top again. The next four matchers (AML-bk, LogMap, AML,
ODGOMS) are relatively close to each other. This year there is a larger group of
matchers (15) which are above the edna baseline than previous years. This is partly because a
couple of previous system matchers improved and a couple of high quality new system
matchers entered the OAEI campaign.</p>
        </sec>
      </sec>
      <sec id="sec-5-4">
        <title>Evaluation based on alignment coherence As in the previous years, we apply the</title>
        <p>Maximum Cardinality measure to evaluate the degree of alignment incoherence, see</p>
      </sec>
    </sec>
    <sec id="sec-6">
      <title>Large biomedical ontologies (largebio)</title>
      <p>The Largebio test case aims at finding alignments between the large and semantically
rich biomedical ontologies FMA, SNOMED-CT, and NCI, which contains 78,989,
306,591 and 66,724 classes, respectively.
The test case has been split into three matching problems: FMA-NCI, FMA-SNOMED
and SNOMED-NCI; and each matching problem in 2 tasks involving different
fragments of the input ontologies.</p>
      <p>
        The UMLS Metathesaurus [4] has been selected as the basis for reference
alignments. UMLS is currently the most comprehensive effort for integrating
independentlydeveloped medical thesauri and ontologies, including FMA, SNOMED-CT, and NCI.
Although the standard UMLS distribution does not directly provide “alignments” (in the
OAEI sense) between the integrated ontologies, it is relatively straightforward to extract
them from the information provided in the distribution files (see [
        <xref ref-type="bibr" rid="ref3">18</xref>
        ] for details).
      </p>
      <p>
        It has been noticed, however, that although the creation of UMLS alignments
combines expert assessment and auditing protocols they lead to a significant number of
logical inconsistencies when integrated with the corresponding source ontologies [
        <xref ref-type="bibr" rid="ref3">18</xref>
        ].
      </p>
      <p>
        To address this problem, in OAEI 2013, unlike previous editions, we have created
a unique refinement of the UMLS mappings combining both Alcomo (mapping)
debugging system [
        <xref ref-type="bibr" rid="ref6">21</xref>
        ] and LogMap’s (mapping) repair facility [
        <xref ref-type="bibr" rid="ref2">17</xref>
        ], and manual curation
when necessary. This refinement of the UMLS mappings, which does not lead to
unsatisfiable classes8, has been used as the Large BioMed reference alignment. Objections
8 For the SNOMED-NCI case we used the OWL 2 EL reasoner ELK, see Section 6.4 for details.
have been raised on the validity (and fairness) of the application of mapping repair
techniques to make reference alignments coherent [
        <xref ref-type="bibr" rid="ref9">24</xref>
        ]. For next year campaign, we intend
to take into consideration their suggestions to mitigate the effect of using repair
techniques. This year reference alignment already aimed at mitigating the fairness effect
by combining two mapping repair techniques, however further improvement should be
done in this line.
      </p>
      <sec id="sec-6-1">
        <title>Evaluation setting, participation and success</title>
        <p>We have run the evaluation in a high performance server with 16 CPUs and allocating
15 Gb RAM. Precision, Recall and F-measure have been computed with respect to the
UMLS-based reference alignment. Systems have been ordered in terms of F-measure.</p>
        <p>In the largebio test case, 13 out of 21 participating systems have been able to cope
with at least one of the tasks of the largebio test case. Synthesis, WeSeEMatch and
WikiMatch failed to complete the smallest task with a time out of 18 hours, while
MapSSS, RiMOM, CIDER-CL, CroMatcher and OntoK threw an exception during the
matching process. The latter two threw an out-of-memory exception. In total we have
evaluated 20 system configurations.
6.3</p>
      </sec>
      <sec id="sec-6-2">
        <title>Tool variants and background knowledge</title>
        <p>There were, in this test case, different variants of tools using background knowledge to
certain degree. These are:
– XMap participates with two variants. XMapSig, which uses a sigmoid function,
and XMapGen, which implements a genetic algorithm. ODGOMS also participates
with two versions (v1.1 and v1.2). ODGOMS-v1.1 is the original submitted version
while ODGOMS-v1.2 includes some bug fixes and extensions.
– LogMap has also been evaluated with two variants: LogMap and LogMap-BK.</p>
        <p>LogMap-BK uses normalisations and spelling variants from the general
(biomedical) purpose UMLS Lexicon9 while LogMap has this feature deactivated.
– AML has been evaluated with 6 different variants depending on the use of repair
techniques (R), general background knowledge (BK) and specialised background
knowledge based on the UMLS Metathesaurus (SBK).
– YAM++ and MaasMatch also use the general purpose background knowledge
provided by WordNet10.</p>
        <p>Since the reference alignment of this test case is based on the UMLS Metathesaurus,
we did not included within the results the alignments provided by AML-SBK and
AMLSBK-R. Nevertheless we consider their results very interesting: AML-SBK and
AMLSBK-R averaged F-measures higher than 0.90 in all 6 tasks.</p>
        <p>We have also re-run the OAEI 2012 version of GOMMA. The results of GOMMA
may slightly vary w.r.t. those in 2012 since we have used a different reference
alignment.
9 http://www.nlm.nih.gov/pubs/factsheets/umlslex.html
10 http://wordnet.princeton.edu/</p>
        <sec id="sec-6-2-1">
          <title>LogMapLt</title>
          <p>IAMA
AML
AML-BK
AML-R
GOMMA2012
AML-BK-R
YAM++
LogMap-BK
LogMap
ServOMap
SPHeRe (*)
XMapSiG
XMapGen
Hertuda
ODGOMS-v1.1
HotMatch
ODGOMS-v1.2
StringsAuto
MaasMatch
# Systems</p>
          <p>FMA-NCI
Task 1 Task 2</p>
        </sec>
        <sec id="sec-6-2-2">
          <title>FMA-SNOMED</title>
          <p>Task 3 Task 4</p>
        </sec>
        <sec id="sec-6-2-3">
          <title>SNOMED-NCI</title>
          <p>Task 5 Task 6
7
13
16
38
18
39
42
93
44
41
140
16
1,476
1,504
3,403
6,366
4,372
10,204
6,358
12,409
20
Together with Precision, Recall, F-measure and Runtimes we have also evaluated the
coherence of alignments. We report (1) the number of unsatisfiabilities when reasoning
with the input ontologies together with the computed mappings, and (2) the ratio of
unsatisfiable classes with respect to the size of the union of the input ontologies.</p>
          <p>
            We have used the OWL 2 reasoner MORe [2] to compute the number of
unsatisfiable classes. For the cases in which MORe could not cope with the input ontologies and
the mappings (in less than 2 hours) we have provided a lower bound on the number of
unsatisfiable classes (indicated by ) using the OWL 2 EL reasoner ELK [
            <xref ref-type="bibr" rid="ref5">20</xref>
            ].
          </p>
          <p>In this OAEI edition, only three systems have shown mapping repair facilities,
namely: YAM++, AML with (R)epair configuration and LogMap. Tables 7-10 show
that even the most precise alignment sets may lead to a huge amount of unsatisfiable
classes. This proves the importance of using techniques to assess the coherence of the
generated alignments.</p>
        </sec>
      </sec>
      <sec id="sec-6-3">
        <title>Runtimes and task completion</title>
        <p>Table 6 shows which systems (including variants) were able to complete each of the
matching tasks in less than 18 hours and the required computation times. Systems have
been ordered with respect to the number of completed tasks and the average time
required to complete them. Times are reported in seconds.</p>
        <p>The last column reports the number of tasks that a system could complete. For
example, 12 system configurations were able to complete all six tasks. The last row
shows the number of systems that could finish each of the tasks. The tasks involving
SNOMED were also harder with respect to both computation times and the number of
systems that completed the tasks.</p>
      </sec>
      <sec id="sec-6-4">
        <title>Results for the FMA-NCI matching problem</title>
        <p>Table 7 summarizes the results for the tasks in the FMA-NCI matching problem.
LogMap-BK and YAM++ provided the best results in terms of both Recall and
Fmeasure in Task 1 and Task 2, respectively. IAMA provided the best results in terms
of precision, although its recall was below average. Hertuda provided competitive
results in terms of recall, but the low precision damaged the final F-measure. On the other
hand, StringsAuto, XMapGen and XMapSiG provided a set of alignments with high
precision, however, the F-measure was damaged due to the low recall of their
alignments. Overall, the results were very positive and many systems obtained an F-measure
higher than 0.80 in the two tasks.</p>
        <p>Efficiency in Task 2 has decreased with respect to Task 1. This is mostly due to
the fact that larger ontologies also involve more possible candidate alignments and it is
harder to keep high precision values without damaging recall, and vice versa.
6.7</p>
      </sec>
      <sec id="sec-6-5">
        <title>Results for the FMA-SNOMED matching problem</title>
        <p>Table 8 summarizes the results for the tasks in the FMA-SNOMED matching problem.
YAM++ provided the best results in terms of F-measure on both Task 3 and Task 4.
YAM++ also provided the best Precision and Recall in Task 3 and Task 4, respectively;
while AML-BK provided the best Recall in Task 3 and AML-R the best Precision in
Task 4.</p>
        <p>Overall, the results were less positive than in the FMA-NCI matching problem
and only YAM++ obtained an F-measure greater than 0.80 in the two tasks.
Furthermore, 9 systems failed to provide a recall higher than 0.4. Thus, matching FMA against
SNOMED represents a significant leap in complexity with respect to the FMA-NCI
matching problem.</p>
        <p>As in the FMA-NCI matching problem, efficiency also decreases as the ontology
size increases. The most important variations were suffered by SPHeRe, IAMA and
GOMMA in terms of precision.
6.8</p>
      </sec>
      <sec id="sec-6-6">
        <title>Results for the SNOMED-NCI matching problem</title>
        <p>Table 9 summarizes the results for the tasks in the SNOMED-NCI matching problem.
LogMap-BK and ServOMap provided the best results in terms of both Recall and
Fmeasure in Task 5 and Task 6, respectively. YAM++ provided the best results in terms
of precision in Task 5 while AML-R in Task 6.</p>
        <sec id="sec-6-6-1">
          <title>LogMap-BK YAM++</title>
          <p>GOMMA2012
AML-BK-R
AML-BK
LogMap
AML-R
ODGOMS-v1.2
AML
LogMapLt
ODGOMS-v1.1
ServOMap
SPHeRe
HotMatch
Average
IAMA
Hertuda
StringsAuto
XMapGen
XMapSiG
MaasMatch</p>
        </sec>
        <sec id="sec-6-6-2">
          <title>System</title>
          <p>YAM++
GOMMA2012
LogMap
LogMap-BK
AML-BK
AML-BK-R
Average
AML-R
AML
SPHeRe
ServOMap
LogMapLt
IAMA
45
94
40
43
39
41
19
10,205
16
8
6,366
141
16
4,372
2,330</p>
          <p>14
3,404
6,359
1,504
1,477
12,410
366
243
162
173
201
205
1,064
194
202
8,136
2,690
60
139</p>
        </sec>
        <sec id="sec-6-6-3">
          <title>Task 2: whole FMA and NCI ontologies</title>
        </sec>
        <sec id="sec-6-6-4">
          <title>Time (s) # Mappings Scores F-m.</title>
          <p>Rec.</p>
        </sec>
        <sec id="sec-6-6-5">
          <title>Task 1: small FMA and NCI fragments Scores # Mappings Prec. F-m.</title>
          <p>2,727
2,561
2,626
2,619
2,695
2,619
2,506
2,558
2,581
2,483
2,456
2,512
2,359
2,280
2,527
1,751
4,309
1,940
1,687
1,564
3,720
2,759
2,843
2,667
2,668
2,828
2,761
2,711
2,368
2,432
2,610
3,235
3,472
1,894</p>
          <p>As in the previous matching problems, efficiency decreases as the ontology size
increases. For example, in Task 6, only ServOMap and YAM++ could reach an F-measure
higher than 0.7. The results were also less positive than in the FMA-SNOMED
matching problem, and thus, the SNOMED-NCI case represented another leap in complexity.
Task 4: whole FMA ontology with SNOMED large fragment</p>
        </sec>
        <sec id="sec-6-6-6">
          <title>Time (s) # Mappings Scores F-m.</title>
        </sec>
      </sec>
      <sec id="sec-6-7">
        <title>Summary results for the top systems</title>
        <p>Table 10 summarizes the results for the systems that completed all 6 tasks of the Large
BioMed Track. The table shows the total time in seconds to complete all tasks and
averages for Precision, Recall, F-measure and Incoherence degree. The systems have
been ordered according to the average F-measure.</p>
        <p>Prec.
Task 6: whole NCI ontology with SNOMED large fragment</p>
        <sec id="sec-6-7-1">
          <title>Time (s) # Mappings Prec. 0.89</title>
          <p>YAM++ was a step ahead and obtained the best average Precision and Recall.
AMLR obtained the second best Precision while AML-BK obtained the second best Recall.</p>
          <p>Regarding mapping incoherence, LogMap-BK computed, on average, the mapping
sets leading to the smallest number of unsatisfiable classes. The configurations of AML
using (R)epair also obtained very good results in terms mapping coherence.</p>
          <p>Finally, LogMapLt was the fastest system. The rest of the tools, apart from
ServoMap and SPHeRe, were also very fast and only needed between 11 and 53 minutes
to complete all 6 tasks. ServOMap required around 4 hours to complete them while
SPHeRe required almost 12 hours.
YAM++
AML-BK
LogMap-BK
AML-BK-R
AML
LogMap
AML-R
ServOMap
GOMMA2012
LogMapLt
SPHeRe
IAMA</p>
        </sec>
        <sec id="sec-6-7-2">
          <title>Total Time (s)</title>
        </sec>
        <sec id="sec-6-7-3">
          <title>Prec. F-m.</title>
        </sec>
        <sec id="sec-6-7-4">
          <title>Average</title>
          <p>Rec. Inc. Degree
Although the proposed matching tasks represent a significant leap in complexity with
respect to the other OAEI test cases, the results have been very promising and 12
systems (including all system configurations) completed all matching tasks with very
competitive results.</p>
          <p>There is, however, plenty of room for improvement: (1) most of the participating
systems disregard the coherence of the generated alignments; (2) the size of the input
ontologies should not significantly affect efficiency, and (3) recall in the tasks involving
SNOMED should be improved while keeping the current precision values.</p>
          <p>
            The alignment coherence measure was the weakest point of the systems
participating in this test case. As shown in Tables 7-10, even highly precise alignment sets may
lead to a huge number of unsatisfiable classes. The use of techniques to assess mapping
coherence is critical if the input ontologies together with the computed mappings are
to be used in practice. Unfortunately, only a few systems in OAEI 2013 have shown to
successfully use such techniques. We encourage ontology matching system developers
to develop their own repair techniques or to use state-of-the-art techniques such as
Alcomo [
            <xref ref-type="bibr" rid="ref6">21</xref>
            ], the repair module of LogMap (LogMap-Repair) [
            <xref ref-type="bibr" rid="ref2">17</xref>
            ] or the repair module
of AML [
            <xref ref-type="bibr" rid="ref11">26</xref>
            ], which have shown to work well in practice [
            <xref ref-type="bibr" rid="ref4">19</xref>
            ].
7
          </p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-7">
      <title>MultiFarm</title>
      <p>
        For evaluating the ability of matching systems to deal with ontologies in different
natural languages, the MultiFarm data set has been proposed [
        <xref ref-type="bibr" rid="ref7">22</xref>
        ]. This data set results from
the translation of 7 Conference test case ontologies (cmt, conference, confOf, iasted,
sigkdd, ekaw and edas), into 8 languages (Chinese, Czech, Dutch, French, German,
Portuguese, Russian, and Spanish, in addition to English). The 9 language versions
result in 36 pairs of languages. For each pair of language, we take into account the
alignment direction (cmten!confOfde and cmtde!confOfen, for instance, as two matching
tasks), what results in 49 alignments. Hence, MultiFarm contains 36
tasks.
49 matching
For the 2013 evaluation campaign, we have used a subset of the whole MultiFarm data
set, omitting all the matching tasks involving the edas and ekaw ontologies (resulting in
36 25 = 900 matching tasks). In this sub set, we can distinguish two types of matching
tasks: (i) those test cases where two different ontologies have been translated in different
languages, e.g., cmt!confOf, and (ii) those test cases where the same ontology has
been translated in different languages, e.g., cmt!cmt. For the test cases of type (ii),
good results are not necessarily related to the use of specific techniques for dealing
with ontologies in different natural languages, but on the ability to exploit the fact that
both ontologies have an identical structure (and that the reference alignment covers all
entities described in the ontologies).
      </p>
      <p>This year, 7 systems (out of 23 participants, see Table 2) use specific cross-lingual11
methods : CIDER-CL, MapSSS, RiMOM2013, StringsAuto, WeSeE, WikiMatch, and
YAM++. This maintains the number of participants implementing specific modules as
in 2012 (ASE, AUTOMSv2, GOMMA, MEDLEY, WeSeE, WikiMatch, and YAM++),
counting on 4 new participants (some of them, extensions of systems participating in
previous campaigns). The other systems are not specifically designed to match
ontologies in different languages nor do they use any component for that purpose.
CIDERCL uses textual definitions of concepts (from Wikipedia articles) and computes
cooccurrence information between multilingual definitions. MapSSS, StringsAuto and
RiMOM201312 apply translation, using Google Translator API, before the matching
step. In particular, RiMOM2013 uses a two-step translation: a first step for translating
labels from the target language into the source language and a second step for
translating all labels into English (for using WordNet). WeSeE uses the Web Translator API
and YAM++ uses Microsoft Bing Translation, where both of them consider English as
pivot language. Finally, WikiMatch exploits Wikipedia for extracting cross-language
links for helping in the task of finding correspondences between the ontologies.</p>
      <sec id="sec-7-1">
        <title>Execution setting and runtime</title>
        <p>
          All systems have been executed on a Debian Linux virtual machine configured with
four processors and 20GB of RAM running under a Dell PowerEdge T610 with 2*Intel
11 We have revised the definitions of multilingual and cross-lingual matching. Initially, as
reported in [
          <xref ref-type="bibr" rid="ref7">22</xref>
          ], MultiFarm was announced as a benchmark for multilingual ontology matching,
i.e., multilingual in the sense that we have a set of ontologies in 8 languages. However, it
is more appropriate to use the term cross-lingual ontology matching. Cross-lingual ontology
matching refers to the matching cases where each ontology uses a different natural language
(or a different set of natural languages) for entity naming, i.e., the intersection of sets is empty.
        </p>
        <p>It is the case of the matching tasks in MultiFarm.
12 These 3 systems have encountered problems for accessing Google servers. New versions of
these tools were received after the deadline, improving, for some test cases, the results reported
here.</p>
        <p>Xeon Quad Core 2.26GHz E5607 processors and 32GB of RAM, under Linux ProxMox
2 (Debian). The runtimes for each system can be found in Table 11. The measurements
are based on 1 run. We can observe large differences between the time required for a
system to complete the 900 matching tasks. While RiMOM requires around 13 minutes,
WeSeE takes around 41 hours. As we have used this year a different setting from the
one in 2012, we are not able to compare runtime measurements over the campaigns.
Overall results Before discussing the results per pairs of languages, we present the
aggregated results for the test cases within type (i) and (ii) matching task. Table 11 shows
the aggregated results. Systems not listed in this table have generated empty alignments,
for most test cases (ServOMap) or have thrown exceptions (CroMatcher, XMapGen,
XMapSiG). For computing these results, we do not distinguish empty and erroneous
alignments. As shown in Table 11, we observe significant differences between the
results obtained for each type of matching task (specially in terms of precision). Most
of the systems that implement specific cross-lingual techniques – YAM++ (.40),
WikiMatch (.27), RiMOM2013 (.21), WeSeE (.15), StringsAuto (.14), and MapSSS (.10) –
generate the best results for test cases of type (i). For the test cases of type (ii), systems
non specifically designed for cross-lingual matching – MaasMatch and OntoK – are
in the top-5 F-measures together with YAM++, WikiMatch and WeSeE. Concerning
CIDER-CL, this system in principle is able to deal with a subset of languages, i.e., DE,
EN, ES, and NL.</p>
        <p>Overall (for both types i and ii), in terms of F-measure, most systems implementing
specific cross-lingual methods outperform non-specific systems: YAM++ (.50),
WikiMatch (.22), RiMOM (.17), WeSeE (.15) – with MaasMatch given its high scores on
cases (ii) – and StringsAuto (.10).</p>
        <p>Comparison with previous campaigns In the first year of evaluation of MultiFarm,
we have used a subset of the whole data set, where we omitted the ontologies edas and
ekaw, and suppressed the test cases where Russian and Chinese were involved. Since
2012, we have included Russian and Chinese translations, but still have not included
edas and ekaw. In the 2011.5 intermediary campaign, 3 participants (out of 19) used
specific techniques – AUTOMSv2, WeSeE, and YAM++. In 2012, 7 systems (out of
24) implemented specific techniques for dealing with ontologies in different natural
languages – ASE, AUTOMSv2, GOMMA, MEDLEY, WeSeE, WikiMatch, and YAM++.
This year, as in 2012, 7 participants out of 21 use specific techniques: 2 of them have
been participating since 2011.5 (WeSeE and YAM), 1 since 2012 (WikiMatch), 3
systems (CIDER-CL, RiMOM2013 and MapSSS) have included cross-lingual approaches
in their implementations, and 1 new system (StringsAuto) has participated.</p>
        <p>Comparing 2012 and 2013 results (on the same basis), WikiMatch improved
precision for both test case types – from .22 to .34 for type (i) and .43 to .65 for type (ii) –
preserving its values of recall. On the other hand, WeSeE has decreased both precision
– from .61 to .22 – and recall – from .32 to .12 – for type (i) and precision – from .90 to
.56 – and recall – from .27 to .09 – for type (ii).</p>
        <sec id="sec-7-1-1">
          <title>Same ontologies (ii)</title>
          <p>System</p>
          <p>CIDER-CL
la MapSSS
gu RiMOM2013
il-rssnoC SWtriinkWgiMseAaSutectEho</p>
          <p>YAM++
c
fi
i
c
e
p
s
n
o
N</p>
          <p>AML
HerTUDA
HotMatch</p>
          <p>IAMA</p>
          <p>LogMap
LogMapLite
MaasMatch
ODGOMS</p>
          <p>OntoK
Synthesis
Language specific results Table 12 shows the results aggregated per language pair,
for the the test cases of type (i). For the sake of readability, we present only F-measure
values. The reader can refer to the OAEI results web page for more detailed results on
precision and recall. As expected and already reported above, the systems that apply
specific strategies to match ontology entities described in different natural languages
outperform the other systems. For most of these systems, the best performance is
observed for the pairs of language including Dutch, English, German, Spanish, and
Portuguese: CIDER-CL (en-es .21, en-nl and es-nl .18, de-nl .16), MapSSS (es-pt and en-es
.33, de-en .28, de-es .26), RiMOM2013 (en-es .42, es-pt .40, de-en .39), StringsAuto
(en-es .37, es-pt .36, de-en .33), WeSeE (en-es .46, en-pt .41, en-nl .40), WikiMatch
(en-es .38, en-pt, es-pt and es-fr .37, es-ru .35). The exception is YAM++ which
generates its best results for the pairs including Czech : cz-en and en-pt .57, cz-pt .56, cz-nl
and fr-pt .53. For all specific systems, English is present in half of the top pairs.</p>
          <p>For non-specific systems, most of them cannot deal at all with Chinese and Russian
languages. 7 out of 10 systems generate their best results for the pair es-pt (followed
by the pair de-en). Again, similarities in the language vocabulary have an important
role in the matching task. On the other hand, although it is likely harder to find
correspondences between cz-pt than es-pt, for some systems Czech is on pairs for the top-5
F-measure (cz-pt, for AML, IAMA, LogMap, LogMapLite and Synthesis). It can be
explained by the specific way systems combine their internal matching techniques
(ontology structure, reasoning, coherence, linguistic similarities, etc).
+
+
M
A</p>
          <p>Y
3
1
0
2
M
O
M
i</p>
          <p>R
e
t
i
L
p
a
M
g
o
L
o
t
u
A
s
g
n
i
r
t
S
As expected, systems using specific methods for dealing with ontologies in different
languages work much better than non specific systems. However, the absolute results
are still not very good, if compared to the top results of the original Conference data set
(approximatively 75% F-measure for the best matcher). For all specific cross-lingual
methods, the techniques implemented in YAM++, as in 2012, generate the best
alignments in terms of F-measure (around 50% overall F-measure for both types of matching
tasks). All systems privilege precision rather than recall. Although we count this year
on 4 new systems implementing specific cross-lingual methods, there is room for
improvements to achieve the same level of compliance as in the original data set.
8</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-8">
      <title>Library</title>
      <p>The library test case was established in 201213. The test case consists of matching of
two real-world thesauri: The Thesaurus for the Social Sciences (TheSoz, maintained
by GESIS) and the Standard Thesaurus for Economics (STW, maintained by ZBW).
The reference alignment is based on a manually created alignment from 2006. As
additional benefit from this test case, the reference alignment is constantly improved by
the maintainers by manually checking the generated correspondences that have not yet
been checked and that are not part of the reference alignment14.
Both thesauri used in this test case are comparable in many respects. They have roughly
the same size (6,000 resp. 8,000 concepts), are both originally developed in German,
are today both multilingual, both have English translations, and, most important,
despite being from two different domains, they have significant overlapping areas. Not
least, both are freely available in RDF using SKOS15. To enable the participation of
all OAEI matchers, an OWL version of both thesauri is provided, effectively by
creating a class hierarchy from the concept hierarchy. Details are provided in the report
of the 2012 campaign [1]. As stated above, we updated the reference alignment with
all correct correspondences found during the 2012 campaign, it now consists of 3161
correspondences.
All matching processes have been performed on a Debian machine with one 2.4GHz
core and 7GB RAM allocated to each system. The evaluation has been executed by
using the SEALS infrastructure. Each participating system uses the OWL version.</p>
      <p>To compare the created alignments with the reference alignment, we use
the Alignment API. For this evaluation, we only included equivalence relations
(skos:exactMatch). We computed precision, recall and F1-measure for each matcher.
Moreover, we measured the runtime, the size of the created alignment and checked
13 There has already been a library test case from 2007 to 2009 using different thesauri, as well
as other thesaurus test cases like the food and the environment test cases.
14 With the reasonable exception of XMapGen, which produces almost 40.000 correspondences.
15 http://www.w3.org/2004/02/skos
whether a 1:1 alignment has been created. To assess the results of the matchers, we
developed three straightforward matching strategies, using the original SKOS version
of the thesauri:
– MatcherPrefDE: Compares the German lower-case preferred labels and generates
a correspondence if these labels are completely equivalent.
– MatcherPrefEN: Compares the English lower-case preferred labels and generates a
correspondence if these labels are completely equivalent.
– MatcherPref: Creates a correspondence, if either MatcherPrefDE or
Matcher</p>
      <p>PrefEN or both create a correspondence.
– MatcherAllLabels: Creates a correspondences whenever at least one label
(preferred or alternative, all languages) of an entity is equivalent to one label of another
entity.
Of all 21 participating matchers (or variants), 12 were able to generate an alignment
within 12 hours. CroMatcher, MaasMatch, RiMOM2013, WeSeE and WikiMatch did
not finish in the time frame, OntoK had heap space problems and CiderCL, MapSSS
and Synthesis threw an exception. The results can be found in Table 13.</p>
      <p>Precision F-Measure Recall Time (ms) Size 1:1</p>
      <p>The best systems in terms of F-measure are ODGOMS and YAM++. These
matchers also have a higher F-measure than MatcherPref. ServOMap and AML are below
this baseline but better than MatcherPrefDE and MatcherAllLabels. A group of
matchers including LogMap, LogMapLite, HerTUDA and HotMatch are above the
MatcherPrefEN baseline. Compared to last year evaluation with the updated reference
alignment, the matchers clearly improved: in 2012, no matcher was able to beat MatcherPref
and MatcherPrefDE, only ServOMapLt was better than MatcherAllLabels. Today, two
matchers outperformed all baselines; further two matchers outperformed all baselines
but MatcherPref. This is remarkable, as the matchers are still not able to consume SKOS
and therefore neglect the distinction between preferred and alternative labels. The
baselines are tailored for very high precision by design, while the matchers usually have a
higher recall. This is reflected in the F-measure, where the highest value increased from
0.72 to 0.76 by almost 5 percentage points since last year. The recall mostly increased,
e.g. YAM++ from 0.76 to 0.81 (without affecting the precision negatively, which also
increased from 0.68 to 0.69).</p>
      <p>Like in the previous year, an additional intellectual evaluation of the alignments
established automatically was done by a domain expert to further improve the reference
alignment. Unsurprisingly, the matching tools predominantly detected matches based
on the character string. This included the term alone as well as the term’s context.
Especially in the case of short terms, this could easily lead to wrong correspondences,
e.g., “tea” 6= “team”, “sheep” 6= “sleep”. Except for its sequence of letters the term’s
context was not taken into account.</p>
      <p>This sole attention to the character string was a main source of error in cases in
which on the term as well as on the context level similar terminological entities
appeared, e.g., “Green revolution” subject category: “Development Politic” 6= “permanent
revolution” subject category: “Political Developments and Processes”.</p>
      <p>Additionally, identical components of a compound frequently lead to incorrect
correspondences, e.g., “prohibition of interest” 6= “prohibition of the use of force”.
Moreover, terms in different domains might look similar, but in fact have very different
meanings. An illustrative example is “Chicago Antitrust Theory” 6= “Chicago School”, where
indeed the same Chicago is referenced, but without any effect on the (dis-)similarity of
both concepts.
The overall performance improvement is encouraging in this test case. While it might
not look impressive to beat simple baselines as ours at first sight, it is actually a notable
achievement. The baselines are not only tailored for very high precision, benefitting
from the fact that in many cases a consistent terminology is used, they also exploit
additional knowledge about the labels. The matchers are general-purpose matchers that
have to perform well in all OAEI test cases. Nonetheless, there does not seems to be
matchers who understand SKOS in order to make use of the many concept hierarchies
provided on the Web.</p>
      <p>Generally, matchers still rely too much on the character string of the labels and
the labels of the concepts in the immediate vicinity. During the intellectual evaluation
process, it became obvious that a multitude of incorrect matches could be prevented
if the subject categories, respectively the thesauri’s classification schemes be matched
beforehand. In many cases, misleading candidate correspondences could be discarded
by taking these higher levels of the hierarchy into account. It could be prevented, for
example, to build up correspondences between personal names and subject headings. A
thesaurus, however, is not a classification system. The disjointness of two subthesauri
is therefore not easy to establish, let alone to detect by automatic means. Nonetheless,
thesauri oftentimes have their own classification schemes which partly follow
classification principles. We believe that further exploiting this context knowledge could be
worthwhile.
9</p>
    </sec>
    <sec id="sec-9">
      <title>Interactive matching</title>
      <p>
        The interactive matching test case was evaluated at OAEI 2013 for the first time. The
goal of this evaluation is to simulate interactive matching [
        <xref ref-type="bibr" rid="ref8">23</xref>
        ], where a human expert
is involved to validate mappings found by the matching system. In the evaluation, we
look at how user interaction may improve matching results.
      </p>
      <p>For the evaluation, we use the conference data set 5 with the ra1 alignment, where
there is quite a bit of room for improvement, with the best fully automatic, i.e.,
noninteractive matcher achieving an F-measure below 80%. The SEALS client was
modified to allow interactive matchers to ask an oracle, which emulates a (perfect) user. The
interactive matcher can present a correspondence to the oracle, which then tells the user
whether the correspondence is right or wrong.</p>
      <p>All matchers participating in the interactive test case support both interactive and
non-interactive matching. This allows us to analyze how much benefit the interaction
brings for the individual matchers.
Overall, five matchers participated in the interactive matching test case: AML and
AML-bk, Hertuda, LogMap, and WeSeE-Match. All of them implement interactive
strategies that run entirely as a post-processing step to the automatic matching, i.e., take
the alignment produced by the base matcher and try to refine it by selecting a suitable
subset.</p>
      <p>AML and AML-bk present all correspondences below a certain confidence
threshold to the oracle, starting with the highest confidence values. They stop adding
references once the false positive rate exceeds a certain threshold. Similarly, LogMap checks
all questionable correspondences using the oracle. Hertuda and WeSeE-Match try to
adaptively set an optimal threshold for selecting correspondences. They perform a
binary search in the space of possible thresholds, presenting a correspondence of average
confidence to the oracle first. If the result is positive, the search is continued with a
higher threshold, otherwise with a lower threshold.</p>
      <p>The results are depicted in Table 14. Please note that the values in this table slightly
differ from the original values in the conference test case, since the latter uses micro
average recall and precision, while we use macro averages, so that we can compute
significance levels using T-Tests on the series of recall and precision values from the
individual test cases. The reason for the strong divergence of the results for WeSeE
to the conference test case results is unknown. Altogether, the biggest improvement in
F-measure, as well as the best overall result (although almost at the same level as
AMLbk), is achieved by LogMap, which increases its F-measure by four percentage points.
Furthermore, LogMap, AML and AML-bk show a statistically significant increase in
recall as well as precision, while all the other tools except for Hertuda show a significant
increase in precision. The increase in precision is in all cases however higher than the
increase of recall. It can be observed for AML, AML-bk and LogMap that a highly
significant increase in precision also increases F-measure at a high significance level,
even if the increase in recall is less significant.</p>
      <p>At the same time, LogMap has the lowest number of interactions with the oracle,
which shows that it also makes the most efficient use of the oracle. In a truly interactive
setting, this would mean that the manual effort is minimized. Furthermore, it is the only
tool that presents more positive than negative examples to the oracle.</p>
      <p>On the other hand, Hertuda and WeSeE even show a decrease in recall, which
cannot be compensated by the increase in precision. The biggest increase in precision (17
percentage points) is achieved by WeSeE, but on an overall lower level than the other
matching systems. Thus, we conclude that their strategy is not as efficient as those of
the other participants.</p>
      <p>
        Compared to the results of the non-interactive conference test case, the best
interactive matcher (in terms of F-measure) is slightly below the best matcher (YAM++)
with a F-measure value of 0.76 (using macro averages). Except for YAM++, the
interactive versions of AML-bk, AML and LogMap achieve better F-measure scores than
all non-interactive matchers.
The results show that current interactive matching tools mainly use interaction as a
means to post-process an alignment found with fully automatic means. There are,
however, other interactive approaches that can be thought of, which include interaction at
an earlier stage of the process, e.g., using interaction for parameter tuning [
        <xref ref-type="bibr" rid="ref10">25</xref>
        ], or
determining anchor elements for structure-based matching approaches using interactive
methods. The maximum F-measure of 0.73 achieved shows that there is still room for
improvement.
      </p>
      <p>Furthermore, different variations of the evaluation method can be thought of,
including different noise levels in the oracle’s responses, i.e., simulating errors made by
the human expert, or allowing other means of interactions than the validation of single
correspondences, e.g., providing a random positive example, or providing the
corresponding element in one ontology, given an element of the other one.</p>
      <p>
        So far, we only compare the final results of the interactive matching process. In
[
        <xref ref-type="bibr" rid="ref8">23</xref>
        ], we have introduced an evaluation method based on learning curves, which gives
insights into how quickly the matcher converges towards its final result. However, we
have not implemented that model for this year’s OAEI, since it requires more changes to
the matchers (each matcher has to provide an intermediate result at any point in time).
10
      </p>
    </sec>
    <sec id="sec-10">
      <title>Instance matching</title>
      <p>
        The instance matching track aims at evaluating the performance of different matching
tools on the task of matching RDF individuals which originate from different sources
but describe the same real-world entity [
        <xref ref-type="bibr" rid="ref1">16</xref>
        ].
Starting from the experience of previous editions of the instance matching track in
OAEI [15], this year we provided a set of RDF-based test cases, called RDFT, that is
automatically generated by introducing controlled transformations in some initial RDF
data sets. The controlled transformations introduce artificial distortions into the data,
which include data value transformations as well as structural transformations. RDFT
includes blind evaluation. The participants are provided with a list of five test cases. For
each test case, we provide training data with the accompanying alignment to be used to
adjust the settings of the tools, and contest data, based on which the final results will be
calculated. The evaluation data set is generated by exploiting the same configuration of
the RDFT transformation tool used for generating training data.
      </p>
      <p>The RDFT test cases have been generated from an initial RDF data set about
wellknown computer scientists data extracted from DBpedia. The initial data set is
composed by 430 resources, 11 RDF properties and 1744 triples. Some descriptive statistics
about the initial data set are available online. Starting from the initial data set, we
provided the participants with five test cases, where different transformations have been
implemented, as follows:
– Testcase 1: value transformation. Values of 5 properties have been changed by
randomly deleting/adding chars, by changing the date format, and/or by randomly
change integer values.
– Testcase 2: structure transformation. The length of property path between
resources and values has been changed. Property assertions have been split in two or
more assertions.
– Testcase 3: languages. The same as Testcase 1, but using French translation for
comments and labels instead of English.</p>
      <p>tescase01 tescase02 tescase03 tescase04 tescase05
system Prec. F-m. Rec. Prec. F-m. Rec. Prec. F-m. Rec. Prec. F-m. Rec. Prec. F-m. Rec.
– Testcase 4: combined. A combination of value and structure transformations using</p>
      <p>French text.
– Testcase 5: cardinality. The same as Testcase 4, but now part of the resources have
none or multiple matching counterparts.
An overview of the precision, recall and F1-measure results for the RDFT test cases is
shown in Table 15.</p>
      <p>All the tools show good performances when dealing with singular type of data
transformation, i.e., Testcases 1-3, either value, structural, and language transformations.
Performances drop when different kinds of transformations are combined together, i.e.,
Testcases 4-5, except for RiMOM2013, which still has performances close to 1.0 for
both precision and recall. This suggests that a possible challenge for instance matching
tools is to work in the direction of improving the combination and balancing of different
matching techniques in a single, general-purpose, configuration scheme.</p>
      <p>In addition to precision, recall and F1-measure results, we performed also a test
based on the similarity values provided by participating tools. In particular, we selected
the provided mappings by different thresholds on the similarity values, in order to
monitor the behavior of precision and recall16. Results of this second evaluation are shown
in Figure 3.</p>
      <p>Testing the results when varying the threshold used for mapping selection is useful
to understand how robust are the mappings retrieved by the participating tools. In
particular, RiMOM2013 is the only tool which has very good results with all the threshold
values that have been tested. This means that the retrieved mappings are generally
correct and associated with high levels of similarity. Other tools, especially LilyIOM and
LogMap, retrieve a high number of mappings which are associated with low levels of
confidence. In such cases, when we rely only on mappings between resources that are
considered very similar by the tool, the quality of results becomes lower.</p>
      <p>Finally, as a general remark suggested from the result analysis, we stress the
opportunity of working toward two main goals in particular: one one side, on the integration
of different matching techniques and the need of conceiving self-adapting tools,
capable of self-configuring the most suitable combination of matching metrics according to
the nature of data heterogeneity that needs to be handled; on the other side, the need for
16 This experiment is partially useful in the case of SLINT+, where the similarity values are not
in the range [0,1] and are not necessarily proportional to the elements similarity.
testcase01, f1−measure</p>
      <p>● ● ● ● ● ● ●</p>
      <p>● ● ● ● ●
testcase03, f1−measure
testcase04, f1−measure
testcase05, f1−measure</p>
      <p>● ● ● ● ● ● ●
testcase03, precision
● ● ● ● ● ● ● ● ●
testcase04, precision</p>
      <p>● ● ● ● ●
testcase05, precision</p>
      <p>● ●
● ● ● ● ● ●
tools capable of providing a degree of confidence which could be used for measuring
the reliability of the provided mappings.</p>
      <p>testcase01, precision
● ● ● ● ● ● ● ●
● ● ● ● ●</p>
      <p>● ● ● ● ●
● ●
● ●
● ●
● ●
testcase01, recall
testcase02, recall
testcase03, recall
testcase04, recall
testcase05, recall
0.25
0.50</p>
    </sec>
    <sec id="sec-11">
      <title>Lesson learned and suggestions</title>
      <p>There are, this year, very few comments about the evaluation execution:
A) This year indicated again that requiring participants to implement a minimal
interface was not a strong obstacle to participation. Moreover, the community seems to
get used to the SEALS infrastructure introduced for OAEI 2011. This might be one
of the reasons for an increasing participation.</p>
      <p>B) Related to the availability of the platform, participants checked that their tools were
working on minimal tests and discovered in September that they were not working
on other tests. For that reason, it would be good to set the preliminary evaluation
results by the end of July.</p>
      <p>C) Now that all tools are run in exactly the same configuration across all test cases,
some discrepancies appear across such cases. For instance, benchmarks expect only
class correspondences in the name space of the ontologies, some other cases expect
something else. This is a problem, which could be solved either by passing
parameters to the SEALS client (this would make its implementation heavier) or by post
processing results (which may be criticized).</p>
      <p>
        D) [
        <xref ref-type="bibr" rid="ref9">24</xref>
        ] raised and documented objections (on validity and fairness) to the way
reference alignments are made coherent with alignment repair techniques. Appropriate
measures should be taken to mitigate this.
      </p>
      <p>E) Last years we reported that we had many new participants. The same trend can be
observed for 2013.</p>
      <p>F) Again and again, given the high number of publications on data interlinking, it is
surprising to have so few participants to the instance matching track.
12
OAEI 2013 saw an increased number of participants and most of the test cases
performed on the SEALS platform. This is good news for the interoperability of matching
systems.</p>
      <p>Compared to the previous years, we observed improvements of runtimes and the
ability of systems to cope with large ontologies and data sets (testified by the largebio
and instance matching results). This comes in addition to progress in overall F-measure,
which is more observable as the test case is more recent. More preoccupying was the
lack of robustness of some systems observed in the simple benchmarks. This seems
to be due to an increased reliance on the network and networked resources that may
time-out systems.</p>
      <p>As usual, most of the systems favour precision over recall. In general,
participating matching systems do not take advantage of alignment repairing system and return
sometimes incoherent alignments. This is a problem if their result has to be taken as
input by a reasoning system. They do not generally use natural language aware strategies,
while the multilingual tests show the worthiness of such an approach.</p>
      <p>A novelty of this year was the evaluation of interactive systems, included in the
SEALS client. It brings interesting insight on the performances of such systems and
should certainly be continued.</p>
      <p>Most of the participants have provided a description of their systems and their
experience in the evaluation. These OAEI papers, like the present one, have not been peer
reviewed. However, they are full contributions to this evaluation exercise and reflect the
hard work and clever insight people put in the development of participating systems.
Reading the papers of the participants should help people involved in ontology
matching to find what makes these algorithms work and what could be improved. Sometimes
participants offer alternate evaluation results.</p>
      <p>
        The Ontology Alignment Evaluation Initiative will continue these tests by
improving both test cases and testing methodology for being more accurate. Matching
evaluation still remains a challenging topic, which is worth further research in order to
facilitate the progress of the field [
        <xref ref-type="bibr" rid="ref12">27</xref>
        ]. Further information can be found at:
      </p>
    </sec>
    <sec id="sec-12">
      <title>Acknowledgements</title>
      <p>We warmly thank the participants of this campaign. We know that they have worked
hard for having their matching tools executable in time and they provided insightful
papers presenting their experience. The best way to learn about the results remains to
read the following papers.</p>
      <p>We are very grateful to STI Innsbruck for providing the necessary infrastructure to
maintain the SEALS repositories.</p>
      <p>We are also grateful to Martin Ringwald and Terry Hayamizu for providing the
reference alignment for the anatomy ontologies and thank Elena Beisswanger for her
thorough support on improving the quality of the data set.</p>
      <p>We thank Christian Meilicke for help with incoherence evaluation within the
conference and his support of the anatomy test case.</p>
      <p>We also thank for their support the other members of the Ontology Alignment
Evaluation Initiative steering committee: Yannis Kalfoglou (Ricoh laboratories, UK),
Miklos Nagy (The Open University (UK), Natasha Noy (Stanford University, USA),
Yuzhong Qu (Southeast University, CN), York Sure (Leibniz Gemeinschaft, DE), Jie
Tang (Tsinghua University, CN), Heiner Stuckenschmidt (Mannheim Universita¨t, DE),
George Vouros (University of the Aegean, GR).</p>
      <p>Bernardo Cuenca Grau, Je´roˆme Euzenat, Ernesto Jimenez-Ruiz, Christian Meilicke,
and Ca´ssia Trojahn dos Santos have been partially supported by the SEALS
(IST-2009238975) European project in the previous years.</p>
      <p>Ernesto and Bernardo have also been partially supported by the Seventh
Framework Program (FP7) of the European Commission under Grant Agreement 318338,
“Optique”, the Royal Society, and the EPSRC projects Score!, ExODA and MaSI3.</p>
      <p>Ca´ssia Trojahn dos Santos and Roger Granada are also partially supported by the
CAPES-COFECUB Cameleon project number 707/11.
1. Jose´ Luis Aguirre, Bernardo Cuenca Grau, Kai Eckert, Je´roˆme Euzenat, Alfio Ferrara,
Robert Willem van Hague, Laura Hollink, Ernesto Jime´nez-Ruiz, Christian Meilicke,
Andriy Nikolov, Dominique Ritze, Franc¸ois Scharffe, Pavel Shvaiko, Ondrej Sva´b-Zamazal,
Ca´ssia Trojahn, and Benjamin Zapilko. Results of the ontology alignment evaluation
initiative 2012. In Proc. 7th ISWC ontology matching workshop (OM), Boston (MA US), pages
73–115, 2012.
2. Ana Armas Romero, Bernardo Cuenca Grau, and Ian Horrocks. MORe: Modular
combination of OWL reasoners for ontology classification. In Proc. 11th International Semantic Web
Conference (ISWC), Boston (MA US), pages 1–16, 2012.
3. Benhamin Ashpole, Marc Ehrig, Je´r oˆme Euzenat, and Heiner Stuckenschmidt, editors. Proc.</p>
      <p>K-Cap Workshop on Integrating Ontologies, Banff (Canada), 2005.
4. Olivier Bodenreider. The unified medical language system (UMLS): integrating biomedical
terminology. Nucleic Acids Research, 32:267–270, 2004.
5. Caterina Caracciolo, Je´roˆ me Euzenat, Laura Hollink, Ryutaro Ichise, Antoine Isaac,
Ve´ronique Malaise´, Christian Meilicke, Juan Pane, Pavel Shvaiko, Heiner Stuckenschmidt,
Ondrej Sva´b-Zamazal, and Vojtech Sva´tek. Results of the ontology alignment evaluation
initiative 2008. In Proc. 3rd ISWC ontology matching workshop (OM), Karlsruhe (DE), pages
73–120, 2008.
6. Je´r oˆme David, Je´roˆ me Euzenat, Franc¸ois Scharffe, and Ca´ssia Trojahn dos Santos. The
alignment API 4.0. Semantic web journal, 2(1):3–10, 2011.
7. Je´r oˆme Euzenat, Alfio Ferrara, Laura Hollink, Antoine Isaac, Cliff Joslyn, Ve´ronique
Malaise´, Christian Meilicke, Andriy Nikolov, Juan Pane, Marta Sabou, Franc¸ois Scharffe,
Pavel Shvaiko, Vassilis Spiliopoulos, Heiner Stuckenschmidt, Ondrej Sva´b-Zamazal,
Vojtech Sva´tek, Ca´ssia Trojahn dos Santos, George Vouros, and Shenghui Wang. Results of
the ontology alignment evaluation initiative 2009. In Proc. 4th ISWC ontology matching
workshop (OM), Chantilly (VA US), pages 73–126, 2009.
8. Je´r oˆme Euzenat, Alfio Ferrara, Christian Meilicke, Andriy Nikolov, Juan Pane, Franc¸ois
Scharffe, Pavel Shvaiko, Heiner Stuckenschmidt, Ondrej Sva´b-Zamazal, Vojtech Sva´tek, and
Ca´ssia Trojahn dos Santos. Results of the ontology alignment evaluation initiative 2010. In
Proc. 5th ISWC ontology matching workshop (OM), Shanghai (CN), pages 85–117, 2010.
9. Je´r oˆme Euzenat, Alfio Ferrara, Robert Willem van Hague, Laura Hollink, Christian
Meilicke, Andriy Nikolov, Franc¸ois Scharffe, Pavel Shvaiko, Heiner Stuckenschmidt, Ondrej
Sva´b-Zamazal, and Ca´ssia Trojahn dos Santos. Results of the ontology alignment
evaluation initiative 2011. In Proc. 6th ISWC ontology matching workshop (OM), Bonn (DE),
pages 85–110, 2011.
10. Je´r oˆme Euzenat, Antoine Isaac, Christian Meilicke, Pavel Shvaiko, Heiner Stuckenschmidt,
Ondrej Svab, Vojtech Svatek, Willem Robert van Hage, and Mikalai Yatskevich. Results of
the ontology alignment evaluation initiative 2007. In Proc. 2nd ISWC ontology matching
workshop (OM), Busan (KR), pages 96–132, 2007.
11. Je´r oˆme Euzenat, Christian Meilicke, Pavel Shvaiko, Heiner Stuckenschmidt, and Ca´ssia
Trojahn dos Santos. Ontology alignment evaluation initiative: six years of experience. Journal
on Data Semantics, XV:158–192, 2011.
12. Je´r oˆme Euzenat, Malgorzata Mochol, Pavel Shvaiko, Heiner Stuckenschmidt, Ondrej Svab,
Vojtech Svatek, Willem Robert van Hage, and Mikalai Yatskevich. Results of the ontology
alignment evaluation initiative 2006. In Proc. 1st ISWC ontology matching workshop (OM),
Athens (GA US), pages 73–95, 2006.
13. Je´r oˆme Euzenat, Maria Rosoiu, and Ca´ssia Trojahn dos Santos. Ontology matching
benchmarks: generation, stability, and discriminability. Journal of web semantics, 21:30–48, 2013.
14. Je´r oˆme Euzenat and Pavel Shvaiko. Ontology matching. Springer-Verlag, Heidelberg (DE),
2nd edition, 2013.
15. Alfio Ferrara, Andriy Nikolov, Jan Noessner, and Franc¸ois Scharffe. Evaluation of instance
matching tools: The experience of OAEI. Journal of Web Semantics, 21:49–60, 2013.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          16.
          <string-name>
            <surname>Alfio</surname>
            <given-names>Ferrara</given-names>
          </string-name>
          , Andriy Nikolov, and
          <article-title>Franc¸ois Scharffe. Data linking for the semantic web</article-title>
          .
          <source>International Journal on Semantic Web and Information Systems</source>
          ,
          <volume>7</volume>
          (
          <issue>3</issue>
          ):
          <fpage>46</fpage>
          -
          <lpage>76</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          17.
          <string-name>
            <surname>Ernesto</surname>
          </string-name>
          <article-title>Jime´nez-Ruiz and Bernardo Cuenca Grau</article-title>
          .
          <article-title>LogMap: Logic-based and scalable ontology matching</article-title>
          .
          <source>In Proc. 10th International Semantic Web Conference (ISWC)</source>
          ,
          <source>Bonn (DE)</source>
          , pages
          <fpage>273</fpage>
          -
          <lpage>288</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          18.
          <string-name>
            <surname>Ernesto</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            , Bernardo Cuenca Grau, Ian Horrocks, and
            <given-names>Rafael</given-names>
          </string-name>
          <string-name>
            <surname>Berlanga</surname>
          </string-name>
          .
          <article-title>Logicbased assessment of the compatibility of UMLS ontology sources</article-title>
          .
          <source>J. Biomed. Sem., 2</source>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          19.
          <string-name>
            <surname>Ernesto</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>Christian</given-names>
          </string-name>
          <string-name>
            <surname>Meilicke</surname>
            , Bernardo Cuenca Grau, and
            <given-names>Ian</given-names>
          </string-name>
          <string-name>
            <surname>Horrocks</surname>
          </string-name>
          .
          <article-title>Evaluating mapping repair systems with large biomedical ontologies</article-title>
          .
          <source>In Proc. 26th Description Logics Workshop</source>
          ,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          20.
          <string-name>
            <surname>Yevgeny</surname>
            <given-names>Kazakov</given-names>
          </string-name>
          , Markus Kro¨ tzsch, and Frantisek Simancik.
          <article-title>Concurrent classification of EL ontologies</article-title>
          .
          <source>In Proc. 10th International Semantic Web Conference (ISWC)</source>
          ,
          <source>Bonn (DE)</source>
          , pages
          <fpage>305</fpage>
          -
          <lpage>320</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          21.
          <string-name>
            <given-names>Christian</given-names>
            <surname>Meilicke</surname>
          </string-name>
          .
          <article-title>Alignment Incoherence in Ontology Matching</article-title>
          .
          <source>PhD thesis</source>
          , University Mannheim,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          22.
          <string-name>
            <surname>Christian</surname>
            <given-names>Meilicke</given-names>
          </string-name>
          , Rau´ l Garc´ıa Castro, Frederico Freitas, Willem Robert van Hage,
          <string-name>
            <surname>Elena</surname>
          </string-name>
          Montiel-Ponsoda, Ryan Ribeiro de Azevedo, Heiner Stuckenschmidt, Ondrej Sva´
          <article-title>b-</article-title>
          <string-name>
            <surname>Zamazal</surname>
            , Vojtech Sva´tek, Andrei Tamilin, Ca´ssia Trojahn, and
            <given-names>Shenghui</given-names>
          </string-name>
          <string-name>
            <surname>Wang</surname>
          </string-name>
          .
          <article-title>MultiFarm: A benchmark for multilingual ontology matching</article-title>
          .
          <source>Journal of web semantics</source>
          ,
          <volume>15</volume>
          (
          <issue>3</issue>
          ):
          <fpage>62</fpage>
          -
          <lpage>68</lpage>
          ,
          <year>2012</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          23.
          <string-name>
            <surname>Heiko</surname>
            <given-names>Paulheim</given-names>
          </string-name>
          , Sven Hertling, and
          <string-name>
            <given-names>Dominique</given-names>
            <surname>Ritze</surname>
          </string-name>
          .
          <article-title>Towards evaluating interactive ontology matching tools</article-title>
          .
          <source>In Proc. 10th Extended Semantic Web Conference (ESWC)</source>
          ,
          <source>Montpellier (FR)</source>
          , pages
          <fpage>31</fpage>
          -
          <lpage>45</lpage>
          ,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          24.
          <string-name>
            <surname>Catia</surname>
            <given-names>Pesquita</given-names>
          </string-name>
          , Daniel Faria, Emanuel Santos, and Francisco Couto.
          <article-title>To repair or not to repair: reconciling correctness and coherence in ontology reference alignments</article-title>
          .
          <source>In Proc. 8th ISWC ontology matching workshop (OM)</source>
          ,
          <source>Sydney (AU)</source>
          , page this volume,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          25.
          <string-name>
            <given-names>Dominique</given-names>
            <surname>Ritze</surname>
          </string-name>
          and
          <string-name>
            <given-names>Heiko</given-names>
            <surname>Paulheim</surname>
          </string-name>
          .
          <article-title>Towards an automatic parameterization of ontology matching tools based on example mappings</article-title>
          .
          <source>In Proc. 6th ISWC ontology matching workshop (OM)</source>
          ,
          <source>Bonn (DE)</source>
          , pages
          <fpage>37</fpage>
          -
          <lpage>48</lpage>
          ,
          <year>2011</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          26.
          <string-name>
            <surname>Emanuel</surname>
            <given-names>Santos</given-names>
          </string-name>
          , Daniel Faria, Catia Pesquita, and Francisco Couto.
          <article-title>Ontology alignment repair through modularization and confidence-based heuristics</article-title>
          .
          <source>CoRR, abs/1307.5322</source>
          ,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          27.
          <article-title>Pavel Shvaiko and Je´roˆ me Euzenat</article-title>
          .
          <article-title>Ontology matching: state of the art and future challenges</article-title>
          .
          <source>IEEE Transactions on Knowledge and Data Engineering</source>
          ,
          <volume>25</volume>
          (
          <issue>1</issue>
          ):
          <fpage>158</fpage>
          -
          <lpage>176</lpage>
          ,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          28. York Sure, Oscar Corcho, Je´roˆ me Euzenat, and Todd Hughes, editors.
          <source>Proc. ISWC Workshop on Evaluation of Ontology-based Tools (EON)</source>
          ,
          <source>Hiroshima (JP)</source>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          29. Ca´ssia Trojahn dos Santos, Christian Meilicke, Je´r oˆme Euzenat, and
          <string-name>
            <given-names>Heiner</given-names>
            <surname>Stuckenschmidt</surname>
          </string-name>
          .
          <source>Automating OAEI campaigns (first report)</source>
          .
          <source>In Proc. ISWC Workshop on Evaluation of Semantic Technologies (iWEST)</source>
          ,
          <source>Shanghai (CN)</source>
          ,
          <year>2010</year>
          .
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>