<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>Overview of the VISCERAL Challenge at ISBI 2015</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Orcun Goksel</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Antonio Foncubierta-Rodr guez</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Oscar Alfonso Jimenez del Toro</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Henning Muller</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Georg Langs</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Marc-Andre Weber</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Bjoern Menze</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ivan Eggel</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Katharina Gruenberg</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Marianne Winterstein</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Markus Holzer</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Markus Krenn</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Georgios Kontokotsios</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sokratis Metallidis</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Roger Schaer</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Abdel Aziz Taha</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andras Jakab</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Tomas Salas Fernandez</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Allan Hanbury</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>; AQuAS</institution>
          ,
          <addr-line>Barcelona</addr-line>
          ,
          <country country="ES">Spain</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>; HES-SO Valais</institution>
          ,
          <addr-line>Sierre</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>; MUW</institution>
          ,
          <addr-line>Vienna</addr-line>
          ,
          <country country="AT">Austria</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>; TUM</institution>
          ,
          <addr-line>Munich</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>; TUWien</institution>
          ,
          <addr-line>Vienna</addr-line>
          ,
          <country country="AT">Austria</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>; University of Heidelberg</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>ETH Zurich</institution>
          ,
          <country country="CH">Switzerland</country>
        </aff>
      </contrib-group>
      <abstract>
        <p>This is an overview paper describing the data and evaluation scheme of the VISCERAL Segmentation Challenge at ISBI 2015. The challenge was organized on a cloud-based virtualmachine environment, where each participant could develop and submit their algorithms. The dataset contains up to 20 anatomical structures annotated in a training and a test set consisting of CT and MR images with and without contrast enhancement. The test-set is not accessible to participants, and the organizers run the virtual-machines with submitted segmentation methods on the test data. The results of the evaluation are then presented to the participant, who can opt to make it public on the challenge leaderboard displaying 20 segmentation quality metrics per-organ and permodality. Dice coe cient and mean-surface distance are presented herein as representative quality metrics. As a continuous evaluation platform, our segmentation challenge leaderboard will be open beyond the duration of the VISCERAL project.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>Introduction</title>
      <p>In this challenge, a set of annotated medical imaging data was provided to the participants, along
with a powerful complimentary cloud-computing instance (8-core CPU with 16GB RAM) where
participant algorithms can be developed and evaluated. The available data contains segmentations
of several di erent anatomical structures in di erent image modalities, e.g. C_T and MRI. Annotated
structures in the training and testing data corpus included the segmentations of left/right kidney,
spleen, liver, left/right lung, urinary bladder, rectus abdominis muscle, 1st lumbar vertebra,
pancreas, left/right psoas major muscle, gallbladder, sternum, aorta, trachea, left/right adrenal gland.</p>
      <p>As training, 20 volumes each were provided for four di erent image modalities and eld-of-views,
with and without contrast enhancement, which add up to 80 volumes in total. In each volume, up to
20 structures were segmented. The missing annotations are due to poor visibility of the structures
in certain image modalities or due to such structures being outside the eld-of-view. Accordingly, in
all 80 volumes, a total of 1295 structures are segmented. A breakdown of annotations per anatomy
can be seen in Figure 1.</p>
      <p>Participants did not need to segment all the structures involved in such data, but rather they
could attempt any single anatomical structure or a combination thereof. For instance, an
algorithm that could segment only some organs in some of the modalities was evaluated only in those
categories for which it outputted any results. Accordingly, our evaluation results were presented
in a per-anatomy, per-modality fashion depending on the attempted segmentation task/s by each
participating algorithm. This is, indeed, in line with the VISCERAL vision of creating a single,
large, and multi-purpose medical image dataset, on which di erent research groups can test their
speci c applications and solutions.</p>
      <p>Participants rst registered for a benchmark account at the VISCERAL registration website.
Among the options during the registration, they could request their choice of operating system
(Linux, Windows, etc) for the virtual machine (VM), in order to get access to the VM and the
data. Having signed the data usage agreement and uploaded it to the participant dashboard, they
could then access the VM for algorithm development and also use the training data accessible
therein. Participants could additionally download the training dataset via FTP for o ine training.</p>
      <p>Participants accordingly developed and installed their algorithms in the VM, while adapting and
testing them on the training data. They then prepared their executable on the VM according to the
input/output speci cations announced by us earlier in the Anatomy3 Guidelines for Participation,
and submitted their VMs (through "Submit VM" button in the online participant dashboard) for
evaluation on the test data. We subsequently ran their VM (and hence their algorithm) on the
test data, and computed the relevant metrics. This evaluation process could be performed several
times during the training phase, nevertheless, we limited submissions to once per week, in order to
prevent the participants \training on the test data". The participants received feedback from their
evaluations in a private leaderboard and had the option to make their results publicly available on
the online public leaderboard, which included the results considered in our benchmark results.
2</p>
    </sec>
    <sec id="sec-2">
      <title>Evaluation</title>
      <p>For the Anatomy3 benchmark, a di erent evaluation approach was implemented compared to the
previous Anatomy benchmarks [LMMH13, JdTGM+14]. For this benchmark, participants had the
opportunity to submit their algorithms several times, giving them the opportunity to improve their
algorithms prior to the nal evaluation analysis during ISBI 2015. They could also choose to make
any of their results from the test-set public at any time. To allow a continuous work ow with this
evaluation approach, the steps during the evaluation phase were automated to a large extent.</p>
      <p>The continuous evaluation approach included the following steps:
1. The participant registers for the challenge; lls, signs, and uploads the participant agreement.
2. The organizers provide the participant with a virtual machine (VM) from the VISCERAL
cloud infrastructure.
3. The participant implements a segmentation algorithm in the VM according to the benchmark
speci cations.
4. The VM is submitted by the participant using the participant dashboard.
5. The organizers isolate the VM to prevent the participant from accessing it during the evaluation
phase.
6. The participant executable is run in a batch-script to test if its output les correspond with
those expected by the evaluation routines.
7. If the previous step is successful, the evaluation proceeds for all the volumes in the test set.
8. Each generated output segmentation le is uploaded by the batch script to the cloud storage
reserved for that participant.
9. Once all the images in the test-set are processed by the participant executable, the output
segmentations are cleared from the VM, which is in turn returned to the participant.
10. The output segmentations uploaded in the cloud storage are then evaluated against the
groundtruth (manual annotations) and the results are presented in the participant dashboard.
11. The participant can then analyze and interpret the results of their submission, and choose to
make them public or not on the public leaderboard.
12. The participant is allowed to submit again for testing only after a minimum of one week from
their latest submission.
Detailed results from 20 metrics can be seen in the online leaderboard1, a snapshot of which
at the time of ISBI 2015 Anatomy3 challenge is shown in Gig. 2. Participant evaluation results
are summarized in tables 1 and 2, respectively for Dice coe cient and mean surface distance,
as commonly-used segmentation evaluation metrics. The former is an overlap metric, describing
how well an algorithm estimates target anatomical region. The latter is a surface distance metric,
summarizing the overall surface estimation error by a given algorithm. The participant row in
the tables contains the citation for the publication contribution within this Anatomy3 proceedings
Part II.</p>
      <p>In the Dice results table, the highest ranking methods per-modality per-organ are marked in
bold. Any other method within 0.01 (1%) Dice of this are also considered a winner (or a tie) due
to the insigni cance of the di erence. Dice values below a threshold are considered unsuccessful
segmentations, and thus are not declared as a winner { even though the reader should note that
depending on particular clinical application such results can potentially still be useful. This threshold
was selected as 0.6 Dice, coinciding with a gap in the reported participant results.</p>
      <p>The results corresponding to the same bold values in the Dice table are also marked in the
mean surface distance table, in order to facilitate comparison of the segmentation surface errors for
the best methods in terms of the Dice metric. For successfully segmented organs (de ned by the
empirical 0.6 Dice cuto ), both metrics agree on the results for all structures and modalities, except
for the rst lumbar vertebra in CT. The reader should note that the mean surface distances are
presented in voxels, therefore the values between modalities (e.g. MR-ce and CT) are not directly
comparable in the latter table.</p>
      <p>According to these tables, there are di erent algorithms performing well for di erent anatomy.
In contrast-enhanced MR modality, we had only a single participant, Heinrich et al., potentially
1 The leaderboard is accessible at http://visceral.eu:8080/register/Leaderboard.xhtml
due to the di culty of automatic segmentations in this modality. This group thus became the
unchallenged winner of MRce for the structures they participated in. Note that the surface
error results are reported in voxels, where MRce has a signi cantly lower resolution than the other
modalities. In CTce, He et al. performed the best for the 6 structures they participated in, with
some ties with Jimenez et al. The latter group segmented all the given structures in CTce, some
of them with satisfactory accuracy, while for the others with potentially unusable results. We
had the most participants for the CT modality, in which the lungs {a relatively easier
segmentation problem{ were segmented successfully by most participants; potentially close to the accuracy
of inter-subject annotations. For most other structures for which successful segmentations were
achieved in CT, Kahl et al.were the winner of the challenge. Nevertheless, for structures where lower
delity segmentations (below the 0.6 Dice cuto ) were attained, Jimenez et al.are seen to provide
better segmentations estimations; likely due to their segmentation approach being atlas-based. It
is also observed that, despite the relatively good contrast of CT, several structures (prominently
the pancreas, gallbladder, thyroid, and adrenal glands) are still quite challenging to segment from
CT | potentially due to the lower sensitivity of CT to those structures also complicated by the
di cult-to-generalize shapes of these anatomies.
4</p>
    </sec>
    <sec id="sec-3">
      <title>Conclusions</title>
      <p>The VISCERAL Anatomy3 Challenge had a total of 23 virtual machines allocated for participants
at a time, although not all participants ultimately submitted results for the challenge. Most
participants relied on atlas-based segmentation methods, although there were also techniques that
use anatomy-based reasoning and locational relations. By using an online leaderboard evaluation
method, more participants are expected to submit results for our Anatomy3 challenge in the future.
5</p>
    </sec>
    <sec id="sec-4">
      <title>Acknowledgments</title>
      <p>The research leading to these results has received funding from the European Union Seventh
Framework Programme (FP7/2007-2014) under grant agreement n 318068 VISCERAL.
[LMMH13]</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          <source>[JdTGM+14] Oscar Alfonso Jimenez del Toro</source>
          , Orcun Goksel, Bjoern Menze, Henning Muller, Georg Langs,
          <string-name>
            <surname>Marc-Andre</surname>
            <given-names>Weber</given-names>
          </string-name>
          , Ivan Eggel, Katharina Gruenberg, Markus Holzer, Georgios Kotsios-Kontokotsios, Markus Krenn, Roger Schaer, Abdel Aziz Taha, Marianne Winterstein, and
          <string-name>
            <given-names>Allan</given-names>
            <surname>Hanbury</surname>
          </string-name>
          . VISCERAL {
          <article-title>VISual Concept Extraction challenge in RAdioLogy: ISBI 2014 challenge organization</article-title>
          . In Orcun Goksel, editor,
          <source>Proceedings of the VISCERAL Challenge at ISBI, number 1194 in CEUR Workshop Proceedings</source>
          , pages
          <volume>6</volume>
          {
          <fpage>15</fpage>
          , Beijing, China, May
          <year>2014</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          <string-name>
            <given-names>Georg</given-names>
            <surname>Langs</surname>
          </string-name>
          , Henning Muller,
          <string-name>
            <surname>Bjoern H. Menze</surname>
            , and
            <given-names>Allan</given-names>
          </string-name>
          <string-name>
            <surname>Hanbury</surname>
          </string-name>
          . Visceral:
          <article-title>Towards large data in medical imaging { challenges and directions</article-title>
          .
          <source>Lecture Notes in Computer Science</source>
          ,
          <volume>7723</volume>
          :
          <fpage>92</fpage>
          {
          <fpage>98</fpage>
          ,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>