<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>Overview of the ImageCLEF 2007 Ob ject Retrieval Task</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Thomas DeselaersRWTH</string-name>
          <email>deselaers@cs.rwth-aachen.de</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Allan HanburyPRIP</string-name>
          <email>hanbury@prip.tuwien.ac.at</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ville ViitaniemiHUT</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andr´as Benczu´rBUDAC</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>M´aty´as BrendelBUDAC</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>B´alint Dar´oczyBUDAC</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Hugo Jair Escalante BalderasINAOE</string-name>
          <email>hugojair@ccc.inaoep.mx</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Theo GeversISLA</string-name>
          <email>gevers@science.uva.nl</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Carlos Arturo Hern´andez GracidasINAOE</string-name>
          <email>carloshg@ccc.inaoep.mx</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Steven C. H. HoiNTU</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jorma LaaksonenHUT</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mingjing LiMSRA</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Heidy Marisol Marin CastroINAOE</string-name>
          <email>hmarinc@ccc.inaoep.mx</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Hermann NeyRWTH Xiaoguang RuiMSRA</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Nicu SebeISLA</string-name>
          <email>nicu@science.uva.nl</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Julian St¨ottingerPRIP</string-name>
          <email>julian@prip.tuwien.ac.at</email>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lei WuMSRA</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Authors: Steven C. H. Hoi Affiliation: School of Computer Engineering, Nanyang Technological University</institution>
          ,
          <country country="SG">Singapore</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Computer and Automation Research Institute of the Hungarian Academy of Sciences</institution>
          ,
          <addr-line>Budapest</addr-line>
          ,
          <country country="HU">Hungary</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>National Institute of Astrophysics</institution>
          ,
          <addr-line>Optics and Electronics, Tonantzintla</addr-line>
          ,
          <country country="MX">Mexico</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Vienna University of Technology</institution>
          ,
          <addr-line>Vienna</addr-line>
          ,
          <country country="AT">Austria</country>
        </aff>
      </contrib-group>
      <abstract>
        <p />
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>-</title>
      <p>We describe the object retrieval task of ImageCLEF 2007, give an overview of the
methods of the participating groups, and present and discuss the results.</p>
      <p>The task was based on the widely used PASCAL object recognition data to train
object recognition methods and on the IAPR TC-12 benchmark dataset from which
images of objects of the ten different classes bicycles, buses, cars, motorbikes, cats,
cows, dogs, horses, sheep, and persons had to be retrieved.</p>
      <p>Seven international groups participated using a wide variety of methods. The results
of the evaluation show that the task was very challenging and that different methods
for relevance assessment can have a strong influence on the results of an evaluation.</p>
    </sec>
    <sec id="sec-2">
      <title>Categories and Subject Descriptors</title>
      <p>H.3 [Information Storage and Retrieval]: H.3.1 Content Analysis and Indexing; H.3.3
Information Search and Retrieval; H.3.4 Systems and Software; H.3.7 Digital Libraries; H.2.3 [Database
Management]: Languages—Query Languages</p>
    </sec>
    <sec id="sec-3">
      <title>General Terms</title>
      <sec id="sec-3-1">
        <title>Measurement, Performance, Experimentation</title>
      </sec>
      <sec id="sec-3-2">
        <title>Image retrieval, image classification, performance evaluation</title>
        <p>1</p>
        <sec id="sec-3-2-1">
          <title>Introduction</title>
          <p>Object class recognition, automatic image annotation, and object retrieval are strongly related
tasks. In object recognition, the aim is to identify whether a certain object is contained in an
image, in automatic image annotation, the aim is to create a textual description of a given image,
and in object retrieval, images containing certain objects or object classes have to be retrieved out
of a large set of images. Each of these techniques are important to allow for semantic retrieval
from image collections.</p>
          <p>Over the last year, research in these areas has strongly grown, and it is becoming clear that
performance evaluation is a very important step to foster progress in research. Several initiatives
create benchmark suites and databases to compare different methods tackling the same problem
quantitatively.</p>
          <p>
            In the last years, evaluation campaign for object detection [
            <xref ref-type="bibr" rid="ref10 ref9">10, 9</xref>
            ], content-based image
retrieval [
            <xref ref-type="bibr" rid="ref5">5</xref>
            ] and image classification [
            <xref ref-type="bibr" rid="ref24">24</xref>
            ] have developed. There is however, no task aiming at
finding images showing particular object from a larger database. Although this task is extremely
similar to the PASCAL visual object classes challenge [
            <xref ref-type="bibr" rid="ref10 ref9">10, 9</xref>
            ], it is not the same. In the PASCAL
object recognition challenge, the probability for an object to be contained in an image is relatively
high and the images to train and test the methods are from the same data collection. In
realistic scenarios, this is not a suitable assumption. Therefore, in the object retrieval task described
here, we use the training data that was carefully assembled by the PASCAL NoE with much
manual work, and the IAPR TC-12 database which has been created under completely different
circumstances as the database from which relevant images are to be retrieved.
          </p>
          <p>In this paper, we present the results of the object retrieval task that was arranged as part
of the CLEF/ImageCLEF 2007 image retrieval evaluation. This task was conceived as a purely
visual task, making it inherently cross-lingual. Once one has a model for the visual appearance of
a specific object, such as a bicycle, it can be used to find images of bicycles independently of the
language or quality of the annotation of an image.</p>
          <p>
            ImageCLEF1 [
            <xref ref-type="bibr" rid="ref5">5</xref>
            ] has started within CLEF2 (Cross Language Evaluation Forum) in 2003. A
medical image retrieval task was added in 2004 to explore domain–specific multilingual information
retrieval and also multi-modal retrieval by combining visual and textual features for retrieval.
Since 2005, a medical retrieval and a medical image annotation task are both presented as part
of ImageCLEF. In 2006, a general object recognition task was presented to see whether interest
in this area existed. Although, only a few groups participated, many groups expressed their
interest and encouraged us to create an object retrieval task. In 2007, beside the here described
object retrieval task, a photographic retrieval task also using the IAPR TC-12 database [
            <xref ref-type="bibr" rid="ref14">14</xref>
            ], a
medical image retrieval task [
            <xref ref-type="bibr" rid="ref26">26</xref>
            ], and a medical automatic annotation task [
            <xref ref-type="bibr" rid="ref26">26</xref>
            ] were arranged in
ImageCLEF 2007.
          </p>
          <p>1http://ir.shef.ac.uk/imageclef/
2http://www.clef-campaign.org/
The task was defined as a visual object retrieval task. Training data was in the form of annotated
example images of ten object classes (PASCAL VOC 2006 data). The task was, after learning
from the provided annotated images, to find all images in the IAPR-TC12 database containing
the learned objects. The particularity of the task is that the training and test images are not from
the same set of images. This makes the task more realistic, but also more challenging.
2.1</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>Datasets</title>
      <p>For this task, two datasets were available. As training data, the organisers of the PASCAL
Network of Excellence visual object classes challenge kindly agreed that we use the training data
they assembled for their 2006 challenge.</p>
      <p>PASCAL VOC 2006 training data The PASCAL Visual Object Classes (VOC) challenge
2006 training data is freely available on the PASCAL web-page3 and consists of approximately
2600 images, where for each image a detailed description is available which of the ten object classes
is visible in which area of the image. Example images from this database are shown in Figure 1
with the corresponding annotation.</p>
      <p>
        IAPR TC-12 dataset The IAPR TC-12 Benchmark database [
        <xref ref-type="bibr" rid="ref15">15</xref>
        ] consists of 20,000 still images
taken from locations around the world and comprising an assorted cross-section of still images
which might for example be found in a personal photo collection. It includes pictures of different
sports and actions, photographs of people, animals, cities, landscapes and many other aspects
of contemporary life. Some example images are shown in Figure 2. This data is also strongly
annotated using textual descriptions of the images and various meta-data. We use only the image
data for this task.
2.2
      </p>
    </sec>
    <sec id="sec-5">
      <title>Object Retrieval Task</title>
      <p>The ten queries correspond to the ten classes of the PASCAL VOC 2006 data: bicycles, buses,
cars, motorbikes, cats, cows, dogs, horses, sheep, and persons. For training, only the “train” and
3http://www.pascal-network.org/challenges/VOC/
“val” sections of the PASCAL VOC database were to be used. For each query, participants were
asked to submit a list of 1000 images obtained by their method from the IAPR-TC12 database,
ranked in the order of best to worst satisfaction of the query.
2.3</p>
    </sec>
    <sec id="sec-6">
      <title>Evaluation Measure</title>
      <p>
        To evaluate the retrieval performance we use the same measure used by most retrieval evaluations
such as the other tasks in CLEF/ImageCLEF [
        <xref ref-type="bibr" rid="ref14 ref26">14, 26</xref>
        ], TREC4 and TRECVid5. The avereage
precision (AP) gives an indicate for the retrieval quality for one topic and the mean average
precision (MAP) provides a single-figure measure of quality across recall levels averaged over all
queries. To calculate these measures, it of course necessary to judge which images are relevant for
a given query and which are not.
2.4
      </p>
    </sec>
    <sec id="sec-7">
      <title>Relevance Assessments</title>
      <p>
        To find relevant images, we created pools per topic [
        <xref ref-type="bibr" rid="ref1">1</xref>
        ] keeping the top 100 results from all submitted
runs resulting in 1,507 images to be judged per topic on average. This resulted in a total of 15,007
images to be assessed. The normal relevance judgement process in information retrieval tasks
envisages that several users judge each document in question for relevance and that for each
image relevance for the particular query is judged. In this case, to judge the relevance is easy
enough that we can postulate that every two persons among the judges would come to the same
conclusion and therefore each image was judged by only one judge. Furthermore, since only
10 queries were to be judged and the concepts to find are simple, the judges complained about
the task being to easy and boring. Therefore, after approximately 3000 images were judged, we
allowed the judges to additionally specify whether an image is relevant with respect to any of
the other queries. The whole judgement process was performed over a web interface which was
quickly created and everybody from the RWTH Aachen University Human Language Technology
and Pattern Recognition Group and from the Vienna University of Technology PRIP group was
invited to judge images. Thus, most of the judges are computer science students and researchers
with a Human Language Technology background. Note, that in the pooling process all images
that are not judged are automatically considered to be not relevant.
      </p>
      <p>The web-interface is shown in Figure 3 to give an impression of the process. On each page,
10 images are shown, and the judge has to decide whether a particular object is present in these
images or not. To reduce boredom for the judges, they are allowed (and recommended) to specify
4http://trec.nist.gov/
5http://www-nlpir.nist.gov/projects/t01v/
class class
id
whether other object classes are present in the images. The judges were told to be rather positive
about the relevance of an image, e.g. to consider sheep-like animals such as llamas to be sheep
and to consider tigers and other non-domestic cats to be cats.</p>
      <p>Furthermore, Ville Viitaniemi from the HUTCIS group, judged all 20,000 images with respect
to relevance for all of the topics with a stricter definition of relevances.</p>
      <p>Results from the Relevance Judgements Table 1 gives an overview how many images were
found to be relevant for each of the given topics. It can be observed that there are far more
relevant images for the person topic than for any other topic. From these numbers it can be seen
that the task at hand is really challenging for most of the classes and very similar to the proverbial
looking for a needle in a haystack. In particular for the sheep topic, only 0.03% of the images in
the database are relevant although some more images were judged to be relevant by the judges
in the additional relevance information. If only the data from the conventional pooling process
is considered for eight of the ten classes less than a thousandth of the images are relevant. The
discrepancy in the results with additional relevance judgements and the results with full relevance
judgement are due to different relevance criteria. The judges that created the additional relevance
information were instructed to judge images as relevant that show sheep-like animals such as
llamas, and to judge tigers as cats, where the full relevance judgement was stricter in this respect.
Table 1 also shows that the additional relevance judgements found more cats and sheep than are
actually in the database.
3</p>
      <sec id="sec-7-1">
        <title>Methods</title>
        <p>Seven international groups from academia participated in the task and submitted a total of 38
runs. The group with the highest number of submissions had 13 submissions. In the following
sections, the methods of the groups are explained (in alphabetical order) and references to further
work are given.
3.1</p>
      </sec>
    </sec>
    <sec id="sec-8">
      <title>Budapest methods</title>
      <sec id="sec-8-1">
        <title>Authors: M´aty´as Brendel, B´alint Dar´oczy, and Andr´as Benczu´r Affiliation: Data Mining and Web search Research Group, Informatics Laboratory Computer and Automation Research Institute of the Hungarian Academy of Sciences</title>
        <p>
          Email: {mbrendel, daroczyb, benczur}@ilab.sztaki.hu
The task of object retrieval is to classify objects found in images. This means to find an objects in
an image, which is similar to sample objects in the pre-classified images. There are two problems
with this task: the first is, how do we model objects. The second is, how do we measure similarity
of objects. Our first answer to the first question is to model objects with image segments. Segment,
region or blob based image similarity is a common method in content based image retrieval, see for
example [
          <xref ref-type="bibr" rid="ref2 ref21 ref27 ref4">4, 27, 2, 21</xref>
          ]. Respectively, the basis of our first method is to find segments on the query
image, which are similar to the objects in the pre-classified images. The image is then classified
to be in that class, to which we find the most similar segment in the query image.
        </p>
        <p>
          Image segmentation in itself is a widely researched and open problem itself. We used an image
segmenter developed by our group to extract segments from the query images. Our method is
based on a graph-based algorithm developed by Felzenszwalb and Huttenlocher [
          <xref ref-type="bibr" rid="ref11">11</xref>
          ]. We
implemented a pre-segmentation method to reduce the computational time and use a different smoothing
technique. All images were sized to a fixed resolution. Gaussian-based smoothing helped us cut
down high frequency noise. Because of the efficiency of OpenCV6 implementation we did not
implement either a resizing or a Gaussian-based smoothing algorithm. As pre-segmentation we built
a three-level Gaussian-Laplacian pyramid to define initial pixel groups. The original
pyramidbased method, which considers the connection between pixels on different levels too, was modified
to eliminate the so-called blocking problem. We used brightness difference to measure distance
between pixels:
dif f Y (P1, P2) = 0.3∗ | RP2 − RP1 | +0.59∗ | GP2 − GP1 | +0.11∗ | BP2 − BP1 |
(1)
        </p>
        <p>
          After pre-segmentation, we had segments of 16 pixels maximum. To detect complex segments,
we modified the original graph-based method by Felzenszwalb and Huttenlocher [
          <xref ref-type="bibr" rid="ref11">11</xref>
          ] with an
adaptive threshold system using Euclidean distance to prefer larger regions instead of small regions
of the image. Felzenszwalb and Huttenlocher defined an undirected graph G = (V, E) where
∀vi ∈ V corresponds to a pixel in the image, and the edges in E connect certain pairs of neighboring
pixels. This graph-based representation of the image reduces the original proposition into a graph
cutting challenge. They made a very efficient and linear algorithm that yields a result near to the
optimal normalized cut which is one of the NP-full graph problems [
          <xref ref-type="bibr" rid="ref11 ref29">11, 29</xref>
          ].
        </p>
        <p>6http://www.intel.com/technology/computing/opencv/
Algorithm 1 Algorithm Segmentation (Isrc, τ1, τ2)
τ1 and τ2 are threshold functions. Let I2 be the source image, I1 and I0 are the down-scaled
images. Let P (x, y, i) be the pixel P (x, y) in the image on level i (Ii). Let G = (V, E) be an
undirected weighted graph where ∀vi ∈ V corresponds to a pixel P (x, y). Each edge (vi, vj ) has a
non-negative weight w(vi, vj ).</p>
      </sec>
      <sec id="sec-8-2">
        <title>Gaussian-Laplacian Pyramid</title>
      </sec>
      <sec id="sec-8-3">
        <title>Graph-based Segmentation</title>
        <p>1. Compute M axweight(R) = maxe∈MST (R,E) w(e) for every coherent group of points R where</p>
        <p>M ST (R, E) is the minimal spanning tree
2. Compute Co(R) = τ2(R) + M axweight(R) as the measure of coherence between points in R
3. J oin(R1, R2) if e ∈ E exists so w(e) &lt; min(Co(R1), Co(R2)) is true, where R1 T R2 = ∅
and w(e) is the weight of the border edge e between R1 and R2
4. Repeat steps 1,2,3 for every neighboring group (R1, R2) until possible to join two groups
Algorithm 1 sometimes does not find relevant parts with low initial thresholds. To find the
relevant borders which would disappear with the graph-based method using high thresholds we
calculated the Sobel-gradient image to separate important edges from other remainders.</p>
        <p>Similarity of complex objects is usually measured on a feature base. This means the the
similarity of the objects is defined by the similarity in a certain feature-space.</p>
        <p>dist(Si, Oj ) = d(F (Si), F (Oj )) : Si ∈ S, Oj ∈ O
where S is the set of segments and O is the set of objects, dist is the distance function of the
objects and segments, d is a distance function in the feature space (usually some of the conventional
metrics in the n-dimensional real space), F is the function which assigns features to objects and
segments. We extracted from the segments features, like mean color, size, shape information,
and histogram information. As shape information a 4 × 4 sized low-resolution variant of the
segment (framed in a rectangle with background) was used. Our histograms had 5 bins in each
channel. Altogether a 35 dimensional, real valued feature-vector was extracted for each of the
segments. The same features were extracted for the objects in the pre-classified images taking
them as segments. The background and those classes, which were not requested were ignored.
The features of the objects were written to a file, with the class-identifiers, which were extracted
from the color-coding. This way we obtained a data-base of class samples, containing features
of objects belonging to the classes. After this, the comparison of the objects of the pre-classified
sample-images and the segments of the query image was possible. We used Euclidean distance to
measure similarity. The distance of the query-image Q was computed as:
dist(Q) = min dist(Si, Oj ) : Si ∈ S, Oj ∈ O</p>
        <p>i,j
where S is the set of segments of image Q, O is the set of the pre-classified sample objects. Q is
classified to be in the class of that object, which minimizes the distance. The score of an image
was computed as:</p>
        <p>score(Q) = 1000/dist(Q)
where Q is the query image.
(2)
(3)
(4)
In our first method (see budapest-acad315) we found that our segments are much smaller than
the objects in the pre-segmented images. It would have been possible to get larger segments
by adjusting the segmentation algorithm, however this way we would not get objects, which
were really similar to the objects. We found that our segmentation algorithm could not generate
segments similar to the the objects in the pre-classified images with any settings of the parameters.
Even, if we tried our algorithm on the sample-images, and the segments were approximately of
the same size, the segments did not match the pre-classified objects. The reason for this is that
pre-segmentation was made by humans and algorithmic segmentation is far from capable of the
same result. For example, it is almost impossible to write an algorithm, which would segment a
shape of a human being as one segment, if his clothes are different. However, people were one of
the classes defined, and the sample images contained people with the entire body as one object.
Therefore we modified our method. Our second method is still segment-based. But we also do a
segmentation on the sample-images. We took the segmented sample-images, and if a segment was
80% inside of an area of a pre-defined object, then we took this segment as a proper sample for
that object. This way a set of sample segments were created. After this the method is similar to
the previous, the difference is only that we have sample-segments instead of sample objects, but
we treat them the same way anyway. The features of the segments were extracted and they were
written to a file, with the identifier of the class, which was extracted from the color-codes. After
this, the comparison of the segments of the pre-classified images and the query image was possible.
We used Euclidean distance again to measure similarity. The closest segment of the image to a
segment in any of the objects was searched.</p>
        <p>dist(Q) = min dist(Si, Sj ) : Si ∈ S, Sj ∈ O
i,j
(5)
where S is the segments of image Q, O is the set of segments belonging to the pre-classified objects.
The image was classified according to the object, to which the closest segment belongs. As we
expected, this modification made the algorithm better.
3.2</p>
        <p>HUTCIS: Conventional Supervised Learning using Fusion of Image
Features</p>
      </sec>
      <sec id="sec-8-4">
        <title>Authors: Ville Viitaniemi, Jorma Laaksonen Affiliation: Adaptive Informatics Research Centre, Laboratory of Computer and Information Science, Helsinki University of Technology, Finland</title>
        <p>Email: firstname.lastname@tkk.fi
All our 13 runs identified with prefix HUTCIS implement a similar general system architecture
with three system stages:
1. Extraction of a large number of global and semi-global image features. Here we interpret
global histograms of local descriptors as one type of global image feature.
2. For each individual feature, conventional supervised classification of the test images using
the VOC2006 trainval images as the training set.</p>
      </sec>
      <sec id="sec-8-5">
        <title>3. Fusion of the feature-wise classifier outputs.</title>
        <p>
          By using this architecture, we knowingly ignored the aspect of qualitatively different training and
test data. The motivation was to provide a baseline performance level that could be achieved by
just applying a well-working implementation of the conventional supervised learning approach.
Table 2 with ROC AUC performances in VOC 2006 test set reveals that the performance of
our principal run HUTCIS SVM FULLIMG ALL is relatively close to the best performances in
the last year’s VOC evaluation [
          <xref ref-type="bibr" rid="ref9">9</xref>
          ]. The last row of the table indicates what the rank of the
HUTCIS SVM FULLIMG ALL run would have been among the 19 VOC 2006 participants.
        </p>
        <p>
          The following briefly describes the components of the architecture. For more detailed
description, see e.g. [
          <xref ref-type="bibr" rid="ref34">34</xref>
          ].
        </p>
        <p>
          Features For different runs, the features are chosen from a set of feature vectors, each with
several components. Table 3 lists 10 of the features. Additionally, the available feature set includes
interest point SIFT feature histograms with different histogram sizes, and concatenations of pairs,
triples and quadruples of the tabulated basic feature vectors. The SIFT histogram bins have been
selected by clustering part of the images with with self-organising map (SOM) algorithm.
Classification and fusion The classification is performed either by a C-SVC implementation
built around LIBSVM support vector machine (SVM) library [
          <xref ref-type="bibr" rid="ref3">3</xref>
          ], or a SOM-based classifier [
          <xref ref-type="bibr" rid="ref19">19</xref>
          ].
The SVM classifiers (prefix HUTCIS SVM) are fused together using an additional SVM layer. For
the SOM classifiers (prefix HUTCIS PICSOM), the fusion is based on summation of normalized
classifier outputs.
        </p>
        <p>The different runs Our principal run HUTCIS SVM FULLIMG ALL realizes all the three
system stages in the best way possible. Other runs use subsets of those image features, inferior
algorithms or are otherwise predicted to be suboptimal.</p>
        <p>The run HUTCIS SVM FULLIMG ALL performs SVM-classification with all the tabulated
features, SIFT histograms and and twelve previously hand-picked concatenations of the tabulated
features, selected on basis of SOM classifier performance in the VOC2006 task. The runs
HUTCIS SVM FULLIMG IP+SC and HUTCIS SVM FULLIMG IP+SC are otherwise similar but use
just subsets of the features: SIFT histograms and color histogram, or just SIFT histograms,
respectively.</p>
        <p>The runs identified by prefix HUTCIS SVM BB are half-baked attempts to account for the
different training and test image distributions. These runs are also based on SIFT histogram and
color histogram features. For the training images, the features are calculated from the bounding
boxes specified in VOC2006 annotations. For the test images, the features are calculated for whole
images. The different runs with this prefix correspond to different ways to select the images as a
basis for SIFT codebook formation.</p>
        <p>The run HUTCIS FULLIMG+BB is the rank based fusion of features extracted from full
images and bounding boxes. The runs HUTCIS PICSOM1 and HUTCIS PICSOM2 are otherwise
identical but use a different setting of SOM classifier parameter. HUTCIS PICSOM2 smooths less
Run id.</p>
        <p>FULLIMG ALL
FULLIMG IP+SC
FULLIMG IP
Best in VOC2006
Rank
the feature spaces, the detection is based on more local information. Both the runs are based on
the full set of features mentioned above. concatenations of pairs, triples and
Results The tabulated MAP results of HUTCIS runs are corrupted by our own slip in ordering
of the queries that prevented us from fully participating the pool selection for query topics 4–10.
For the normal pooling, our mistake excluded us completely, for additional relevance judgments
only partly. This is expected to degrade our results quite much for the normal pooling and
somewhat less for the additional pool, as participating the pool selection is known to be essential
for obtaining good performance figures.</p>
        <p>As expected, HUTCIS SVM FULLIMG ALL turned out to be the best of our runs for the
uncorrupted query topics 1–3. The idea of extracting features only from bounding boxes of objects
in order to reduce the effect of different backgrounds in the training and test images seemed
usually not to work as well as features extracted from full images, although results were not very
conclusive. Partly this is could be to the asymmetry in feature extraction: the features extracted
from bounding boxes of the training objects were compared with the features of whole of the
test images. The results of the SOM classifier runs did not provide information that would be of
general interest, besides confirming the previously known result of SOM classifiers being inferior
to SVMs.
3.3</p>
      </sec>
    </sec>
    <sec id="sec-9">
      <title>INAOE’s Annotation-based object retrieval approaches</title>
      <sec id="sec-9-1">
        <title>Authors: Heidy Marisol Marin Castro, Hugo Jair Escalante Balderas, and Carlos Arturo Hern´andez Gracidas Affiliation: TIA Research Group, Computer Science Department, National Institute of Astrophysics, Optics and Electronics, Tonantzintla, Mexico</title>
        <p>
          Email: {hmarinc,hugojair,carloshg}@ccc.inaoep.mx
The TIA research group at INAOE, Mexico proposed two methods based on image labeling.
Automatic image annotation methods were used for labeling regions within segmented images, and
then we performed object retrieval based on the generated annotations. Results were not what
we expected, though it can be due to the fact that annotations were defined subjectively and that
not enough images were annotated for creating the training set for the annotation method. Two
approaches were proposed: a semi-supervised classifier based on unlabeled data and a supervised
one, the last method was enhanced with a recent proposed method based on semantic cohesion
[
          <xref ref-type="bibr" rid="ref8">8</xref>
          ]. Both approaches followed the following steps:
1. Image segmentation
2. Feature extraction
3. Manual labeling of a small subset of the training set
4. Training a classifier
5. Using the classifier for labeling the test-images
6. Using labels assigned to regions images for object retrieval
For both approaches the full collection of images was segmented with the normalized cuts algorithm
[
          <xref ref-type="bibr" rid="ref30">30</xref>
          ]. A set of 30 features were extracted from each region; we considered color, shape and texture
attributes. We used our own tools for image segmentation, feature extraction and manual labeling
[
          <xref ref-type="bibr" rid="ref22">22</xref>
          ]. The considered annotations were the labels of the 10 objects defined for this task. The features
for each region together with the manual annotations for each region were used as training set
with the two approaches proposed. Each classifier was trained with this dataset and then all of the
test images were annotated with such a classifier. Finally, the generated annotations were used
for retrieving objects with queries. Queries were created using the labels of the objects defined
for this task; and selected as relevant those images with the highest number of regions annotated
with the object label. Sample segmented images with their corresponding manual annotations are
shown in Figure 4. As we can see the segmentation algorithm works well for some images (isolated
cows, close-up of people), however for other objects segmentation is poor (a bicycle, for example).
KNN+MRFI, A supervised approach For the supervised approach we used a simple knn
classifier for automatically labeling regions. Euclidean distance was used as similarity function.
The label of the nearest neighbor (in the training set) for each test-region was assigned as
annotation for this region. This was our baseline run (INAOE-TIA-INAOE-RB-KNN ).
        </p>
        <p>
          The next step consisted of improving annotation performance of knn using an approach called
(MRFI ) [
          <xref ref-type="bibr" rid="ref8">8</xref>
          ] which we recently proposed for improving annotation systems. This approach consists
of modeling each image (region-annotations pairs) with a Markov random field (MRF ),
introducing semantic knowledge, see Figure 5. The top−k more likely annotations for each region are
considered. Each of these annotations have a confidence weight related to the relevance of the
label to being the correct annotation for that region, according to knn. The MRFI approach uses
the relevance weights with semantic information for choosing a unique (the correct) label for each
region. Semantic information is considered in the MRF for keeping coherence among annotations
assigned to regions within a common image; while the relevance weight is considered for taking
into account the confidence of the annotation method (k − nn) on each of the labels, see Figure
5. The (pseudo) optimal configuration of regions-annotations for each image is obtained by
minimizing an energy function defined by potentials. For optimization we used standard simulated
annealing.
        </p>
        <p>
          The intuitive idea of the MRFI approach is to guarantee that the labels assigned to regions
are coherent among them, taking into account semantic knowledge and the confidence of the
annotation system. In previous work, semantic information was obtained from cooccurrences of
labels on an external corpus. However for this work semantic association between a pair of labels is
given by the normalized number of relevant documents returned by GoogleR to queries generated
using the pair of labels. This run is named INAOE-TIA-INAOE-RB-KNN+MRFI, see [
          <xref ref-type="bibr" rid="ref8">8</xref>
          ] for
details.
        </p>
        <p>
          SSAssemble: Semi-supervised Weighted AdaBoost The semi-supervised approach consist
of using a recently proposed ensemble of classifiers, called WSA [
          <xref ref-type="bibr" rid="ref22">22</xref>
          ]. Our WSA ensemble uses
naive Bayes as its base classifier. A set of these is combined in a cascade based on the AdaBoost
technique [
          <xref ref-type="bibr" rid="ref13">13</xref>
          ]. Ensemble methods work by combining a set of base classifiers in some way, such
as a voting scheme, producing a combined classifier which usually outperforms a single classifier.
When training the ensemble of Bayesian classifiers, WSA considers the unlabeled images on each
stage. These are annotated based on the classifier from the previous stage, and then used to
train the next classifier. The unlabeled instances are weighted according to a confidence measure
based on their predicted probability value; while the labeled instances are weighted according to
the classifier error, as in standard AdaBoost. Our method is based on the supervised multi-class
AdaBoost ensemble, which has shown to be an efficient scheme to reduce the rate error of different
classifiers.
        </p>
        <p>Formally WSA algorithm receives a set of labeled data (L) and a set of unlabeled data (U ).
An initial classifier N B1 is build using L. The labels in L are used to evaluate the error of N B1.
As in AdaBoost the error is used to weight the examples, increasing the weight of the misclassified
examples and keeping the same weight of the correctly classified examples. The classifier is used
to predict a class for U with certain probability. In the case of U , the weights are multiplied by
the predicted probability of the majority class. Unlabeled examples with high probability of their
predicted class will have more influence in the construction of the next classifier than examples
with lower probabilities. The next classifier N B2 is build using the weights and predicted class
of L ∪ U . N B2 makes new predictions on U and the error of N B2 on all the examples is used to
re–weight the examples. This process continues, as in AdaBoost, for a predefined number of cycles
or when a classifier has a weighted error greater or equal to 0.5. As in AdaBoost, new instances
are classified using a weighted sum of the predicted class of all the constructed base classifiers.
WSA is described in algorithm 2.</p>
        <p>We faced several problems when performing the annotation image task. The first one was that
the training set and the test set were different, so this caused a classification with high error ratio.
The second one was due the segmentation algorithm. The automatic segmentation algorithm did
not perform well for all images leading to have incorrect segmentation of the objects in the images.
The last one concerns to the different criteria for manual labeling of the training set. Due all these
facts we did not get good results. We hope improving the annotation task by changing part of the
labeling strategy.
3.4</p>
      </sec>
    </sec>
    <sec id="sec-10">
      <title>MSRA: Object Retrieval</title>
      <sec id="sec-10-1">
        <title>Authors:</title>
      </sec>
      <sec id="sec-10-2">
        <title>Mingjing Li, Xiaoguang Rui, and Lei Wu Affiliation: Email: Microsoft Research Asia</title>
        <p>
          Two approaches were adopted by Microsoft Research Asia (MSRA) to perform the object retrieval
task in ImageCLEF 2007. One is based on the visual topic model (VTM); the other is the visual
language modelling (VLM) method [
          <xref ref-type="bibr" rid="ref35">35</xref>
          ]. VTM represents an image by a vector of probabilities
that the image belongs to a set of visual topics, and categorizes images using SVM classifiers. VLM
represents an image as a 2-D document consisting of visual words, trains a statistical language
model for each image category, and classifies an image to the category that generates the image
with the highest probability.
Probabilistic Latent Semantic Analysis (pLSA) [
          <xref ref-type="bibr" rid="ref12">12</xref>
          ], which is a generative model from the text
literature, is adopted to find out the visual topics from training images. Different from traditional
pLSA, all training images of 10 categories are put together in the training process and about 100
visual topics are discovered finally.
        </p>
        <p>The training process consists of five steps: local feature extraction, visual vocabulary
construction, visual topic construction, histogram computation, and classifier training. At first, salient
image regions are detected using scale invariant interest point detectors such as the Harris-Laplace
and the Laplacian detectors. For each image, about 1,000 to 2,000 salient regions are extracted.
Those regions are described by the SIFT descriptor which computes a gradient orientation
histogram within the support region. Next, 300 local descriptors are randomly selected from each
category and combined together to build a global vocabulary of 3,000 visual words. Based on the
vocabulary, images are represented by the frequency of visual words. Then, pLSA is performed to
discover the visual topics in the training images. pLSA is also applied to estimate how likely an
image belongs to each visual topic. The histogram of the estimated probabilities is taken as the
feature representation of that image for classification. For multi-class classification problem, we
adopt the one-against-one scheme, and train an SVM classifier with RBF kernel for each possible
pair of categories.
3.4.2</p>
        <p>VLM: Visual Language Modeling
The approach consists of three steps: image representation, visual language model training and
object retrieval. Each image is transformed into a matrix of visual words. First, an image is simply
segmented into 8 x 8 patches, and the texture histogram feature is extracted from each path. Then
all patches in the training set are grouped into 256 clusters based on their features. Next, each
path cluster is represented using an 8-bit hash code, which is defined as the visual word. Finally,
an image is represented by a matrix of visual words, which is called visual document.</p>
        <p>Visual words in a visual document are not independent to each other, but correlated with other
words. To simply the model training, we assume that visual words are generated in the order from
left to right, and top to bottom and each word is only conditionally dependent on its immediate
top and left neighbors, and train a trigram language model for each image category. Given a test
image, it is transformed into a matrix of visual words in the same way, and the probability that it
is generated by each category is estimated respectively. Finally, the image categories are ranked
in the descending order of these probabilities.
3.5</p>
      </sec>
    </sec>
    <sec id="sec-11">
      <title>NTU: Solution for the Object Retrieval Task</title>
      <p>
        Object retrieval is an interdisciplinary research problem between object recognition and
contentbased image retrieval (CBIR). It is commonly expected that object retrieval can be solved more
effectively with the joint maximization of CBIR and object recognition techniques. The goal of
this paper is to study a typical CBIR solution with application to the object retrieval tasks [
        <xref ref-type="bibr" rid="ref17 ref18">17, 18</xref>
        ].
We expected that the empirical study in this work will serve as a baseline for future research when
applying CBIR techniques for object recognition.
3.5.2
      </p>
      <p>Overview of Our Solution
We study a typical CBIR solution for the object retrieval problem. In our approach, we focus on
two key tasks. One is the feature representation, the other is the supervised learning scheme with
support vector machines.</p>
      <p>Feature Representation In our approach, three kinds of global features are extracted to
represent an image, including color, shape, and texture.</p>
      <p>For color, we study the Grid Color Moment feature (GCM). For each image, we split it into
3 × 3 equal grids and extract color moments to represent each of the 9 grids. Three color moments
are then computed: color mean, color variance and color skewness in each color channel (H, S,
and V), respectively. Thus, an 81-dimensional color moment is adopted as the color feature for
each image.</p>
      <p>For shape, we employ the edge direction histogram. First, an input color image is first converted
into a gray image. Then a Canny edge detector is applied to obtain its edge image. Based on the
edge images, the edge direction histogram can then be computed. Each edge direction histogram
is quantized into 36 bins of 10 degrees each. In addition, we use a bin to count the number of
pixels without edge. Hence, a 37-dimensional edge direction histogram is used for shape.</p>
      <p>For texture, we investigate the Gabor feature. Each image is first scaled to the size of 64 × 64.
Then, the Gabor wavelet transformation is applied to the scaled image at 5 scale levels and 8
orientations, which results in a total of 40 subimages for each input image. For each subimage,
we calculate three statistical moments to represent the texture, including mean, variance, and
skewness. Therefore, a 120-dimensional feature vector is used for texture.</p>
      <p>
        In total, a 238-dimensional feature vector is used to represent each image. The set of visual
features has been shown to be effective for content-based image retrieval in our previous
experiments [
        <xref ref-type="bibr" rid="ref17 ref18">17, 18</xref>
        ].
      </p>
      <p>
        Supervised Learning for Object Retrieval The object retrieval task defined in ImageCLEF
2007 is similar to a relevance feedback task in CBIR, in which a number of possible and negative
labeled examples are given for learning. This can be treated as a supervised classification task.
To solve it, we employ the support vector machines (SVM) technique for training the classifiers
on the given examples [
        <xref ref-type="bibr" rid="ref17">17</xref>
        ]. In our experiment, a standard SVM package is used to train the SVM
classifier with RBF kernels. The parameters C and γ are best tuned on the VOC2006 training
set, in which the training precision is 84.2% for the classification tasks. Finally, we apply the
trained classifiers to do the object retrieval by ranking the distances of the objects apart from the
classifier’s decision boundary.
3.5.3
      </p>
      <p>Concluding Remarks
We described our solution for the object retrieval task in the ImageCLEF 2007 by a typical CBIR
solution. We found that the current solution, though it was trained with good performance in an
object recognition test-bed, did not achieve promising results in the tough object retrieval tasks.
In our future work, several directions can be explored to improve the performance, including local
feature representation and better machine learning techniques.
3.6</p>
    </sec>
    <sec id="sec-12">
      <title>PRIP: Color Interest Points and SIFT features</title>
      <p>Authors: Julian St¨ottinger1, Allan Hanbury1, Nicu Sebe2, Theo Gevers2
Affiliation: 1 PRIP, Institute of Computer-Aided Automation,</p>
      <p>Vienna University of Technology, Vienna, Austria
2 Intelligent Systems Lab Amsterdam,</p>
      <p>University of Amsterdam, The Netherlands
Email: {julian,hanbury}@prip.tuwien.ac.at, {nicu,gevers}@science.uva.nl
In the field of retrieval, detection, recognition and classification of objects, many state of the art
methods use interest point detection at an early stage. This initial step typically aims to find
meaningful regions in which descriptors are calculated. Finding salient locations in image data is
crucial for these tasks. Most current methods use only the luminance information of the images.
This approach focuses on the use of color information in interest point detection and its gain in
performance. Based on the Harris corner detector, multi-channel visual information transformed
into different color spaces is the basis to extract the most salient interest points. To determine the
characteristic scale of an interest point, a global method of investigating the color information on a
global scope is used. The two different PRIP-PRIP ScIvHarris approaches differ in the properties
of these interest points only.</p>
      <p>The method consists of the following stages:
1. Extraction of multi-channel based interest points</p>
      <sec id="sec-12-1">
        <title>2. Local descriptions of interest points</title>
      </sec>
      <sec id="sec-12-2">
        <title>3. Estimating the signature of an image</title>
      </sec>
      <sec id="sec-12-3">
        <title>4. Classification</title>
        <p>
          Extraction of multi-channel based interest points An extension of the intensity-based
Harris detector [
          <xref ref-type="bibr" rid="ref16">16</xref>
          ] is proposed in [
          <xref ref-type="bibr" rid="ref25">25</xref>
          ]. Because of common photometric variations in imaging
conditions such as shading, shadows, specularities and object reflectance, the components of the
RGB color system are correlated and therefore sensitive to illumination changes. However, in
natural images, high contrast changes may appear. Therefore, a color Harris detector in RGB
space does not dramatically change the position of the corners compared to a luminance based
approach. Normalized rgb overcomes the correlation of RGB and favors color changes. The main
drawback, however, is its instability in dark regions. We can overcome this by using quasi invariant
color spaces.
        </p>
        <p>
          The approach PRIP-PRIP HSI ScIvHarris uses the HSI color space [
          <xref ref-type="bibr" rid="ref32">32</xref>
          ], which is
quasiinvariant to shadowing and specular effects. Therefore, changes in lighting conditions in images
should not affect the positions of the interest points, resulting in more stable locations.
Additionally, the HSI color space discriminates between luminance and color. Therefore, much information
can be discarded, and the locations get more sparse and distinct.
        </p>
        <p>
          The PRIP cbOCS ScIvHarris approach follows a different idea. As proposed in [
          <xref ref-type="bibr" rid="ref33">33</xref>
          ], colors
have different occurrence probabilities and therefore different information content. Therefore, rare
colors are regarded as more salient than common ones. We wish to find a boosting function so
that color vectors having equal information content have equal impact on the saliency function.
This transformation can be found analyzing the occurrence probabilities of colors in large image
databases. With this change of focus towards rare colors, we aim to discard many repetitive
locations and get more stable results on rare features.
        </p>
        <p>
          The characteristic scale of an interest point is chosen by applying a principal component
analysis (PCA) on the image and thus finding a description for the correlation of the multi-channel
information [
          <xref ref-type="bibr" rid="ref31">31</xref>
          ]. The characteristic scale is decided when the Laplacian of Gaussian function of
this projection and the Harris energy is a maximum at the same location in the image. The final
extraction of these interest points and corresponding scales is done by preferring locations with
high Harris energy and huge scales. A maximum number of 300 locations per image has been
defined, as over-description diminish the overall recognition ability dramatically.
Local descriptions of interest points The scale invariant feature transform (SIFT) [
          <xref ref-type="bibr" rid="ref20">20</xref>
          ]
showed to give best results in a broad variety of applications [
          <xref ref-type="bibr" rid="ref23">23</xref>
          ]. We used the areas of the
extracted interest points as a basis for the description phase. SIFT are basically sampled and
normalized gradient histograms, which can lead to multiple descriptions per location. This occurs
if there is more than one direction of the gradients regarded as predominant.
        </p>
        <p>
          Estimating the signature of an image In this bag of visual features approach [
          <xref ref-type="bibr" rid="ref36">36</xref>
          ], we cluster
the descriptions of one image to a fixed number of 40 clusters using k-means. The centroids and
the proportional sizes of the clusters build the signature of one image having a fixed dimensionality
of 40 by 129.
        </p>
        <p>
          Classification The Earth Mover’s Distance (EMD) [
          <xref ref-type="bibr" rid="ref28">28</xref>
          ] showed to be a suitable metric for
comparing image signatures. It takes the proportional sizes of the clusters into account, which
seems to gain a lot of discriminative power. The classification itself is done in the most straight
forward way possible: for every object category, the smallest distances to another signature build
the classification.
3.7
        </p>
      </sec>
    </sec>
    <sec id="sec-13">
      <title>RWTHi6: Patch-Histograms and Log-Linear Models</title>
      <sec id="sec-13-1">
        <title>Authors: Thomas Deselaers, Hermann Ney Affiliation: Human Language Technology and Pattern Recognition, RWTH Aachen University, Aachen, Germany</title>
        <p>Email: surname@cs.rwth-aachen.de
The approach used by the Human Language Technology and Pattern Recognition group of the
RWTH Aachen University, Aachen, Germany, to participate in the PASCAL Visual Object Classes
Challenge consists of four steps:</p>
      </sec>
      <sec id="sec-13-2">
        <title>1. patch extraction</title>
        <p>
          2. clustering
3. creation of histograms
4. training of a log-linear model
where the first three steps are feature extraction steps and the last is the actual classification step.
This approach was first published in [
          <xref ref-type="bibr" rid="ref6 ref7">6, 7</xref>
          ].
        </p>
        <p>The method follows the promising approach of considering objects to be constellations of
parts which offers the immediate advantages that occlusions can be handled very well, that the
geometrical relationship between parts can be modelled (or neglected), and that one can focus on
the discriminative parts of an object. That is, one can focus on the image parts that distinguish
a certain object from other objects.</p>
        <p>The steps of the method are briefly outlined in the following paragraphs. To model the
difference in the training and test data, the first three steps have been done for the training and test
data individually, and then the according histograms have been extracted for the respective other,
so that once the vocabulary was learnt for the training data and once for the test data and the
histograms are created for each using both vocabularies. Results, however show that this seems
not to be a working approach to tackle divergence in training and testing data.
Patch Extraction Given an image, we extract square image patches at up to 500 image points.
Additionally, 300 points from a uniform grid of 15×20 cells that is projected onto the image
are used. At each of these points a set of square image patches of varying sizes (in this case
7 × 7, 11 × 11, 21 × 21, and 31 × 31 pixels) are extracted and scaled to a common size (in this case
15 × 15 pixels).</p>
        <p>In contrast to the interest points from the detector, the grid-points can also fall onto very
homogeneous areas of the image. This property is on the one hand important for capturing
homogeneity in objects which is not found by the interest point detector and on the other hand
it captures parts of the background which usually is a good indicator for an object, as in natural
image objects are often found in a “natural” environment.</p>
        <p>After the patches are extracted and scaled to a common size, a PCA dimensionality reduction
is applied to reduce the large dimensionality of the data, keeping 39 coefficients corresponding
to the 40 components of largest variance but discarding the first coefficient corresponding to the
largest variance. The first coefficient is discarded to achieve a partial brightness invariance. This
approach is suitable because the first PCA coefficient usually accounts for global brightness.
Clustering The data are then clustered using a k-means style iterative splitting clustering
algorithm to obtain a partition of all extracted patches. To do so, first one Gaussian density is
estimated which is then iteratively split to obtain more densities. These densities are then
reestimated using k-means until convergence is reached and then the next split is done. It has be
shown experimentally that results consistently improve up to 4096 clusters but for more than 4096
clusters the improvement is so small that it is not worth the higher computational demands.
Creation of Histograms Once we have the cluster model, we discard all information for each
patch except its closest corresponding cluster center identifier. For the test data, this identifier
is determined by evaluating the Euclidean distance to all cluster centers for each patch. Thus,
the clustering assigns a cluster c(x) ∈ {1, . . . C} to each image patch x and allows us to create
histograms of cluster frequencies by counting how many of the extracted patches belong to each
of the clusters. The histogram representation h(X) with C bins is then determined by counting
and normalization such that hc(X) = L1X PlL=X1 δ(c, c(xl)), where δ denotes the Kronecker delta
function, c(xl) is the closest cluster center to xl, and xl is the l-th image patch extracted from
image X, from which a total of LX patches are extracted.</p>
        <p>Training &amp; Classification Having obtained this representation by histograms of image patches,
we define a decision rule for the classification of images. The approach based on maximum
likelihood of the class-conditional distributions does not take into account the information of competing
classes during training. We can use this information by maximizing the class posterior
probability QkK=1 QnN=k 1 p(k|Xkn) instead. Assuming a Gaussian density with pooled covariances for the
class-conditional distribution, this maximization is equivalent to maximizing the parameters of a
log-linear or maximum entropy model
p(k|h) =</p>
        <p>1
Z(h)
exp</p>
        <p>C !
αk + X λkchc ,
c=1
where Z(h) = PK c=1 λkchc is the renormalization factor. We use a modified
k=1 exp αk + PC
version of generalized iterative scaling. Bayes’ decision rule is used for classification.
4</p>
        <sec id="sec-13-2-1">
          <title>Results</title>
          <p>The results for all submissions are given in Table 4 using the normal Qrels and in Table 5 using
the Qrels with additional information. Furthermore in Table 6, the results are presented using
the relevance information estimated for each image in the database for all classes. All results are
given as AP for the individual topics (first 10 columns) and MAP over all topics (last column).</p>
          <p>The tables are ordered by MAP (last column). When considering the individual topics, it can
be observed that some methods work well for some of the topics but completely fail for others.
To highlight this, the highest AP in each column in shown in bold. The two runs from Budapest
perform well for the bicycles (1) and motorbikes (4) topics respectively, but have bad results for
all other topics.</p>
          <p>rank
avg
avg</p>
          <p>rank
One issue that should be taken into account when interpreting the results is that 50% of the
evaluated runs are from HUTCIS and thus this group had a significant impact on the pools for
relevance assessment. This effect is further boosted by the fact that the initial runs from HUTCIS
(which were used for the pooling) had the queries 4–10 in wrong order. This implies that for
the topics where the runs were in correct order, HUTCIS has a good chance of having many of
their images in the pools and that furthermore for the runs where HUTCIS’ runs were broken,
none of their images were considered in pooling. Therefore, it can be observed that the additional
relevance information leads to some significant changes in some of the results. A particular strong
change of the results can be seen for the runs from HUTCIS: the APs of the queries which were
in correct order for the pooling process are strongly reduced whereas those which were in wrong
order are significantly improved, which shows that the large number of runs from HUTCIS had a
strong influence on the results of the pooling.</p>
          <p>The drawbacks of using pooling is visible in the results as the MAPs for some methods increase
when the additional relevance judgements are used. This effect is particular strong for topics with
only very few relevant images.</p>
          <p>The results clearly show that the task is a very difficult one, and that most recent methods are
not yet able to fully cope with tasks where training and test data do not match very well. This
means that although the other image- and object retrieval and recognition tasks are very good to
foster advancement in research, they do not yet pose a fully realistic challenge as their
trainingand testing data match up pretty well. A higher variability between the test- and the training
data would make these tasks more realistic.</p>
          <p>The participating methods use a wide variety of different techniques, some methods use global
image descriptors, other methods use local descriptors and discriminative models, which are
currently the state of the art in object recognition and detection.</p>
          <p>5
avg</p>
          <p>For the classes cat (5), cow (6), and sheep (9), AP values are so small that it is not obvious
which method produces the best results. For the classes cat and sheep this effect is even stronger
since only 6 resp. 7 images from the whole database are relevant.</p>
          <p>No single method obtained the highest AP values in all classes. A very strong outlier in this
respect are the two methods from the Budapest-group which have clearly the best results for the
bicycle and the motorbike class respectively but have very bad performance for all other classes.
The fact that the queries are very dissimilar in terms of 1) image statistics (number of correct
images) and 2) visual difficulty, makes the AP values vary strongly among the different topics and
therefore MAP is not a very reliable measure judge the overall retrieval quality here.</p>
          <p>Finally, although a maximum of 1000 images were to be returned for each query, the number
of relevant images exceeded 1000 for two classes (car and person). This implies that the maximum
possible AP (using the full annotations) for “car” is 0.789, and for “person” is 0.089, thus the
performance of the best runs for the person query is nearly perfect.
6</p>
        </sec>
        <sec id="sec-13-2-2">
          <title>Conclusion</title>
          <p>We presented the object retrieval task of ImageCLEF 2007, the methods of the participating
groups, and the results. The results show that none of the methods really solves the assigned task
and therewith that, although large advances in object detection and recognition were achieved
over the last years, still many improvements are necessary to solve realistic tasks on a web-scale
size with a high variety of the data to be analysed. It can however be observed that some of the
methods are able to obtain reasonable results for a limited set of classes.</p>
          <p>Furthermore, the analysis of the results showed that the use of pooling techniques for relevance
assessment can be problematic if the pools are biased due to erroneous runs or due to many strongly
correlated submissions as it was the case in this evaluation.</p>
        </sec>
        <sec id="sec-13-2-3">
          <title>Acknowledgements</title>
          <p>We would like to thank the CLEF campaign for supporting the ImageCLEF initiative. This work
was partially funded by the European Commission MUSCLE NoE (FP6-507752) and the DFG
(German research foundation) under contract Ne-572/6.</p>
          <p>We would like to thank the PASCAL NoE for allowing us to use the training data of the
PASCAL 2006 Visual object classes challenge as training data for this task, in particular, we
would like to thank Mark Everingham for his cooperation.</p>
          <p>We would like to thank Paul Clough from Sheffield University for support in creating the pools
and Jan Hosang, Jens Forster, Pascal Steingrube, Christian Plahl, Tobias Gass, Daniel Stein,
Morteza Zahedi, Richard Zens, Yuqi Zhang, Markus Nu¨sbaum, Gregor Leusch, Michael Arens,
Lech Szumilas, Jan Bungeroth, David Rybach, Peter Fritz, Arne Mauser, Saˇsa Hasan, and Stefan
Hahn for helping to create relevance assessments for the images.</p>
          <p>The work of the Budapest group was supported by a Yahoo! Faculty Research Grant and by
grants MOLINGV NKFP-2/0024/2005, NKFP-2004 project Language Miner.</p>
        </sec>
      </sec>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [1]
          <string-name>
            <given-names>M.</given-names>
            <surname>Braschler</surname>
          </string-name>
          and
          <string-name>
            <given-names>C.</given-names>
            <surname>Peters</surname>
          </string-name>
          .
          <article-title>CLEF Methodology and Metrics</article-title>
          . In C. Peters (Ed.),
          <article-title>Crosslanguage information retrieval and evaluation:</article-title>
          <source>Proceedings of the CLEF2001 Workshop, Lecture Notes in Computer Science 2406</source>
          , Springer Verlag, pages
          <fpage>394</fpage>
          -
          <lpage>404</lpage>
          ,
          <year>2002</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [2]
          <string-name>
            <given-names>C.</given-names>
            <surname>Carson</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Belongie</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Greenspan</surname>
          </string-name>
          , and
          <string-name>
            <given-names>J.</given-names>
            <surname>Malik</surname>
          </string-name>
          . Blobworld:
          <article-title>Image Segmentation Using Expectation-Maximization and Its Application to Image Querying</article-title>
          .
          <source>IEEE Trans. Pattern Anal. Mach</source>
          . Intell.,
          <volume>24</volume>
          (
          <issue>8</issue>
          ):
          <fpage>1026</fpage>
          -
          <lpage>1038</lpage>
          ,
          <year>2002</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          [3]
          <string-name>
            <given-names>C.-C.</given-names>
            <surname>Chang</surname>
          </string-name>
          and
          <string-name>
            <given-names>C.-J.</given-names>
            <surname>Lin</surname>
          </string-name>
          .
          <article-title>LIBSVM: a library for support vector machines</article-title>
          ,
          <year>2001</year>
          . Software available at http://www.csie.ntu.edu.tw/∼cjlin/libsvm.
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [4]
          <string-name>
            <given-names>Y.</given-names>
            <surname>Chen</surname>
          </string-name>
          and
          <string-name>
            <given-names>J. Z.</given-names>
            <surname>Wang</surname>
          </string-name>
          .
          <article-title>Image Categorization by Learning and Reasoning with Regions</article-title>
          .
          <source>J. Mach. Learn. Res.</source>
          ,
          <volume>5</volume>
          :
          <fpage>913</fpage>
          -
          <lpage>939</lpage>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          [5]
          <string-name>
            <given-names>P.</given-names>
            <surname>Clough</surname>
          </string-name>
          , H. Mu¨ller, and
          <string-name>
            <given-names>M.</given-names>
            <surname>Sanderson</surname>
          </string-name>
          .
          <article-title>Overview of the CLEF Cross-Language Image Retrieval Track (ImageCLEF) 2004</article-title>
          . In C. Peters,
          <string-name>
            <given-names>P. D.</given-names>
            <surname>Clough</surname>
          </string-name>
          ,
          <string-name>
            <given-names>G. J. F.</given-names>
            <surname>Jones</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Gonzalo</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Kluck</surname>
          </string-name>
          , and B. Magnini, editors,
          <source>Multilingual Information Access for Text</source>
          ,
          <article-title>Speech and Images: Result of the fifth CLEF evaluation campaign</article-title>
          ,
          <source>Lecture Notes in Computer Science</source>
          , Springer-Verlag, Bath, England,
          <year>2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          [6]
          <string-name>
            <given-names>T.</given-names>
            <surname>Deselaers</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Keysers</surname>
          </string-name>
          , and
          <string-name>
            <given-names>H.</given-names>
            <surname>Ney</surname>
          </string-name>
          .
          <article-title>Discriminative Training for Object Recognition using Image Patches</article-title>
          .
          <source>In IEEE Conference on Computer Vision and Pattern Recognition</source>
          , volume
          <volume>2</volume>
          , San Diego, CA, pages
          <fpage>157</fpage>
          -
          <lpage>162</lpage>
          ,
          <year>June 2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          [7]
          <string-name>
            <given-names>T.</given-names>
            <surname>Deselaers</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Keysers</surname>
          </string-name>
          , and
          <string-name>
            <given-names>H.</given-names>
            <surname>Ney</surname>
          </string-name>
          .
          <article-title>Improving a Discriminative Approach to Object Recognition using Image Patches</article-title>
          .
          <source>In DAGM</source>
          <year>2005</year>
          ,
          <string-name>
            <given-names>Pattern</given-names>
            <surname>Recognition</surname>
          </string-name>
          ,
          <source>26th DAGM Symposium, number 3663 in Lecture Notes in Computer Science</source>
          , Vienna, Austria, pages
          <fpage>326</fpage>
          -
          <lpage>333</lpage>
          ,
          <year>August 2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          [8]
          <string-name>
            <given-names>H. J.</given-names>
            <surname>Escalante</surname>
          </string-name>
          , M. M. y
          <string-name>
            <surname>G</surname>
          </string-name>
          ´omez, and
          <string-name>
            <given-names>L. E.</given-names>
            <surname>Sucar</surname>
          </string-name>
          .
          <article-title>Word Co-occurrence and MRF's for Improving Automatic Image Annotation</article-title>
          .
          <source>In In Proceedings of the 18th British Machine Vision Conference (BMVC</source>
          <year>2007</year>
          ), Warwick, UK, September,
          <year>2007</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          [9]
          <string-name>
            <given-names>M.</given-names>
            <surname>Everingham</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Zisserman</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Williams</surname>
          </string-name>
          , and
          <string-name>
            <given-names>L. V.</given-names>
            <surname>Gool</surname>
          </string-name>
          .
          <source>The Pascal Visual Object Classes Challenge</source>
          <year>2006</year>
          (
          <article-title>VOC2006) Results</article-title>
          .
          <source>Technical report</source>
          ,
          <year>2006</year>
          . Available on-line at http://www.pascal-network.org/.
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          [10]
          <string-name>
            <given-names>M.</given-names>
            <surname>Everingham</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Zisserman</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C. K. I.</given-names>
            <surname>Williams</surname>
          </string-name>
          ,
          <string-name>
            <surname>L. van Gool</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Allan</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C. M.</given-names>
            <surname>Bishop</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Chapelle</surname>
          </string-name>
          ,
          <string-name>
            <given-names>N.</given-names>
            <surname>Dalal</surname>
          </string-name>
          ,
          <string-name>
            <given-names>T.</given-names>
            <surname>Deselaers</surname>
          </string-name>
          , G. Dorko,
          <string-name>
            <given-names>S.</given-names>
            <surname>Duffner</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Eichhorn</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J. D. R.</given-names>
            <surname>Farquhar</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Fritz</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Garcia</surname>
          </string-name>
          ,
          <string-name>
            <given-names>T.</given-names>
            <surname>Griffiths</surname>
          </string-name>
          ,
          <string-name>
            <given-names>F.</given-names>
            <surname>Jurie</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Keysers</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Koskela</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Laaksonen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Larlus</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.</given-names>
            <surname>Leibe</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Meng</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Ney</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.</given-names>
            <surname>Schiele</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Schmid</surname>
          </string-name>
          ,
          <string-name>
            <given-names>E.</given-names>
            <surname>Seemann</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Shawe-Taylor</surname>
          </string-name>
          , A. Storkey,
          <string-name>
            <given-names>S.</given-names>
            <surname>Szedmak</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.</given-names>
            <surname>Triggs</surname>
          </string-name>
          ,
          <string-name>
            <given-names>I.</given-names>
            <surname>Ulusoy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            <surname>Viitaniemi</surname>
          </string-name>
          , and
          <string-name>
            <surname>J. Zhang.</surname>
          </string-name>
          <article-title>The 2005 PASCAL Visual Object Classes Challenge</article-title>
          .
          <source>In Machine Learning Challenges. Evaluating Predictive Uncertainty, Visual Object Classification, and Recognising Tectual Entailment (PASCAL Workshop 05), number 3944 in Lecture Notes in Artificial Intelligence</source>
          , Southampton, UK, pages
          <fpage>117</fpage>
          -
          <lpage>176</lpage>
          ,
          <year>2006</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          [11]
          <string-name>
            <given-names>P. F.</given-names>
            <surname>Felzenszwalb</surname>
          </string-name>
          and
          <string-name>
            <given-names>D. P.</given-names>
            <surname>Huttenlocher</surname>
          </string-name>
          .
          <article-title>Efficient Graph-Based Image Segmentation</article-title>
          .
          <source>International Journal of Computer Vision</source>
          ,
          <volume>59</volume>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          [12]
          <string-name>
            <given-names>R.</given-names>
            <surname>Fergus</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Fei-Fei</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Perona</surname>
          </string-name>
          ,
          <article-title>and</article-title>
          <string-name>
            <given-names>A.</given-names>
            <surname>Zisserman</surname>
          </string-name>
          .
          <article-title>Learning object categories from Google's image search</article-title>
          .
          <source>In International Conference on Computer Vision</source>
          , Beijing, China, Oct
          <year>2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          [13]
          <string-name>
            <given-names>Y.</given-names>
            <surname>Freund</surname>
          </string-name>
          and
          <string-name>
            <given-names>R.</given-names>
            <surname>Schapire</surname>
          </string-name>
          .
          <article-title>Experiments with a New Boosting Algorithm</article-title>
          .
          <source>In International Conference on Machine Learning</source>
          , pages
          <fpage>148</fpage>
          -
          <lpage>156</lpage>
          ,
          <year>1996</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          [14]
          <string-name>
            <given-names>M.</given-names>
            <surname>Grubinger</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Clough</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Hanbury</surname>
          </string-name>
          , and
          <string-name>
            <given-names>H.</given-names>
            <surname>Mu</surname>
          </string-name>
          <article-title>¨ller. Overview of the ImageCLEF 2007 Photographic Retrieval Task</article-title>
          .
          <source>In Working Notes of the 2007 CLEF Workshop</source>
          , Budapest, Hungary,
          <year>September 2007</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref15">
        <mixed-citation>
          [15]
          <string-name>
            <given-names>M.</given-names>
            <surname>Grubinger</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Clough</surname>
          </string-name>
          , H. Mu¨ller, and
          <string-name>
            <given-names>T.</given-names>
            <surname>Deselaers</surname>
          </string-name>
          .
          <article-title>The IAPR Benchmark: A New Evaluation Resource for Visual Information Systems</article-title>
          .
          <source>In LREC 06 OntoImage</source>
          <year>2006</year>
          :
          <article-title>Language Resources for Content-Based Image Retrieval</article-title>
          , Genoa, Italy, page in press, May
          <year>2006</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref16">
        <mixed-citation>
          [16]
          <string-name>
            <given-names>C.</given-names>
            <surname>Harris</surname>
          </string-name>
          and
          <string-name>
            <given-names>M.</given-names>
            <surname>Stephens</surname>
          </string-name>
          .
          <article-title>A Combined Corner and Edge Detection</article-title>
          .
          <source>In 4th Alvey Vision Conference</source>
          , pages
          <fpage>147</fpage>
          -
          <lpage>151</lpage>
          ,
          <year>1988</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref17">
        <mixed-citation>
          [17]
          <string-name>
            <given-names>S. C. H.</given-names>
            <surname>Hoi</surname>
          </string-name>
          and
          <string-name>
            <given-names>M. R.</given-names>
            <surname>Lyu</surname>
          </string-name>
          .
          <article-title>A novel log-based relevance feedback technique in content-based image retrieval</article-title>
          .
          <source>In 12th ACM International Conference on Multimedia (MM</source>
          <year>2004</year>
          ), New York, NY, USA, pages
          <fpage>24</fpage>
          -
          <lpage>31</lpage>
          , oct
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref18">
        <mixed-citation>
          [18]
          <string-name>
            <given-names>S. C.</given-names>
            <surname>Hoi</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M. R.</given-names>
            <surname>Lyu</surname>
          </string-name>
          , and
          <string-name>
            <given-names>R.</given-names>
            <surname>Jin</surname>
          </string-name>
          .
          <article-title>A unified log-based relevance feedback scheme for image retrieval</article-title>
          .
          <source>IEEE Transactions on Knowledge and Data Engineering</source>
          ,
          <volume>18</volume>
          (
          <issue>4</issue>
          ):
          <fpage>509</fpage>
          -
          <lpage>524</lpage>
          ,
          <year>2006</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref19">
        <mixed-citation>
          [19]
          <string-name>
            <given-names>J.</given-names>
            <surname>Laaksonen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Koskela</surname>
          </string-name>
          , and
          <string-name>
            <surname>E. Oja.</surname>
          </string-name>
          <article-title>PicSOM-Self-Organizing Image Retrieval with MPEG-7 Content Descriptions</article-title>
          .
          <source>IEEE Transactions on Neural Networks, Special Issue on Intelligent Multimedia Processing</source>
          ,
          <volume>13</volume>
          (
          <issue>4</issue>
          ):
          <fpage>841</fpage>
          -
          <lpage>853</lpage>
          ,
          <year>July 2002</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref20">
        <mixed-citation>
          [20]
          <string-name>
            <given-names>D.</given-names>
            <surname>Lowe</surname>
          </string-name>
          .
          <article-title>Distinctive Image Features from Scale-Invariant Keypoints</article-title>
          . IJCV,
          <volume>60</volume>
          (
          <issue>2</issue>
          ):
          <fpage>91</fpage>
          -
          <lpage>110</lpage>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref21">
        <mixed-citation>
          [21]
          <string-name>
            <given-names>Q.</given-names>
            <surname>Lv</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Charikar</surname>
          </string-name>
          , and
          <string-name>
            <given-names>K.</given-names>
            <surname>Li</surname>
          </string-name>
          .
          <article-title>Image similarity search with compact data structures</article-title>
          .
          <source>In CIKM '04: Proceedings of the thirteenth ACM international conference on Information and knowledge management</source>
          , ACM Press, New York, NY, USA, pages
          <fpage>208</fpage>
          -
          <lpage>217</lpage>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref22">
        <mixed-citation>
          [22]
          <string-name>
            <surname>H.-M. Marin-Castro</surname>
            ,
            <given-names>L. E.</given-names>
          </string-name>
          <string-name>
            <surname>Sucar</surname>
            , and
            <given-names>E. F.</given-names>
          </string-name>
          <string-name>
            <surname>Morales</surname>
          </string-name>
          .
          <article-title>Automatic image annotation using a semi-supervised ensemble of classifiers</article-title>
          . To appear.
          <source>In 12th Iberoamerican Congress on Pattern Recognition CIARP 2007, Lecture Notes in Computer Science</source>
          , Springer,
          <source>Vin˜a del Mar</source>
          , Valparaiso, Chile,
          <year>2007</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref23">
        <mixed-citation>
          [23]
          <string-name>
            <given-names>K.</given-names>
            <surname>Mikolaczyk</surname>
          </string-name>
          and
          <string-name>
            <surname>C. Schmid.</surname>
          </string-name>
          <article-title>A performance evaluation of local descriptors</article-title>
          .
          <source>IEEE Transactions on Pattern Analysis &amp; Machine Intelligence</source>
          ,
          <volume>27</volume>
          (
          <issue>10</issue>
          ):
          <fpage>1615</fpage>
          -
          <lpage>1630</lpage>
          ,
          <year>2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref24">
        <mixed-citation>
          [24]
          <string-name>
            <given-names>P.-A.</given-names>
            <surname>Moellic</surname>
          </string-name>
          and
          <string-name>
            <given-names>C.</given-names>
            <surname>Fluhr</surname>
          </string-name>
          .
          <article-title>ImageEVAL 2006 Official Campaign</article-title>
          .
          <source>Technical report, ImagEVAL</source>
          ,
          <year>2006</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref25">
        <mixed-citation>
          [25]
          <string-name>
            <given-names>P.</given-names>
            <surname>Montesinos</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            <surname>Gouet</surname>
          </string-name>
          , and
          <string-name>
            <given-names>R.</given-names>
            <surname>Deriche.</surname>
          </string-name>
          <article-title>Differential Invariants for Color Images</article-title>
          .
          <source>In ICPR, page 838</source>
          ,
          <year>1998</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref26">
        <mixed-citation>
          [26]
          <string-name>
            <given-names>H.</given-names>
            <surname>Mu</surname>
          </string-name>
          ¨ller, T. Deselaers, E. Kim,
          <string-name>
            <given-names>J.</given-names>
            <surname>Kalpathy-Cramer</surname>
          </string-name>
          ,
          <string-name>
            <given-names>T. M.</given-names>
            <surname>Deserno</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Clough</surname>
          </string-name>
          , and
          <string-name>
            <given-names>W.</given-names>
            <surname>Hersh</surname>
          </string-name>
          .
          <article-title>Overview of the ImageCLEFmed 2007 Medical Retrieval and Annotation Tasks</article-title>
          .
          <source>In Working Notes of the 2007 CLEF Workshop</source>
          , Budapest, Hungary,
          <year>September 2007</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref27">
        <mixed-citation>
          [27]
          <string-name>
            <given-names>B. G.</given-names>
            <surname>Prasad</surname>
          </string-name>
          ,
          <string-name>
            <given-names>K. K.</given-names>
            <surname>Biswas</surname>
          </string-name>
          , and
          <string-name>
            <given-names>S. K.</given-names>
            <surname>Gupta</surname>
          </string-name>
          .
          <article-title>Region-based image retrieval using integrated color, shape, and location index</article-title>
          .
          <source>Comput. Vis. Image Underst</source>
          .,
          <volume>94</volume>
          (
          <issue>1-3</issue>
          ):
          <fpage>193</fpage>
          -
          <lpage>233</lpage>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref28">
        <mixed-citation>
          [28]
          <string-name>
            <given-names>Y.</given-names>
            <surname>Rubner</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Tomasi</surname>
          </string-name>
          , and
          <string-name>
            <given-names>L. J.</given-names>
            <surname>Guibas</surname>
          </string-name>
          .
          <article-title>The Earth Mover's Distance as a Metric for Image Retrieval</article-title>
          . IJCV,
          <volume>40</volume>
          (
          <issue>2</issue>
          ):
          <fpage>99</fpage>
          -
          <lpage>121</lpage>
          ,
          <year>2000</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref29">
        <mixed-citation>
          [29]
          <string-name>
            <given-names>J.</given-names>
            <surname>Shi</surname>
          </string-name>
          and
          <string-name>
            <given-names>J.</given-names>
            <surname>Malik</surname>
          </string-name>
          .
          <article-title>Normalized Cuts and Image Segmentation</article-title>
          .
          <source>IEEE Transactions on Pattern and Machine Intelligence</source>
          ,
          <volume>22</volume>
          :
          <fpage>888</fpage>
          -
          <lpage>905</lpage>
          ,
          <year>2000</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref30">
        <mixed-citation>
          [30]
          <string-name>
            <given-names>J.</given-names>
            <surname>Shi</surname>
          </string-name>
          and
          <string-name>
            <given-names>J.</given-names>
            <surname>Malik</surname>
          </string-name>
          .
          <article-title>Normalized Cuts and Image Segmentation</article-title>
          .
          <source>IEEE Trans. Pattern Anal. Mach</source>
          . Intell.,
          <volume>22</volume>
          (
          <issue>8</issue>
          ):
          <fpage>888</fpage>
          -
          <lpage>905</lpage>
          ,
          <year>2000</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref31">
        <mixed-citation>
          [31]
          <string-name>
            <given-names>J.</given-names>
            <surname>St</surname>
          </string-name>
          <article-title>¨ottinger, A</article-title>
          . Hanbury,
          <string-name>
            <given-names>N.</given-names>
            <surname>Sebe</surname>
          </string-name>
          , and
          <string-name>
            <given-names>T.</given-names>
            <surname>Gevers</surname>
          </string-name>
          .
          <source>Do Colour Interest Points Improve Image Retrieval? ICIP</source>
          ,
          <year>2007</year>
          . to appear.
        </mixed-citation>
      </ref>
      <ref id="ref32">
        <mixed-citation>
          [32]
          <string-name>
            <surname>J. van de Weijer</surname>
            and
            <given-names>T.</given-names>
          </string-name>
          <string-name>
            <surname>Gevers</surname>
          </string-name>
          .
          <article-title>Edge and Corner Detection by Photometric Quasi-Invariants</article-title>
          . PAMI,
          <volume>27</volume>
          (
          <issue>4</issue>
          ):
          <fpage>625</fpage>
          -
          <lpage>630</lpage>
          ,
          <year>2005</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref33">
        <mixed-citation>
          [33]
          <string-name>
            <surname>J. van de Weijer</surname>
          </string-name>
          , T. Gevers,
          <article-title>and</article-title>
          <string-name>
            <given-names>A.</given-names>
            <surname>Bagdanov</surname>
          </string-name>
          .
          <article-title>Boosting Color Saliency in Image Feature Detection</article-title>
          . PAMI,
          <volume>28</volume>
          (
          <issue>1</issue>
          ):
          <fpage>150</fpage>
          -
          <lpage>156</lpage>
          ,
          <year>2006</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref34">
        <mixed-citation>
          [34]
          <string-name>
            <given-names>V.</given-names>
            <surname>Viitaniemi</surname>
          </string-name>
          and
          <string-name>
            <given-names>J.</given-names>
            <surname>Laaksonen</surname>
          </string-name>
          .
          <article-title>Improving the accuracy of global feature fusion based image categorisation</article-title>
          .
          <source>In Proceedings of the 2nd International Conference on Semantic and Digital Media Technologies (SAMT</source>
          <year>2007</year>
          ), Lecture Notes in Computer Science, Springer, Genova, Italy,
          <year>December 2007</year>
          . Accepted.
        </mixed-citation>
      </ref>
      <ref id="ref35">
        <mixed-citation>
          [35]
          <string-name>
            <given-names>L.</given-names>
            <surname>Wu</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M. J.</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Z. W.</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>W. Y.</given-names>
            <surname>Ma</surname>
          </string-name>
          , and
          <string-name>
            <given-names>N. H.</given-names>
            <surname>Yu</surname>
          </string-name>
          .
          <article-title>Visual Language Modeling for Image Classification</article-title>
          .
          <source>In 9th ACM SIGMM International Workshop on Multimedia Information Retrieval</source>
          ,
          <source>(MIR'07)</source>
          , Augsburg, Germany, sep
          <year>2007</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref36">
        <mixed-citation>
          [36]
          <string-name>
            <given-names>J.</given-names>
            <surname>Zhang</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Marszalek</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Lazebnik</surname>
          </string-name>
          , and
          <string-name>
            <given-names>C.</given-names>
            <surname>Schmid</surname>
          </string-name>
          .
          <article-title>Local Features and Kernels for Classification of Texture and Object Categories: A Comprehensive Study</article-title>
          .
          <source>CWPR</source>
          ,
          <volume>73</volume>
          (
          <issue>2</issue>
          ):
          <fpage>213</fpage>
          -
          <lpage>238</lpage>
          ,
          <year>2006</year>
          .
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>