<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>QURATOR: Innovative Technologies for Content and Data Curation</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Georg Rehm</string-name>
          <email>georg.rehm@dfki.de</email>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Peter Bourgonje</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Stefanie Hegele</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Florian Kintzel</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Julian Moreno Schneider</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Malte Ostendor</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Karolina Zaczynska</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Armin Berger</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Stefan Grill</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Soren Rauchle</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jens Rauenbusch</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lisa Rutenburg</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andre Schmidt</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mikka Wild</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Henry Ho mann</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Julian Fink</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sarah Schulz</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jurica Seva</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Joachim Quantz</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Joachim Bottger</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jose ne Matthey</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Rolf Fricke</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jan Thomsen</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Adrian Paschke</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jamal Al Qundus</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Thomas Hoppe</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Naouel Karam</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Frauke Weichhardt</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Christian Fillies</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Clemens Neudecker</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mike Gerber</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Kai Labusch</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Vahid Rezanezhad</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Robin Schaefer</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>David Zellhofer</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Siewert</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Bunk</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Julia Katharina Schlichting</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lydia Pintscher</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Elena Aleynikova</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Franziska Heine</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>3pc GmbH Neue Kommunikation</institution>
          ,
          <addr-line>Prinzessinnenstra e 1, 10969 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Ada Health GmbH</institution>
          ,
          <addr-line>Adalbertstra e 20, 10997 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>COM AG</institution>
          ,
          <addr-line>Kleiststra e 23-26, 10787 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Condat AG</institution>
          ,
          <addr-line>Alt-Moabit 91d, 10559 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>DFKI GmbH, Alt-Moabit 91c</institution>
          ,
          <addr-line>10559 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Fraunhofer FOKUS</institution>
          ,
          <addr-line>Kaiserin-Augusta-Allee 31, 10589 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>Semtation GmbH</institution>
          ,
          <addr-line>Geschwister-Scholl-Stra e 38, 14471 Potsdam</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Staatsbibliothek zu Berlin (SPK)</institution>
          ,
          <addr-line>Potsdamer Stra e 33 10785 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>Ubermetrics Technologies GmbH</institution>
          ,
          <addr-line>Kronenstra e 1, 10117 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>Wikimedia Deutschland e. V.</institution>
          ,
          <addr-line>Tempelhofer Ufer 23-24, 10963 Berlin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
      </contrib-group>
      <abstract>
        <p>In all domains and sectors, the demand for intelligent systems to support the processing and generation of digital content is rapidly increasing. The availability of vast amounts of content and the pressure to publish new content quickly and in rapid succession requires faster, more e cient and smarter processing and generation methods. With a consortium of ten partners from research and industry and a broad range of expertise in AI, Machine Learning and Language Technologies, the QURATOR project, funded by the German Federal Ministry of Education and Research, develops a sustainable and innovative technology platform that provides services to support knowledge workers in various industries to address the challenges they face when curating digital content. The project's vision and ambition is to establish an ecosystem for content curation technologies that signi cantly pushes the current state of the art and transforms its region, the metropolitan area BerlinBrandenburg, into a global centre of excellence for curation technologies.</p>
      </abstract>
      <kwd-group>
        <kwd>Curation Technologies</kwd>
        <kwd>Language Technologies</kwd>
        <kwd>Semantic</kwd>
        <kwd>Technologies</kwd>
        <kwd>Knowledge Technologies</kwd>
        <kwd>Arti cial Intelligence</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>-</title>
      <p>Copyright c 2020 for this paper by its authors.</p>
      <p>Use permitted under Creative Commons License Attribution 4.0 International (CC BY 4.0).</p>
    </sec>
    <sec id="sec-2">
      <title>Introduction</title>
      <p>
        Digital content and online media have gained immense importance, especially in
business, but also in politics and many other areas of society. Some of the many
challenges include better support and smarter technologies for digital content
curators who are exposed to an ever increasing stream of heterogeneous information
they need to process further. For example, professionals in a digital agency create
websites or mobile apps for customers who provide documents, data, pictures,
videos etc. that are processed and then deployed as new websites or mobile apps.
Knowledge workers in libraries digitize archives, add metadata and publish them
online. Journalists need to continuously stay up to date to be able to write a
new article on a speci c topic. Many more examples exist in various industries
and media sectors (television, radio, blogs, journalism, etc.). These diverse work
environments can bene t immensely from smart semantic technologies that help
content curators, who are usually under great time pressure, to support their
processes. Currently, they use a wide range of non-integrated, isolated, and
fragmented tools such as search engines, Wikipedia, databases, content management
systems, or enterprise wikis to perform their curation work. Largely manual tasks
such as smart content search and production, summarization, classi cation as
well as visualization are only partially supported by existing tools [
        <xref ref-type="bibr" rid="ref14">14</xref>
        ].
      </p>
      <p>The QURATOR project1, funded by the German Federal Ministry of
Education and Research (BMBF), with a project runtime of three years
(11/201810/2021), is based in the metropolitan region Berlin/Brandenburg. The
consortium of ten project partners from research and industry combines vast expertise
in areas such as Language Technologies as well as Knowledge Technologies,
Articial Intelligence and Machine Learning. The projects main goal is the
development of a sustainable technology platform that supports knowledge workers in
various industries. Non-e cient process chains increase the manual processing
e ort for workers even more. The platform will simplify the curation of digital
content and accelerate it dramatically. AI techniques are integrated into curation
technologies and curation work ows in the form of industry solutions covering
the entire life cycle of content curation. The solutions being developed focus on
curation services for the sectors of culture, media, health and industry.</p>
      <p>In Section 2 we describe the emerging QURATOR technology platform. In
the main part of this article, Section 3, we provide brief summaries of the ten
partner projects. Section 4 concludes the article with a short summary.
2</p>
    </sec>
    <sec id="sec-3">
      <title>The QURATOR Curation Technology Platform</title>
      <p>The centerpiece of the project is the development of a platform for digital
curation technologies. The project develops, integrates and evaluates various services
for importing, preprocessing, analyzing and generating content that covers a wide
range of information sources and data formats, spanning use cases from several</p>
      <sec id="sec-3-1">
        <title>1 https://qurator.ai</title>
        <p>industries and domains. A special focus of the project is on the integration of
AI methods to improve the quality, exibility and coverage of the services.</p>
        <p>Figure 1 outlines the concept of the QURATOR Curation Technology
Platform which can be divided into three main layers. In order to process and
transform incoming data, text and multimedia streams from di erent sources to
device-adapted, publishable content, various groups of components, services and
technologies are applied. First, the set of basic technologies includes adapters
to data, content and knowledge sources, as well as infrastructural tools and
smart AI methods for the acquisition, analysis and generation of content.
Second, knowledge workers can make use of curation tools and services which have
knowledge sources and intelligent procedures already integrated in order to
process content. Third, there are selected use case and application areas (culture,
media, health, industry), i. e., the respective integration of curation tools and
services. Each of the three layers has already been populated with technology
components which are to be further developed and also extended in the following
two years of the project.
The project consortium includes ten partners: the research centers DFKI and
Fraunhofer FOKUS, the industry partners 3pc, Ada Health, ART+COM,
Condat, Semtation, Ubermetrics as well as Wikimedia Germany and the Berlin State
Library (Stiftung Preu ischer Kulturbesitz). Several of these partners already
contributed to previous BMBF-funded projects including Corporate Smart
Content2 and Digital Curation Technologies3, which focused on semiautomatic
methods for the e cient processing, creation and distribution of high quality media
content and laid the groundwork for the QURATOR project.</p>
      </sec>
      <sec id="sec-3-2">
        <title>2 https://www.unternehmen-region.de/de/7923.php</title>
      </sec>
      <sec id="sec-3-3">
        <title>3 http://digitale-kuratierung.de</title>
        <p>In the following, we brie y introduce each partner and provide an overview
of their respective projects focusing upon the current state and the next steps.
3.1</p>
        <sec id="sec-3-3-1">
          <title>DFKI GmbH: A Flexible AI Platform for the Adaptive Analysis and Creative Generation of Digital Content</title>
          <p>DFKI (Deutsches Forschungszentrum fur Kunstliche Intelligenz GmbH) is
Germany's leading research center in the eld of innovative software technologies
based on AI methods. The Speech and Language Technology Lab conducts
advanced research in language technology and provides novel computational
techniques for processing text, speech and knowledge.</p>
          <p>
            In QURATOR, DFKI focuses on the development of an innovative platform
for digital curation technologies [
            <xref ref-type="bibr" rid="ref13 ref2 ref21 ref22">2, 13, 21, 22</xref>
            ] as well as on the population of
this platform with various processing services. This platform plays a crucial
role in the project as it is being designed together with all partners who also
contribute services to the platform.4 Ultimately, the QURATOR platform will
contain services, data sets and components that are able to handle and to process
di erent types and classes of content as well as content sources. The DFKI
services can be divided into three classes.
          </p>
          <p>
            Preprocessing encompasses the services that are responsible for obtaining and
processing information from di erent content sources so that they can be used
in the platform and integrated into other services [
            <xref ref-type="bibr" rid="ref23">23</xref>
            ]. These services include
the provisioning of data sets and content (web pages, RSS feeds etc.), language
and duplicate detection as well as document structure recognition.
          </p>
          <p>
            Semantic analysis includes services that process a document (or part of it)
and add information in the form of annotations. These services are named entity
recognition and linking, temporal expression analysis, relation extraction, event
detection, fake news as well as discourse analysis [
            <xref ref-type="bibr" rid="ref16 ref25 ref3 ref9">3, 25, 16, 9</xref>
            ].
          </p>
          <p>
            Content generation contains services that make use of annotated information
(semantic analysis) to help create a new piece of information. These services are
summarization, paraphrasing, automatic translation and semantic storytelling
for both text and multimedia content [
            <xref ref-type="bibr" rid="ref12 ref15 ref17 ref19 ref20 ref7">17, 15, 7, 12, 20, 19</xref>
            ].
          </p>
          <p>DFKI will continue the development of the di erent services as well as the
infrastructure. Since a exible organization needs to be guaranteed, DFKI is also
responsible for the design and implementation of work ows. These will ultimately
enable the joint use of (almost) all the services available in the platform.
3.2</p>
        </sec>
        <sec id="sec-3-3-2">
          <title>3pc GmbH: Curation Technologies for Interactive Storytelling</title>
          <p>3pc creates solutions for the digital age, combining strategy, design, technology,
and communication in a holistic and user-centered approach. As experts in the</p>
        </sec>
      </sec>
      <sec id="sec-3-4">
        <title>4 This platform is developed in close collaboration with the EU project European Lan</title>
        <p>
          guage Grid, which is also coordinated by DFKI, see
https://www.european-languagegrid.eu and [
          <xref ref-type="bibr" rid="ref11">11</xref>
          ] for more details.
development of novel and unique digital products, 3pc identi es core challenges
and key content within complex subject matters.
        </p>
        <p>
          As part of the QURATOR project, 3pc develops intelligent tools for
interactive storytelling [
          <xref ref-type="bibr" rid="ref1 ref15">1, 15</xref>
          ] in order to assist editors, content curators, and publishers
from cultural and scienti c institutions, corporate communication divisions, and
media agencies. The providers for storytelling face an increasing challenge telling
engaging stories built from vast amounts of data for a broad range of devices,
including wearables, augmented and virtual reality systems, voice-based user
interfaces { and whatever the future holds. In this context, interactive storytelling
is de ned as novel media formats and implementations that exceed todays rather
static websites by far. 3pc is currently building an asset management tool that
enables users to access media analysis algorithms in an intuitive and e cient
way (Figure 2). Media analysis processes text, images, videos, and audio les in
order to enrich them with additional information such as content description,
sentiment or topic, which is usually a labor-intensive and, therefore, expensive
process often neglected in busy publishing environments. Enriched media
becomes machine-readable, allowing storytellers to nd content faster and for new
connections to be forged in order to create richer, interactive stories. Ultimately,
a semantic storytelling machine becomes possible, generating semi-automatically
unique and tailored stories, based entirely on user preferences. At 3pc, research
is conducted through an iterative process by creating functional prototypes and
testing their usefulness on representative members of di erent user groups. 3pc
ensures that all novel technology solutions are adapted to each users needs,
taking into consideration their tasks, behaviour and knowledge.
        </p>
        <p>Next up, 3pc will extend traditional forms of interactive storytelling by
exploring space, voice, and generated audio as means of human-computer
interaction. Further research will also be conducted on training algorithms for
domainspeci c tasks in order to develop curation tools for di erent areas of expertise.</p>
        <sec id="sec-3-4-1">
          <title>Ada Health GmbH: Curation of Biomedical Knowledge</title>
          <p>Ada Health GmbH is a global health company founded by doctors, scientists,
and industry pioneers to create new possibilities for personal health. Ada's core
system connects medical knowledge with intelligent technology to help people
actively manage their health and medical professionals to deliver e ective care.</p>
          <p>Within the QURATOR project, Ada focuses on supporting the structured
medical knowledge creation by providing a tool for pre-extracting information
from unstructured text. This tool utilizes methods from biomedical Natural
Language Processing (NLP). Since the quality of the medical database is of utmost
importance to ensure accurate diagnosis support, the \human in the loop"
approach leverages the deep medical knowledge provided by Ada's doctors and the
e ciency of AI methods.</p>
          <p>As a rst step, Ada's researchers focus on the extraction of medical
entities from medical case reports. These descriptions of a patient's symptoms are
usually semi-structured and can function as test cases for Ada's quality control.
The extraction of a structured case requires NLP methods such as the detection
of relevant paragraphs, named entity recognition and named entity
normalization. Challenging characteristics of named entities in the biomedical domain are
their discontinuous nature in text as well as their high heterogeneity in terms of
linguistic features. Thus, these domain-speci c characteristics require the
adaptation and implementation of domain-tailored NLP solutions. In order to do so,
data is required which Ada acquires through a combination of manual annotation
and active learning from feedback given by the medical content editors.
3.4</p>
        </sec>
        <sec id="sec-3-4-2">
          <title>ART+COM AG: Curation Tools for Multimedia Content</title>
          <p>ART+COM Studios designs and develops new media spaces and installations.
New technology is not only used as an artistic medium of expression but as a
medium for the interactive communication of complex information. In the
process, ART+COM improves existing technologies constantly and explores their
applications both independently and in cooperation with other companies and
academic institutions.</p>
          <p>The main focus within QURATOR is to develop basic technologies to
automatically process and assess multimedia content and ultimately create smart
exhibits that can organize and generate content automatically.</p>
          <p>The current objective is to categorize entities and visualize their relations
in huge datasets. ART+COM's curation tool will import results from machine
learning (ML) methods, including NLP, action detection, image recognition and
processing of interconnected data. Subsequently, it should be used to create
automated content analyses and visualizations that are navigable and support
knowledge workers as well as the creation of interactive museum exhibits. The
interconnected entities in Wikidata, one of the most relevant databases for
researchers, are subject of one of the ongoing sub-projects. The project focuses on
an interactive software to visualize, explore, and curate knowledge contained in
Wikidata. Of particular interest are interaction techniques to lter and select
the objects of interest and their relationships between each other. The software
framework also explores potential data arrangements in two-dimensional and
three-dimensional space, tailored to their relevance to particular questions and
interests. Figure 3 shows the prototype of the Wikidata knowledge graph tool
displaying a selection of items and their connections to each other.
3.5</p>
        </sec>
        <sec id="sec-3-4-3">
          <title>Condat AG: Smart Newsboard</title>
          <p>Condat AG has a strong focus on the media industry, mainly on public
broadcasters. Condat support all parts of the distribution chain for these broadcasters.</p>
          <p>Within the QURATOR project, Condat develops a Smart Newsboard
(Figure 4) as a means to produce content based on news around a particular topic.
This involves a chain of di erent curation services such as 1) nding the original
sources around a topic by searches and subscription to RSS feeds, Twitter
channels etc., 2) categorizing and classifying sources into meaningful groups, either
through topic detection (if the categories are not prede ned) or text classi
cation (if the categories are xed), or even a combination of both, 3) applying text
analysis, named entity recognition and enrichment by linking the entities to
speci c resources such as Wikidata which allows the building of a knowledge graph
to nd connections between, e. g., people and events in di erent contexts and,
thus, enable journalists to pursue a deeper analysis. Condat also explores the
identi cation of temporal expressions which enables the possibility to generate
story outlines based on the sequence of events as extracted from multiple
documents (timelining). Another curation service relevant for the Smart Newsboard
is the summarization of sources. This includes not only the summarization of
individual texts but more importantly the summarization of multiple documents.</p>
          <p>Curation services for summarization, named entity recognition and topic
detection have already been implemented. The next steps are the integration of
additional services, as they become available, and the design of the user interface.
3.6</p>
        </sec>
        <sec id="sec-3-4-4">
          <title>Fraunhofer FOKUS: Corporate Smart Insights</title>
          <p>The Fraunhofer Institute for Open Communication Systems (FOKUS) develops
solutions for the communication infrastructure of the future. With more than
30 years of experience, FOKUS is one of the most important actors in the ICT
research landscape both nationally and worldwide. Fraunhofer FOKUS develops
innovative processes from the original concept up to the pre-product in
companies and institutions. As a member of important standardization bodies, the
institute contributes to the de nition of new ICT standards. It researches and
develops application-orientated solutions for partners in industry, research and
public administration in various ICT elds.</p>
          <p>Fraunhofer FOKUS has signi cant experience in semantic data intelligence
and AI, concentrated in the DANA group, which drives the research and the
development in the area of Corporate Semantic Web. Using this experience,
the aspect realized in the QURATOR project is an insight-driven AI approach
(Figure 5) which bene ts the technological innovation of an Insight Driven
Organisation (IDO).5 An IDO embeds corporate knowledge, reasoning and smart</p>
        </sec>
      </sec>
      <sec id="sec-3-5">
        <title>5 https://qurator.ai/partner/fraunhofer-fokus/</title>
        <p>insights learned from data analytics into the daily decisions and actions of an
organisation, including their argumentation and interpretation.</p>
        <p>
          The technical CSI framework consists of knowledge repositories for the
distributed management of knowledge artefacts, such as semantic knowledge graphs
and terminologies, a standardized API for knowledge bases [
          <xref ref-type="bibr" rid="ref10">10</xref>
          ], a knowledge
extraction and analytics6 service, and methods for corporate smart insights
knowledge evolution, as well as services to reuse the learned CSI knowledge for AI,
including inference, explainability and plausibility/validation.
3.7
        </p>
        <sec id="sec-3-5-1">
          <title>Semtation GmbH: Intelligent Business Process Modelling</title>
          <p>Semtation GmbH provides the platform SemTalk (registered trademark) for
modeling and supporting business processes and knowledge structures, an easy
to use but very powerful modeling and portal tool based on Microsoft Visio and
the Microsoft Cloud. SemTalk technology makes use of various tools provided in
Microsoft 365 in order to o er best-in-class portal experiences when it comes to
supporting business processes.</p>
          <p>In QURATOR, Semtation pursues the enhancement of business process model
usage. It consists of several tasks that aim in two directions, namely to 1) present
models on other devices but a monitor and 2) recommend information
dynamically based on the process context of the current user. Integrating various AI
technologies is necessary to obtain suitable results for both scenarios. It helps
to recognize your surroundings in order to recommend adequate information in
mixed reality settings and to understand natural language in a chat scenario. It
also makes it easier to understand the current process context in order to
recommend available documents and team members in various settings based on text
analysis and other machine learning use cases. Semtation has already integrated
knowledge graph information in process portals in order to use available internal
information to recommend suitable documents and people based on the current
process or project instance and the current task.</p>
          <p>The next step will be to de ne a scenario with one of the customers in order
to check requirements and results on a real world basis.
3.8</p>
        </sec>
        <sec id="sec-3-5-2">
          <title>Stiftung Preu ischer Kulturbesitz, Staatsbibliothek zu Berlin:</title>
        </sec>
        <sec id="sec-3-5-3">
          <title>Automated Curation Technologies for Digitized Cultural</title>
        </sec>
        <sec id="sec-3-5-4">
          <title>Heritage</title>
          <p>The Berlin State Library (Staatsbibliothek zu Berlin, SBB) is the largest research
library in Germany with more than 12 million documents in its holdings and
more than 2.5 PB of digital data stored throughout various repositories (as of
Oct. 2019). The collection encompasses texts, media and cultural works from all
elds in all languages, from all time periods and all countries of the world.</p>
          <p>
            Within QURATOR, the Berlin State Library is taking part in the R&amp;D
activities on behalf of the Prussian Heritage Foundation (Stiftung Preu ischer
Kulturbesitz, SPK). SBB aims to digitize all its copyright-free historical collections
6 https://www.cyber-akademie.de/anlage.jsp?id=959
and to make them available on the web7 as facsimile images with high-quality
text, logically structured and semantically annotated for use by researchers. In
order to achieve this goal, SBB works in a number of research areas in the
context of QURATOR { from layout and text recognition (OCR) and unsupervised
post-correction to named entity recognition (NER), disambiguation and linking.
Due to the huge volume and variety of the documents published between 1475
and 1945, solutions are required that are particularly robust and that can be
netuned to the complexities of historical fonts, layout, language and orthography.
While the SBB adopts state-of-the-art convolutional neural networks (CNN) like
ResNet50 [
            <xref ref-type="bibr" rid="ref5">5</xref>
            ] and UNet [
            <xref ref-type="bibr" rid="ref18">18</xref>
            ] in combination with attention and adds rule-based
domain adaptation for layout recognition and the classi cation of structural
elements, it follows a more classical RNN-LSTM-CTC approach for text recognition
[
            <xref ref-type="bibr" rid="ref26">26</xref>
            ], achieving character error rates below 1% with voting between multiple
models trained on su ciently large amounts of historical document ground truth data
[
            <xref ref-type="bibr" rid="ref24">24</xref>
            ]. In a related e ort, SBB is also taking part in the development of an open
end-to-end framework for historical document analysis and recognition based on
AI methods [
            <xref ref-type="bibr" rid="ref8">8</xref>
            ]. For NER, the recent transformer architecture BERT [
            <xref ref-type="bibr" rid="ref4">4</xref>
            ] is
utilized and adapted to historical German through a combination of unsupervised
pre-training and supervised learning [
            <xref ref-type="bibr" rid="ref6">6</xref>
            ]. The nal goal is to identify and classify
named entities found in the digitized documents, and to disambiguate and link
them to an online knowledge base, e. g., Wikidata.
          </p>
        </sec>
      </sec>
      <sec id="sec-3-6">
        <title>7 https://digital.staatsbibliothek-berlin.de</title>
        <p>Eventually, the digitized historical documents shall be made fully searchable
with semantic markup enabling advanced content retrieval scenarios and rich
contextualization of documents with knowledge from third party sources. In a
further step, image-based classi cation methods will be added to enhance the
document metadata and to complement the functionalities o ered through the
full-text search. In the course of 2020, a demonstrator will be launched in SBB's
research and innovation lab.8
3.9</p>
        <sec id="sec-3-6-1">
          <title>Ubermetrics Technologies GmbH: Curation Technologies for the Monitoring of Online Content and Risks</title>
          <p>Ubermetrics is a leading provider of cloud-based social media monitoring and
content intelligence software. Ubermetrics analyses public data from online,
print, TV and radio sources with a proprietary technology to identify critical
information in order to help organizations to optimize decision processes and
increase their performance (see Figure 7).</p>
          <p>In QURATOR, Ubermetrics researches how to use social media for the
monitoring of both external and internal risks. The focus areas are an easy setup
of risk-related search queries thanks to automated query suggestions and a
condensation of the results found with the help of summarization and duplicate
detection technology. The project aims at showing the developed capabilities in
demonstrators to get feedback for a later product for risk monitoring.</p>
          <p>Automatic connections to risk related sources have already been developed
and a rst version of query suggestions is available. The next steps are to improve
the query suggestions especially in the risk context and start with the research
and development of text summarization methods.</p>
        </sec>
        <sec id="sec-3-6-2">
          <title>Wikimedia Deutschland e. V.: Data quality in Wikidata</title>
          <p>Wikidata is Wikimedia's knowledge base. It is a sister project of Wikipedia and
collects general purpose data about the world. Wikidata currently describes over</p>
        </sec>
      </sec>
      <sec id="sec-3-7">
        <title>8 https://lab.sbb.berlin</title>
        <p>63 million entities such as people, geographic entities, events and works of art.
Wikidata, just like Wikipedia, is built by a community { currently consisting of
more than 20,000 editors from all around the world { that collects and maintains
that data. Wikidata's data powers a large number of applications, among them
search engine instant answers, digital personal assistants, educational websites,
as well as information boxes on Wikipedia articles. By its nature, Wikidata is an
open project. It relies on contributions from volunteer editors. To build a large
enough community for building and maintaining a general purpose knowledge
base the entry barrier needs to be low. At the same time the pressure to provide
high-quality data is increasing as more and more people are exposed to its data
in their day-to-day life. It is vital for the long-term sustainability of Wikidata to
nd ways to stay open while keeping the quality of its data high. On top of that
Wikidata can only follow its mission of giving more people more access to more
knowledge if the data is easily accessible for everyone. In QURATOR, we work
on improving both the quality and accessibility of the data in Wikidata, which
is supposed to become a viable basic building block of the QURATOR platform
by providing easily accessible high-quality data for all partners.</p>
        <p>So far three important components with a focus on quality improvements
have been developed. The rst presents a way to de ne schemas for data in
order to allow editors to quickly nd data that does not conform to the speci ed
schema. It is based on the Shape Expression standard (see Figure 8). The second
entails the ability to automatically judge the quality of a data item using machine
learning. Editors can then nd especially high and low quality data items to
showcase and improve them respectively. The third is an improved way to add
references for individual data points to improve the veri ability of the data.</p>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>Summary and Next Steps</title>
      <p>This paper provides a snapshot of the technologies and approaches developed
as part of the QURATOR project. Its vision is to o er a broad portfolio of
integrated solutions to the cross-industry challenges that are associated with the
curation of digital content. A platform strategy has been developed to transform
fragmented market areas for curation technologies into a new stand-alone market
and greatly expand it by displacing existing isolated solutions. QURATOR aims
to establish an ecosystem for curation technologies that improve the state of the
art and transform the Berlin-Brandenburg area into a global center of excellence
for curation technologies and the development of e cient industrial applications.</p>
    </sec>
    <sec id="sec-5">
      <title>Acknowledgements</title>
      <p>The research presented in this article is funded by the German Federal
Ministry of Education and Research (BMBF) through the project QURATOR
(Unternehmen Region, Wachstumskern, grant no. 03WKDA1A). http://qurator.ai</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          1.
          <string-name>
            <surname>Berger</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          :
          <article-title>Archive zum Sprechen Bringen { Semantic Storytelling oder der Redaktionswork ow der Zukunft</article-title>
          . In: 23.
          <string-name>
            <surname>Berliner</surname>
          </string-name>
          <article-title>Veranstaltung der Internationalen EVASerie Electronic Media</article-title>
          and
          <string-name>
            <given-names>Visual</given-names>
            <surname>Arts</surname>
          </string-name>
          . pp.
          <volume>135</volume>
          {
          <issue>141</issue>
          (
          <year>2017</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          2.
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Moreno-Schneider</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nehring</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sasaki</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Srivastava</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          :
          <article-title>Towards a Platform for Curation Technologies: Enriching Text Collections with a Semantic-Web Layer</article-title>
          . In: Sack,
          <string-name>
            <given-names>H.</given-names>
            ,
            <surname>Rizzo</surname>
          </string-name>
          ,
          <string-name>
            <given-names>G.</given-names>
            ,
            <surname>Steinmetz</surname>
          </string-name>
          ,
          <string-name>
            <given-names>N.</given-names>
            ,
            <surname>Mladeni</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            ,
            <surname>Auer</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            ,
            <surname>Lange</surname>
          </string-name>
          , C. (eds.)
          <article-title>The Semantic Web</article-title>
          . pp.
          <volume>65</volume>
          {
          <fpage>68</fpage>
          . No. 9989
          <source>in Lecture Notes in Computer Science</source>
          , Springer (
          <year>June 2016</year>
          ),
          <article-title>eSWC 2016 Satellite Events</article-title>
          . Heraklion, Crete, Greece, May 29 { June 2, 2016 Revised Selected Papers
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          3.
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sasaki</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          :
          <article-title>Processing Document Collections to Automatically Extract Linked Data: Semantic Storytelling Technologies for Smart Curation Work ows</article-title>
          . In: Gangemi,
          <string-name>
            <given-names>A.</given-names>
            ,
            <surname>Gardent</surname>
          </string-name>
          , C. (eds.)
          <source>Proceedings of the 2nd International Workshop on Natural Language Generation and the Semantic Web (WebNLG</source>
          <year>2016</year>
          ). pp.
          <volume>13</volume>
          {
          <fpage>16</fpage>
          .
          <article-title>The Association for Computational Linguistics</article-title>
          , Edinburgh, UK (
          <year>September 2016</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          4.
          <string-name>
            <surname>Devlin</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Chang</surname>
            ,
            <given-names>M.W.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lee</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Toutanova</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          :
          <article-title>Bert: Pre-training of Deep Bidirectional Transformers for Language Understanding</article-title>
          . arXiv preprint arXiv:
          <year>1810</year>
          .
          <volume>04805</volume>
          (
          <year>2018</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          5.
          <string-name>
            <surname>He</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zhang</surname>
            ,
            <given-names>X.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Ren</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sun</surname>
          </string-name>
          , J.:
          <article-title>Deep Residual Learning for Image Recognition</article-title>
          .
          <source>In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>
          . pp.
          <volume>770</volume>
          {
          <issue>778</issue>
          (
          <year>2016</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          6.
          <string-name>
            <surname>Labusch</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Neudecker</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          , Zellhofer, D.:
          <article-title>BERT for Named Entity Recognition in Contemporary and Historic German</article-title>
          .
          <source>In: Preliminary Proceedings of the 15th Conference on Natural Language Processing (KONVENS</source>
          <year>2019</year>
          )
          <article-title>: Long Papers</article-title>
          . pp.
          <volume>1</volume>
          {
          <issue>9</issue>
          . German Society for Computational Linguistics &amp; Language
          <string-name>
            <surname>Technology</surname>
          </string-name>
          , Erlangen, Germany (
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          7.
          <string-name>
            <surname>Moreno-Schneider</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Srivastava</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Wabnitz</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
          </string-name>
          , G.:
          <article-title>Semantic Storytelling, Cross-lingual Event Detection and other Semantic Services for a Newsroom Content Curation Dashboard</article-title>
          . In: Popescu,
          <string-name>
            <given-names>O.</given-names>
            ,
            <surname>Strapparava</surname>
          </string-name>
          , C. (eds.)
          <source>Proc. of the Second Workshop on Natural Language Processing meets Journalism { EMNLP 2017 Workshop (NLPMJ</source>
          <year>2017</year>
          ). pp.
          <volume>68</volume>
          {
          <fpage>73</fpage>
          . Copenhagen, Denmark (
          <year>2017</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          8.
          <string-name>
            <surname>Neudecker</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Baierer</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Federbusch</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Boenig</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          , Wurzner,
          <string-name>
            <given-names>K.M.</given-names>
            ,
            <surname>Hartmann</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            ,
            <surname>Herrmann</surname>
          </string-name>
          , E.:
          <string-name>
            <surname>OCR-D</surname>
          </string-name>
          :
          <article-title>An End-to-end Open Source OCR Framework for Historical Printed Documents</article-title>
          .
          <source>In: Proceedings of the 3rd International Conference on Digital Access to Textual Cultural Heritage</source>
          . pp.
          <volume>53</volume>
          {
          <fpage>58</fpage>
          . DATeCH2019, ACM, New York, NY, USA (
          <year>2019</year>
          ). https://doi.org/10.1145/3322905.3322917, http://doi.acm.
          <source>org/10</source>
          .1145/3322905.3322917
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          9.
          <string-name>
            <surname>Ostendor</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Berger</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Moreno-Schneider</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
          </string-name>
          , G.:
          <article-title>Enriching BERT with Knowledge Graph Embeddings for Document Classi cation</article-title>
          . In: Remus,
          <string-name>
            <given-names>S.</given-names>
            ,
            <surname>Aly</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            ,
            <surname>Biemann</surname>
          </string-name>
          , C. (eds.)
          <source>Proceedings of the GermEval Workshop</source>
          <year>2019</year>
          {
          <article-title>Shared Task on the Hierarchical Classi cation of Blurbs</article-title>
          . Erlangen, Germany (10
          <year>2019</year>
          ), 8 October 2019
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          10.
          <string-name>
            <surname>Paschke</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Athan</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sottara</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kendall</surname>
            ,
            <given-names>E.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bell</surname>
          </string-name>
          , R.:
          <article-title>A Representational Analysis of the API4KP Metamodel</article-title>
          . In: International Workshop Formal Ontologies Meet Industries. pp.
          <volume>1</volume>
          {
          <fpage>12</fpage>
          . Springer (
          <year>2015</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          11.
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Berger</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Elsholz</surname>
            ,
            <given-names>E.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hegele</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kintzel</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Marheinecke</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Piperidis</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Deligiannis</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Galanis</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gkirtzou</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Labropoulou</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bontcheva</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Jones</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Roberts</surname>
            ,
            <given-names>I.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hajic</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hamrlova</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Kacena</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Choukri</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Arranz</surname>
            ,
            <given-names>V.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Mapelli</surname>
            ,
            <given-names>V.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Vasiljevs</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Anvari</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lagzdins</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Melnika</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Backfried</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Dikici</surname>
            ,
            <given-names>E.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Janosik</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Prinz</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Prinz</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Stampler</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>ThomasAniola</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Perez</surname>
            ,
            <given-names>J.M.G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Silva</surname>
            ,
            <given-names>A.G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Berrio</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Germann</surname>
            ,
            <given-names>U.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Renals</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Klejch</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          :
          <article-title>European Language Grid: An Overview (</article-title>
          <year>2020</year>
          ), submitted to LREC 2020. Marseille, France.
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          12.
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>He</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nehring</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Quantz</surname>
          </string-name>
          , J.:
          <article-title>Designing User Interfaces for Curation Technologies</article-title>
          . In: Yamamoto,
          <string-name>
            <surname>S</surname>
          </string-name>
          . (ed.)
          <article-title>Human Interface and the Management of Information: Information, Knowledge and Interaction Design</article-title>
          , 19th International Conference,
          <source>HCI International</source>
          <year>2017</year>
          (Vancouver, Canada). pp.
          <volume>388</volume>
          {
          <fpage>406</fpage>
          . No. 10273
          <source>in Lecture Notes in Computer Science (LNCS)</source>
          , Springer, Cham,
          <source>Switzerland (July</source>
          <year>2017</year>
          ), part I
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          13.
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lee</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          :
          <article-title>Curation Technologies for a Cultural Heritage Archive: Analysing and Transforming a Heterogeneous Data Set into an Interactive Curation Workbench</article-title>
          . In: Antonacopoulos,
          <string-name>
            <given-names>A.</given-names>
            ,
            <surname>Bechler</surname>
          </string-name>
          , M. (eds.)
          <source>Proceedings of DATeCH</source>
          <year>2019</year>
          :
          <article-title>Digital Access to Textual Cultural Heritage</article-title>
          . Brussels, Belgium (May
          <year>2019</year>
          ),
          <fpage>8</fpage>
          -10 May
          <year>2019</year>
          . In print.
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          14.
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sasaki</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          :
          <article-title>Digitale Kuratierungstechnologien { Verfahren fur die Efziente Verarbeitung, Erstellung und Verteilung Qualitativ Hochwertiger Medieninhalte</article-title>
          . In: Proceedings der Fru
          <article-title>hjahrstagung der Gesellschaft fur Sprachtechnologie und Computerlinguistik (GSCL</article-title>
          <year>2015</year>
          ). pp.
          <volume>138</volume>
          {
          <fpage>139</fpage>
          .
          <string-name>
            <surname>Duisburg</surname>
          </string-name>
          (9
          <year>2015</year>
          ),
          <fpage>30</fpage>
          . September{2. Oktober
        </mixed-citation>
      </ref>
      <ref id="ref15">
        <mixed-citation>
          15.
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Srivastava</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Fricke</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Thomsen</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>He</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Quantz</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Berger</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          , Konig, L., Rauchle,
          <string-name>
            <given-names>S.</given-names>
            ,
            <surname>Gerth</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            ,
            <surname>Wabnitz</surname>
          </string-name>
          ,
          <string-name>
            <surname>D.</surname>
          </string-name>
          :
          <article-title>Di erent Types of Automated and Semi-Automated Semantic Storytelling: Curation Technologies for Di erent Sectors</article-title>
          . In: Rehm,
          <string-name>
            <given-names>G.</given-names>
            ,
            <surname>Declerck</surname>
          </string-name>
          , T. (eds.)
          <article-title>Language Technologies for the Challenges of the Digital Age: 27th International Conference</article-title>
          ,
          <source>GSCL 2017</source>
          , Berlin, Germany,
          <source>September 13-14</source>
          ,
          <year>2017</year>
          , Proceedings. pp.
          <volume>232</volume>
          {
          <fpage>247</fpage>
          . No. 10713
          <source>in Lecture Notes in Arti cial Intelligence (LNAI)</source>
          ,
          <article-title>Gesellschaft fur Sprachtechnologie und Computerlinguistik e</article-title>
          .V., Springer, Cham, Switzerland (
          <year>January 2018</year>
          ),
          <volume>13</volume>
          /
          <issue>14</issue>
          <year>September 2017</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref16">
        <mixed-citation>
          16.
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Srivastava</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nehring</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Berger</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          , Konig, L., Rauchle,
          <string-name>
            <given-names>S.</given-names>
            ,
            <surname>Gerth</surname>
          </string-name>
          , J.:
          <article-title>Event Detection and Semantic Storytelling: Generating a Travelogue from a large Collection of Personal Letters</article-title>
          . In: Caselli,
          <string-name>
            <given-names>T.</given-names>
            ,
            <surname>Miller</surname>
          </string-name>
          ,
          <string-name>
            <surname>B.</surname>
          </string-name>
          , van Erp,
          <string-name>
            <given-names>M.</given-names>
            ,
            <surname>Vossen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            ,
            <surname>Palmer</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            ,
            <surname>Hovy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>E.</given-names>
            ,
            <surname>Mitamura</surname>
          </string-name>
          , T. (eds.)
          <source>Proc. of the Events and Stories in the News Workshop</source>
          . pp.
          <volume>42</volume>
          {
          <fpage>51</fpage>
          . Association for Computational Linguistics, Vancouver, Canada (
          <year>August 2017</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref17">
        <mixed-citation>
          17.
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zaczynska</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          : Semantic Storytelling:
          <article-title>Towards Identifying Storylines in Large Amounts of Text Content</article-title>
          . In: Jorge,
          <string-name>
            <given-names>A.</given-names>
            ,
            <surname>Campos</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            ,
            <surname>Jatowt</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            ,
            <surname>Bhatia</surname>
          </string-name>
          , S. (eds.)
          <source>Proc. of Text2Story { Second Workshop on Narrative Extraction From Texts co-located with 41th European Conf. on Information Retrieval (ECIR</source>
          <year>2019</year>
          ). pp.
          <volume>63</volume>
          {
          <fpage>70</fpage>
          . Cologne, Germany (April
          <year>2019</year>
          ), 14 April 2019
        </mixed-citation>
      </ref>
      <ref id="ref18">
        <mixed-citation>
          18.
          <string-name>
            <surname>Ronneberger</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Fischer</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Brox</surname>
          </string-name>
          , T.:
          <article-title>U-net: Convolutional Networks for Biomedical Image Segmentation</article-title>
          .
          <source>In: International Conference on Medical Image Computing and Computer-assisted Intervention</source>
          . pp.
          <volume>234</volume>
          {
          <fpage>241</fpage>
          . Springer (
          <year>2015</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref19">
        <mixed-citation>
          19.
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nehring</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sasaki</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Srivastava</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          :
          <article-title>Towards Semantic Story Telling with Digital Curation Technologies</article-title>
          . In: Birnbaum,
          <string-name>
            <given-names>L.</given-names>
            ,
            <surname>Popescu</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            ,
            <surname>Strapparava</surname>
          </string-name>
          , C. (eds.)
          <source>Proceedings of Natural Language Processing Meets Journalism { IJCAI-16 Workshop (NLPMJ</source>
          <year>2016</year>
          ). New York (
          <year>July 2016</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref20">
        <mixed-citation>
          20.
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
          </string-name>
          , G.:
          <article-title>Towards User Interfaces for Semantic Storytelling</article-title>
          . In: Yamamoto,
          <string-name>
            <surname>S</surname>
          </string-name>
          . (ed.)
          <article-title>Human Interface and the Management of Information: Information, Knowledge and Interaction Design, 19th</article-title>
          <string-name>
            <surname>Int. Conf.</surname>
          </string-name>
          ,
          <source>HCI International</source>
          <year>2017</year>
          (Vancouver, Canada). pp.
          <volume>403</volume>
          {
          <fpage>421</fpage>
          . No. 10274
          <source>in Lecture Notes in Computer Science (LNCS)</source>
          , Springer, Cham,
          <source>Switzerland (July</source>
          <year>2017</year>
          ), part II
        </mixed-citation>
      </ref>
      <ref id="ref21">
        <mixed-citation>
          21.
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
          </string-name>
          , G.:
          <article-title>Curation Technologies for the Construction and Utilisation of Legal Knowledge Graphs</article-title>
          . In: Rehm,
          <string-name>
            <given-names>G.</given-names>
            ,
            <surname>Rodriguez-Doncel</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            ,
            <surname>Schneider</surname>
          </string-name>
          ,
          <string-name>
            <surname>J.M</surname>
          </string-name>
          . (eds.)
          <source>Proc. of the LREC 2018 Workshop on Language Resources</source>
          and
          <article-title>Technologies for the Legal Knowledge Graph</article-title>
          . pp.
          <volume>23</volume>
          {
          <fpage>29</fpage>
          .
          <string-name>
            <surname>Miyazaki</surname>
          </string-name>
          , Japan (May
          <year>2018</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref22">
        <mixed-citation>
          22.
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
          </string-name>
          , G.:
          <article-title>Towards a Work ow Manager for Curation Technologies in the Legal Domain</article-title>
          . In: Rehm,
          <string-name>
            <given-names>G.</given-names>
            ,
            <surname>Rodriguez-Doncel</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            ,
            <surname>Schneider</surname>
          </string-name>
          ,
          <string-name>
            <surname>J.M</surname>
          </string-name>
          . (eds.)
          <source>Proc. of the LREC 2018 Workshop on Language Resources</source>
          and
          <article-title>Technologies for the Legal Knowledge Graph</article-title>
          . pp.
          <volume>30</volume>
          {
          <fpage>35</fpage>
          .
          <string-name>
            <surname>Miyazaki</surname>
          </string-name>
          , Japan (May
          <year>2018</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref23">
        <mixed-citation>
          23.
          <string-name>
            <surname>Schneider</surname>
            ,
            <given-names>J.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Roller</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hegele</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
          </string-name>
          , G.:
          <article-title>Towards the Automatic Classi cation of O ensive Language and Related Phenomena in German Tweets</article-title>
          . In: Ruppenhofer,
          <string-name>
            <given-names>J.</given-names>
            ,
            <surname>Siegel</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            ,
            <surname>Wiegand</surname>
          </string-name>
          , M. (eds.)
          <source>Proceedings of the GermEval Workshop</source>
          <year>2018</year>
          {
          <article-title>Shared Task on the Identi cation of O ensive Language</article-title>
          . pp.
          <volume>95</volume>
          {
          <fpage>103</fpage>
          . Vienna, Austria (
          <year>September 2018</year>
          ), 21 September 2018
        </mixed-citation>
      </ref>
      <ref id="ref24">
        <mixed-citation>
          24.
          <string-name>
            <surname>Springmann</surname>
            ,
            <given-names>U.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Reul</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Dipper</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Baiter</surname>
          </string-name>
          , J.:
          <article-title>Ground Truth for Training OCR Engines on Historical Documents in German Fraktur and Early Modern Latin</article-title>
          . arXiv preprint arXiv:
          <year>1809</year>
          .
          <volume>05501</volume>
          (
          <year>2018</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref25">
        <mixed-citation>
          25.
          <string-name>
            <surname>Srivastava</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Sasaki</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bourgonje</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Moreno-Schneider</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nehring</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Rehm</surname>
          </string-name>
          , G.:
          <article-title>How to Con gure Statistical Machine Translation with Linked Open Data Resources</article-title>
          . In:
          <string-name>
            <surname>Esteves-Ferreira</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Macan</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Mitkov</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Stefanov</surname>
            ,
            <given-names>O.M.</given-names>
          </string-name>
          <article-title>(eds</article-title>
          .)
          <source>Proceedings of Translating and the Computer</source>
          <volume>38</volume>
          (
          <issue>TC38</issue>
          ). pp.
          <volume>138</volume>
          {
          <fpage>148</fpage>
          .
          <string-name>
            <surname>Editions</surname>
            <given-names>Tradulex</given-names>
          </string-name>
          , London, UK (November
          <year>2016</year>
          ), http://www.asling.org/tc38/
        </mixed-citation>
      </ref>
      <ref id="ref26">
        <mixed-citation>
          26.
          <string-name>
            <surname>Wick</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Reul</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Puppe</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          :
          <article-title>Calamari-A High-Performance Tensor ow-based Deep Learning Package for Optical Character Recognition</article-title>
          . arXiv preprint arXiv:
          <year>1807</year>
          .
          <year>02004</year>
          (
          <year>2018</year>
          )
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>