<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>A computational framework for the analysis of the Uruguayan dictatorship archives</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Lorena Etcheverry</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Leopoldo Agorio</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Virginia Bacigalupe</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sof a Barreiro</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Elena Bing</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Samuel Blixen</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Calegari</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lautaro Cardozo</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Fernando Carpani</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Felipe Chavat</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Diego Garat</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alvaro Gomez</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Fernandez</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Federico Fioritto</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Fabian Hernandez</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Rodrigo Laguna</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Victor Marabotto</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Guillermo Moncecchi</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ignacio Ram rez</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Aiala Rosa</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Javier Stabile</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jorge Tiscornia</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Nilo Patin~o</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>L a Rivero</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Dina Wonsever</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Guillermo Zorron</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Gregory Randall</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Facultad de Informacion y Comunicacion, Universidad de la Republica</institution>
          ,
          <addr-line>Montevideo</addr-line>
          ,
          <country country="UY">Uruguay</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Instituto de Computacion, Facultad de Ingenier a, Universidad de la Republica</institution>
          ,
          <addr-line>Montevideo</addr-line>
          ,
          <country country="UY">Uruguay</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Instituto de Ingenier a Electrica, Facultad de Ingenier a, Universidad de la Republica</institution>
          ,
          <addr-line>Montevideo</addr-line>
          ,
          <country country="UY">Uruguay</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Madres y Familiares de Uruguayos Detenidos Desaparecidos</institution>
          ,
          <addr-line>Montevideo</addr-line>
          ,
          <country country="UY">Uruguay</country>
        </aff>
      </contrib-group>
      <abstract>
        <p>Between 1973 and 1985, a civic-military dictatorship ruled in Uruguay. Systematic violations of human rights marked this period. Project Cruzar.uy aims to develop tools and methodologies to analyze historical documents from that period. We present the advances in this ongoing project. We describe a set of tools to automatize the extraction and organization of information from the archives using computational tools including image processing, machine learning, natural language processing, information extraction and integration.</p>
      </abstract>
      <kwd-group>
        <kwd>Documents analysis and recognition</kwd>
        <kwd>OCR</kwd>
        <kwd>historical documents retrieval</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>-</title>
      <p>Between 1973 and 1985, a civic-military dictatorship ruled in Uruguay,
preceded by state terrorism from 1968 to 1973. Systematic violations of human
rights, including the use of torture, raping, kidnapping, and the forced
disappearance of hundreds of Uruguayans marked this period. The investigation of
these crimes has been hampered by a complicity web involving military
personnel, institutions, and politicians involved in those events, obstructing the access
to the di erent sources of information that still exist from that period. However,
over the years, access to some data sets has been achieved. One is the so-called
Archivo Berrutti, which contains approximately 3 million pages of diverse
material produced by the security agencies during the dictatorship and ensuing
years. The documents in this collection consist of digital scans of micro lm reels
of the original paper documents, no longer available. Other collections are
partially digitized: e.g. the Historical Archive of the former National Directorate of
Information and Intelligence, and the Archive of the Naval Fusiliers Corps.</p>
      <p>The documents in these archives are heterogeneous. Among their contents are
interrogation reports, press clippings, lists of people and places, personal records,
pictures, passports, political a liation, and any other background information
deemed useful by the security agencies. These documents also contain codes
that allow them to be related to additional les. However, without collaboration
from the military, such codes' meaning is unclear and of small help for organizing
them. A similar situation occurs with codenames used to designate places and
targets.</p>
      <p>
        These archives' contents are precious to understand this period of Uruguayan
history better, assist in searching for missing persons, contribute elements to the
ongoing trials, and build citizenship and awareness of this historical period. At
least two projects focused on analyzing this documentation and related
information of the dictatorship period [
        <xref ref-type="bibr" rid="ref22 ref4">4,22</xref>
        ]. In both cases, transcription and analysis
were manually performed on a tiny subset of the archives to the best of our
knowledge. Unfortunately, this methodology is not su cient to process the
millions of document pages that are yet to be analyzed.
      </p>
      <p>More than two years ago, the project Cruzar.uy was formulated to develop
a methodology capable of processing the massive amount of information in the
military archives in a relatively short time. This interdisciplinary project involves
researchers and collaborators ranging from electrical engineers and computer
scientists to journalists, archivists, historians, and members of Uruguayan human
rights organizations. The present paper presents the advances in this ongoing
project. In particular, we describe a set of tools that we are deploying to
automatize the extraction and organization of information from the archives employing
computational tools and techniques including image processing, machine
learning, and natural language processing (NLP), and information extraction and
integration.
2</p>
    </sec>
    <sec id="sec-2">
      <title>Related work</title>
      <p>
        From a technical point of view, our work falls within the trans-disciplinary eld
of Digital Humanities, where digital technology and computational methods are
applied to humanities research [
        <xref ref-type="bibr" rid="ref7">7</xref>
        ]. In this context, some works explore the
design of tools that help researchers to nd historical documents [
        <xref ref-type="bibr" rid="ref24 ref25">24,25</xref>
        ], following
a similar approach to the one proposed in the current project. It is also worth
mentioning the International Consortium of Investigative Journalists' work that
analyzes the Panama Papers [
        <xref ref-type="bibr" rid="ref14">14</xref>
        ]. In the current project context, it is crucial
to extract events and facts together with temporal information; related work on
temporal information retrieval is collected in [
        <xref ref-type="bibr" rid="ref8">8</xref>
        ]. Latin American countries have
been subject to several dictatorships during the XX century. The analysis of the
documents produced during those periods has been carried on manually by
Human Rights organizations, in general with little support from governments. As
a consequence, the technical resources available for such analysis are minimal.
Some exceptions include the Equipo Argentino de Arqueolog a Forense with
extensive experience and known successes in forensic approaches in Argentina and
many countries with similar situations [
        <xref ref-type="bibr" rid="ref12">12</xref>
        ]. The Comision Provincial de Memoria
analyses the Police Intelligence Directorate's archive of the Province of Buenos
Aires, Argentina. In 1992, in Paraguay, Mart n Almada found what is known as
the \terror archives": a substantial body of documents related to the Stroessner
dictatorship and the \Plan Condor" that articulated the repressive actions of
several South American countries during the 1970s and 1980s[
        <xref ref-type="bibr" rid="ref2">2</xref>
        ]. The project
Memoria Ojo Publico has analyzed extensive documentation about the state
terrorism during the dictatorships in Peru' [
        <xref ref-type="bibr" rid="ref1">1</xref>
        ]. An international e ort, based in
the University of Texas at Austin, analyzes about 80 million digitized documents
obtained from the Guatemalan National Police Historical Archive [
        <xref ref-type="bibr" rid="ref26">26</xref>
        ].
      </p>
      <p>The preceding examples share similarities with the one we are dealing with
at Cruzar. In all these cases, the archives have been digitized, processed, and
analyzed manually to the best of our knowledge. No tools have been developed
to automatize those time-consuming tasks, which could signi cantly boost the
capacity to extract and correlate information that could enable more in-depth
and broader investigations to occur.
3</p>
    </sec>
    <sec id="sec-3">
      <title>Our approach</title>
      <p>The Cruzar Project's goal is to systematize and organize the aforementioned
historical documents to maximize the quantity and quality of extracted
knowledge. Hopefully, such information will allow us to understand the mechanisms
and operation of the repressive system, to identify the processes that led to the
disappearance of prisoners (which could,in turn, allow to nd their remnants
and provide solace to their relatives), and help in the process of making justice.
Contributing to the understanding of that historical period may help to avoid
future repetitions.</p>
      <p>In this context, we are continually building and improving information
systems to help researchers nd, extract and discover useful information from the
documents and their relationships. Our approach combines tools and techniques
from Image Processing, Computer Vision, NLP and Knowledge Graphs to deal
with several use cases. These include nding all the documents that mention a
speci c person, place, or organization; building timelines on people, places and
facts; and reconstructing the repressive corps' organization charts. Controlled
vocabularies and ontologies have a crucial role in our approach, guiding the
information extraction process and assisting in data integration tasks. Given the
nature of this project, to guarantee the traceability of the information, it is
necessary to maintain the link between the transcribed text, the information
subsequently extracted, and the original documents at all times..</p>
      <p>Figure 1 shows an overview of our strategy. The data preparation stage
consists of several image pre-processing tasks, targeted at improving image quality,
and classifying images according to di erent criteria (e.g., document type, image</p>
      <p>Transcription
Extract text from images.</p>
      <p>Automatic and
humanassisted</p>
      <p>Integration
Integrate obtained graphs.</p>
      <p>Perform entity-matching
1
2
3
4
5</p>
      <p>Preparation
Image pre-processing
tasks</p>
      <p>Extraction
Extract information from
text (entities and relations)
and store it as graphs.</p>
      <p>Provenance</p>
      <p>Exploration
Navigate the graph.</p>
      <p>
        Maintaining the link between transcribed text, extracted and derived information,  and the original documents
quality, etc.). The text contained in the digitized document images is obtained
in the transcription stage. The resulting text is stored in a relational database,
keeping track of data provenance as the relationship between the source images
and the produced text. Several information extraction tasks are performed in a
third stage, using NLP methods to extract Named Entities and relations.
Ontologies and controlled vocabularies guide this process and are used to annotate
and represent the extracted assertions; these are stored as Resource
Description Framework (RDF)[
        <xref ref-type="bibr" rid="ref28">28</xref>
        ]] triples, atomic pieces of knowledge encoded as three
entities (e.g., \subject-predicate-object"). Extracted triples are stored in a
triplestore: a database capable of handling RDF data. In this stage, we also keep track
of each assertion's provenance in terms of text segments. The integration stage
provides a uni ed view of the sub-graphs extracted from each document.
Entity Matching tasks produce integrated Knowledge Graphs required to navigate
statements pulled from di erent documents in this stage. Finally, exploration
tools combine di erent approaches as text-search, faceted-search based on the
ontology, and timelines to explore documents' collection. In the following
sections, we provide further details on each stage.
3.1
      </p>
      <sec id="sec-3-1">
        <title>Data preparation</title>
        <p>Our current work is focused on the so-called Berrutti Archive: an extensive
collection of documents found on military facilities while Azucena Berrutti was the
Uruguayan Defense Minister. This collection of micro lmed documents contains
around 3 million pages span from 1965 to 1999, arranged in micro lm rolls of
about 2500 images each. There is no apparent organization within each roll,
besides some rough chronological order related to each one. The original
documents include handwritten text, machine written pages, pictures, and portions
of printed material (as newspapers). The process of digitizing the micro lm rolls
took place long before this project started; the digitized images are the sole
material we have. Instead of colour, or even gray-scale, the digitized images are
binary (black or white, no shades in between); this poses a challenge to any
image processing or computer vision task (such as Optical Character Recognition).
Figure 2 shows some examples.</p>
        <p>
          Annotation The annotation stage's goal is to enrich documents with relevant
metadata such as the date, the type, the origin and eventually, a brief
description. Moreover, documents may have several pages, and we want to signal these
relations in the corresponding images. Most of the image annotation tools focus
on the segmentation of an image. In our case, the segmentation of regions of
interest (e.g. stamps, signatures) is important, but the complete annotation of
a document must include other metadata. Since the addressed documents have
sensitive content, web applications were not considered. Among the standalone
applications, we selected LabelMe [
          <xref ref-type="bibr" rid="ref23 ref29">23,29</xref>
          ] because it has both segmenting and
annotating tools and some basic global tagging. The open-source code was adapted
to include the features mentioned above speci c for documents.
        </p>
        <p>This customized version of LabelMe allows annotating di erent aspects of a
document, including its type, date, index of the page in a multi-page document,
among other things. We have identi ed about 74 di erent types of documents
(e.g.reports, letters, police les) and 52 di erent origins (e.g. divisions of the
army or security agencies). So far, over 140000 documents have been annotated
in this way by about 150 students and volunteers since 2019. Besides its direct
use as valuable information about the documents, the labeled documents also
constitute a reputable source of labeled data for training the various automatic
classi cation algorithms that we are currently working on. For example, LabelMe
lets the user select regions of the image containing stamps and signatures (see
Figure 3) to test automatic signature and stamp detection algorithms.
3.2</p>
      </sec>
      <sec id="sec-3-2">
        <title>Text Transcription and Image Readability Assessment</title>
        <p>Automatic transcription and readability assessment The low quality of the source
images implies that even sophisticated commercial OCRs are ine ective for a
signi cant part of the digitized archives. On the other hand, humans can easily
read several of these problematic documents. Based on the above observation,
we have implemented a hybrid manual-automatic transcription system which
consists of three stages. First, an automatic transcription is obtained using the
o -the-shelf, freely available Tesseract OCR (version 4.1) 5. Then, the quality of
the OCR output is estimated; this has to be done without access to the correct
transcription. If the quality is deemed satisfactory, the OCR transcription is
fed directly to the database. Otherwise, the document is queued for manual
transcription in the next step. Note that this stage is critical, as documents
which are incorrectly labeled as well transcribed will likely not be read by anyone
afterwards. Thus, we put a signi cant e ort to produce a reasonable estimation
of the document readability.</p>
        <p>Manual transcription { LUISA: If the OCR transcription of a document is
deemed unsatisfactory, the image is fed to a web-based crowd-sourced
transcription platform, named LUISA, which was developed by our team for this sole
purpose. We will now provide more detail on the last two stages.
Readability score Given the sheer number of documents to be processed, a
manual assessment of the readability of each document is out of reach. Instead, we
must rely on some automatic method for estimating such readability. To achieve
this, we developed a readability score r of the text produced by the OCR. We
use a custom enriched dictionary D, whose quality is crucial in computing such
score. Besides that, we need to ascertain that the score works as expected, at
least on a number of known cases for which we do have a complete transcription.
Given a text t formed by n words, t = fw1; w2; : : : ; wng, the score r(t) is de ned
as r(t) = PL</p>
        <p>l=1 alpl(t); where pl(t) is the fraction of words of length l in t that
are present in the dictionary D, and al is a coe cient to be determined.</p>
        <sec id="sec-3-2-1">
          <title>5 https://github.com/tesseract-ocr/tesseract</title>
          <p>A) &lt;cuexa ) Spoga e dd cdo Ct traslado echerada al 3% Le 2... nnldsa Lora 1630 OF) eouaada
Y oaun~tesia ue. 20 GApradoro hora 1719 An POvedad,y &gt; 334) imovuades &gt; 3 Lieoye obdaple
en Urala blorso y rear Sr cosmanics e) ina/ Procoue poreonal de -00Ce dd0. $1 luzer ol 131
2ategd. 230 esipvistdo a exvayo cuerda or dae porko 3rasoru y.
3: 11,15: da.qio efectua servicio de aniulancia desde el iterivr el rado al Sanaterio Casa de
talicia roer choque entre un camion y una moto. :raslada a :lisabeth canci &gt;lehans &lt;iartinoz bel
1.20kh.&lt;%6 61 estado
grave.- La fuente informa que los d as 11 y 13 del:corriente mes, en el local de dicho frente se real zara
un pequen~o cursillo sobre las TESIS de "LENIN" (tesis de abril). - Se comenta que invitaron
en forma abierta a todos los que quieran participar y debido a esto se espera la presencia del
M.L.N., Mov. 26 de Marzo y aquellas personas</p>
          <p>
            To determine the coe cients al, 50 documents were selected and divided
into ve groups of 10 documents each according to our subjective assessment
of readability (very bad, bad, mediocre, good, very good); this is illustrated in
Figure 4. We then manually transcribed those documents as best as possible,
and computed an ideal readability score based on the Levenshtein distance [
            <xref ref-type="bibr" rid="ref18">18</xref>
            ]
between the OCR and manual transcriptions. Concretely, given the ideal score
s(tj ) and the vector of proportions (p1; : : : ; pL)j computed for each j-th of the
N = 50 selected documents, the weights (a1; : : : ; aL) were estimated so that r(tj )
and s(tj ) were as close as possible employing a linear regression. Figure 5 shows
the performance of the adjusted score in terms of the best t to the ideal score,
and its relationship to our subjective quality assessment. Using a threshold of
60%, about 300000 images were deemed of enough quality to be fed directly to
the following stages.
          </p>
          <p>Enriched dictionary Since the dictionary determines the validity of a transcribed
word, it must include all the words that may appear in a document. In our case,
the list of valid words goes far beyond those found in o -the-shelf Spanish
dictionaries: names and surnames derived from many countries and languages,
including all possible spellings; idiosyncratic expressions; military jargon, acronyms,
abbreviations, as well as common misspellings of words.</p>
          <p>An initial dictionary contained words from a standard Spanish dictionary and
lists of names and surnames found online. A list of domain-speci c acronyms was
then added. Then, the OCR was used to transcribe a subset of the documents,
and produce a list of words which did not appear in the initial dictionary. This
list, of about 200000 words, was manually ltered in search for valid terms. In
this way we obtained 16000 additional words which were added to produce the
nal enriched dictionary.6
LUISA: a crowd-sourcing transcription platform The goal of LUISA (Leyendo
Unidos para Interpretar loS Archivos ) is to transcribe low quality documents
through collaborative networking. The tool is named after the Uruguayan
human rights activist LUISA Cuesta (Nebio Ariel Melo Cuesta's mother,
disappeared). We developed a crowd-sourcing approach in order to promote social
participation in the project. Users of this web application 7, which can be used
on both computers and smartphones, are asked to transcribe what they see at
small portions of the documents (blocks): letters, numbers, symbols, etc. Each
block of text is presented at least three times randomly to di erent users to have
a set of transcriptions for each one. Programs developed by our team
automatically align the images and generate these blocks. These are then combined to
obtain transcripts of the entire document using a majority vote mechanism to
determine each block of text's transcription. In its design, LUISA takes care of
the collaborators' privacy and considers the handled documents' sensitivity: no
information that could help identify people who have collaborated is kept,
neither whole documents are shown. LUISA's social impact is twofold: on the one
hand, it helps to recover the documents. On the other, it creates awareness in
6 The dictionary is freely available and can be downloaded from http://iie. ng.edu.</p>
          <p>uy/ nacho/data/luisa/luisa-dic.zip.
7 https://www. ng.edu.uy/mh/luisa/
the participants by confronting them with these materials' reality and allowing
them to collaborate practically with the search for truth and justice.</p>
          <p>Since its launch in May 2019, LUISA has transcribed more than 400,000
blocks. On average, the platform has 5,800 accesses per day. We estimate that
over 4,000 documents have been transcribed so far by more than 10,000
collaborators. Along with the original document and the OCR generated
transcription, these transcriptions constitute a third type of document that enriches the
database. LUISA is also useful to collect data to develop a tailored OCR system
or improve the quality of the OCR transcriptions. The texts generated
reconstructing the original pages using LUISA annotated blocks, constitute a
groundtruth set essential for applying natural language processing (NLP) techniques,
which are discussed next.
3.3</p>
        </sec>
      </sec>
      <sec id="sec-3-3">
        <title>Improving the quality of automatic transcriptions</title>
        <p>To improve the OCR output quality, we pursued two objectives: a) to correctly
assemble the manual transcriptions from LUISA to generate a ground truth
version of OCR documents and b) improve the outputs obtained by the OCR
using advanced correction techniques.</p>
        <p>Quality metric We use the output of the SequenceMatcher.ratio function of
di ib, 8 which produces scores in the interval [0; 1]. Given a reference R text,
the quality of a target text T is computed as follows. First, the longest common
subsequence (LCS) of contiguous characters between both strings is identi ed.
This sequence is extracted from both strings, leaving a pair of left sub-strings
and a pair of right sub-strings. Then, recursively perform the same procedure
between both left sub-strings and both right sub-strings. E.g., if R=\la casa de
su madre" and T =\la caso ce eu nadre" (both of jT j = jRj = 19 characters),
their LCS is C1: \la cas". The remaining left sub-strings are empty and the right
sub-strings are \a de su madre" and =\o ce eu nadre". The new LCS is \adre"
and the process continues until no new LCS can be found. The quality is given
as (2 sumijCij=(jRj + jT j)), that is, the ratio of the sum of the double of the
lengths of all LCSs to the sum of the lengths of the two sequences R and T . In
the preceding example, this value would be (6+4+2+1+2)=19 = 15=19 = 0:789.
For several texts, we report on the average of this metric.</p>
        <p>Reconstruction This stage takes the block transcriptions from LUISA and
rebuilds complete texts using each word's coordinates within the original
document. Issues such as identifying and joining erroneously separated blocks,
removing blocks belonging to stamps, and identifying line jumps between blocks
are taken care of in this process to obtain a faithful transcription. A preliminary
study of this procedure based on a set of 15 manually transcribed documents
yielded an average score of 0:87 for the ratio function explained above, which is
a good result.</p>
        <sec id="sec-3-3-1">
          <title>8 https://docs.python.org/3/library/di ib.html.</title>
          <p>
            Correction of OCR output Two di erent approaches were used for correcting the
OCR transcriptions: language models [
            <xref ref-type="bibr" rid="ref15 ref3">15,3</xref>
            ], and Statistical Machine Translation
(SMT) [
            <xref ref-type="bibr" rid="ref6">6</xref>
            ]. The rst case uses a Spanish language model and a dictionary of valid
words. The language model chooses a set of the words which are most likely
to occur within a given context, and these are further re ned as those in the
dictionary having a small Levenshtein [
            <xref ref-type="bibr" rid="ref18">18</xref>
            ] distance to the one being replaced.
Di erent language models were evaluated, including various n-gram models and
ELMO [
            <xref ref-type="bibr" rid="ref21">21</xref>
            ]. The best results were ultimately obtained using [
            <xref ref-type="bibr" rid="ref19">19</xref>
            ]. Figure 6 shows
an example of the n-gram approach. Some incorrect words are detected and
successfully corrected while others are erroneously replaced by another incorrect
word. In other cases, correct words are considered errors because they do not
belong to the dictionary and are therefore replaced. The OCR replaces some
words by others belonging to the dictionary, so the algorithm considers them
correct. Finally, some incorrect words remain undetected due to other reasons
(e.g., strange symbols). We use Moses [
            <xref ref-type="bibr" rid="ref17">17</xref>
            ] to implement the SMT approach. The
OCR and LUISA outputs were aligned to obtain a data set suitable for training
and testing, resulting in a set of 17460 examples. Figure 7 shows an example
of this approach. One advantage of SMT with respect to language models is
the ability to correct erroneous words that are erroneous even if listed in the
dictionary. On the other hand, STM can introduce words that are not present
in the OCR output nor in the source document. The average value of the ratio
mentioned above was computed on a test set of 544 documents. The result, when
comparing the unaltered OCR outputs to the ground truth was 0:543. After the
n-gram correction, this value was slightly improved to 0:550. However, the SMT
correction resulted in a sensibly low value of 0:514.
The aim of the LUZ sub-project9 aims to build a Knowledge Base (KB) that
provides an integrated view of the information extracted from the
transcriptions. This KB is designed following Linked Data principles, using Semantic
Web standards such as RDF and the Web Ontology Language (OWL) 10. The
KB construction is a two-stage process: i) information extraction (IE), and ii)
integration of the extracted information. Ontologies play a crucial role in both
stages.
          </p>
          <p>
            Information Extraction (IE The IE process comprises three sub-stages: Named
Entity Recognition (NER), Co-reference resolution, and Relation Extraction
(RE). Our strategy combines traditional NLP tools with the use of ontologies,
following a similar approach to the one proposed in [
            <xref ref-type="bibr" rid="ref10">10</xref>
            ]. In the NER stage,
persons, places, dates, organizations, and facts are recognized. We are currently
evaluating various NER modules from NLP tools like spaCy11, Stanford [
            <xref ref-type="bibr" rid="ref11">11</xref>
            ],
Freeling[
            <xref ref-type="bibr" rid="ref20">20</xref>
            ], Python Natural Language Toolkit (NLTK)12. The output of the
NER task is an RDF graph per document, based on the NLP Interchange
Format (NIF) Core ontology13. This graph represents structural elements of the
text, such as words and sentences. The annotation of these elements uses
concepts from our Content Ontology (CO). This ontology de nes the domain's key
concepts (such as Persons, Organizations, Places, and Events) and relevant
relations between them, such as Person pe is in Place pl at a certain Datetime dt, or
Person pe is member of Organization org during Interval i1. Our CO is strongly
9 Named after the Uruguayan human rights activist Luz Ibarburu. Also, luz means
light in Spanish.
10 Web Ontology Language https://www.w3.org/OWL/
11 https://spacy.io/
12 https://www.nltk.org/
13 NIF Core Ontology https://persistence.uni-leipzig.org/nlp2rdf/ontologies/nif-core/
nif-core.html
based on The Simple Event Model Ontology (SEM) [
            <xref ref-type="bibr" rid="ref27">27</xref>
            ]. Figure 8 depicts the
main classes and properties in our CO. To deal with co-reference between named
entities we developed a co-reference resolution module; the output of this module
is also represented in terms of the NIF Core Ontology. We are currently working
on a Relation Extraction module based on NLTK Relextract 14. This module
annotates existing graphs with relations from our CO. These modules have been
integrated into INCEpTION [
            <xref ref-type="bibr" rid="ref16">16</xref>
            ], a semantic annotation platform that o ers a
collaborative environment. This tool is particularly relevant during the curation
of the results from the extraction process, as di erent actors may interact with
the output. Our approach enables to control the semantic heterogeneity that
may arise during the extraction, while allowing the evolution of the underlying
vocabularies. As a byproduct, a new corpus is obtained for further improving
the performance of our IE modules.
          </p>
          <p>We now illustrate the Information Extraction process with an example. Let us
consider the following sentences taken from testimonies collected during trials 15
{ On October 5th 1976 ight 511 from Transporte Aereo Militar Uruguayo
(TAMU) ew from Buenos Aires to Montevideo, illegally transporting 22
uruguayan citizens kidnapped in Buenos Aires at\Orletti" detention center.
{ Mayor Walter Pintos was the pilot of ight 511 and Soldier Ernesto Soca
collaborated in Buenos Aires, at \Orletti" detention center.
{ Ernesto \Dracula" Soca tortured prisoners at\Automotores Orletti", in Buenos</p>
          <p>Aires.
14 https://www.nltk.org/ modules/nltk/sem/relextract.html
15 Ernesto Soca Prado's judicial le and sentence available at https://sitiosdememoria.
uy/index.php/causas/1242</p>
          <p>Listing 1.1. Sample RDF triples produced by the IE process
1 :PINTOS_Walter rdf:type :Person.
2 :SOCA_Ernesto rdf:type :Person.
3 :SOCA_Ernesto_Dracula rdf:type :Person.
4 :MONTEVIDEO rdf:type :Place
5 :ORLETTI rdf:type :Place
6 :ORLETTI_Automotores rdf:type :Place
7
8 :AR1 rdf:type :AgentInRole ;
9 :agentRoleHasAgent :PINTOS_Walter ;
10 :isMemberOf :Transporte_Aereo_Militar_Uruguayo ;
11 :agentRoleHasRole :Pilot .
12 :AR2 rdf:type :AgentInRol ;
13 :agentRoleHasAgent :SOCA_Ernesto;
14 :agentRoleHasRole :Collaborator .
15 :Inst1 rdf:type :Instant ;
16 :instantDate "1976-10-05T00:00:00"^^xsd:dateTime .
17 :Action1 rdf:type :Actions ;
18 :actionName "Flight 511 TAMU"
19 :hasAgent :AR1,:AR2;
20 :hasPlace :ORLETTI;
21 :hasPlace :MONTEVIDEO;
22 :hasTime :Inst1 .</p>
          <p>A subset of the triples extracted from these sentences is depicted in Listing 1.1
using the Turtle format.</p>
          <p>Integration and Entity Matching The use of the terms of our CO in all the graphs
produced by our Information Extraction process enforces a shared schema, which
reduces the integration process to Semantic Reconciliation, a.k.a. Semantic
Matching, Entity Matching, and Entity Resolution. In the example, the NER process
tags Mayor Walter Pintos, Soldier Ernesto Soca, and Ernesto \Dracula" Soca
as instances of class Person, while Automotores Orletti is tagged as a Place. The
RE process extracts the relations between these concepts and uses predicates
from the ontology to represent them. Then, semantic reconciliation allows us,
e.g., to identify that Soldier Ernesto Soca and Ernesto \Dracula" Soca are the
same person.</p>
          <p>
            The problem of Entity Resolution in the Semantic Web has received much
attention in the literature [
            <xref ref-type="bibr" rid="ref5 ref9">5,9</xref>
            ]. We are currently exploring di erent approaches
to identify corresponding entities. Using semantic web rules engines based on
SWRL [
            <xref ref-type="bibr" rid="ref13">13</xref>
            ], employing triplestores with support of entailment regimes and
inference forms like Stardog 16 or Virtuoso17, or inserting owl:sameAs triples in
another graph using query expressions that implement our equality conditions,
are under consideration.
16 https://www.stardog.com/
17 http://vos.openlinksw.com/owiki/wiki/VOS
          </p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>Conclusion and Future Work</title>
      <p>This ongoing project has created an environment that will facilitate the analysis
of millions of documents and therefore contribute to the clari cation of di erent
issues of the last dictatorship period and the advancement of the justice in
Uruguay. It is essential to use all the existent technological capabilities to advance
further and help the judges, journalists, historians, and other specialists do their
job and contribute to the development of citizen consciousness.</p>
      <p>Several work lines are open, among them: to use the data collected by LUISA
(blocks of images with their human transcripts) to build a tailored OCR; an
automatic method to detect stamps and signatures; the use of LabelMe annotations
to train an automatic classi cation algorithm. The integration of these utilities
in a system capable of managing a vast amount of documents and an adequate
human interface is crucial to facilitate the use by interdisciplinary teams devoted
to analyzing the data. Some of this work is being carried on today but is in its
early stages. Finally, all the methods and software developed for this project are
available to other communities dealing with similar problems searching for truth
and justice.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          1.
          <string-name>
            <surname>Amancio</surname>
            ,
            <given-names>N.L.</given-names>
          </string-name>
          :
          <article-title>Proyecto memoria: Datos contra el olvido (</article-title>
          <year>2017</year>
          ), https://memoria. ojo-publico.com/
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          2.
          <article-title>Los archivos del terror, contra la desmemoria y la cultura autoritaria de hoy (</article-title>
          <year>2018</year>
          ), https://www.nodalcultura.am/
          <year>2018</year>
          /12/archivos-del-terror/
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          3.
          <string-name>
            <surname>Bengio</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schwenk</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Senecal</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Morin</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gauvain</surname>
          </string-name>
          , J.:
          <source>Neural Probabilistic Language Models</source>
          , vol.
          <volume>194</volume>
          , pp.
          <volume>137</volume>
          {
          <fpage>189</fpage>
          . Springer Berlin Heidelberg (
          <year>2006</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          4.
          <string-name>
            <surname>Blixen</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          : Propuesta de Proyecto:
          <article-title>Sistematizacion, tratamiento y difusion de la informacion digital vinculada con las investigaciones en materia de graves violaciones a los derechos humanos en el pasado reciente</article-title>
          y terrorismo de Estado. (
          <year>2017</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          5. Bohm,
          <string-name>
            <given-names>C.</given-names>
            ,
            <surname>De Melo</surname>
          </string-name>
          ,
          <string-name>
            <given-names>G.</given-names>
            ,
            <surname>Naumann</surname>
          </string-name>
          ,
          <string-name>
            <given-names>F.</given-names>
            ,
            <surname>Weikum</surname>
          </string-name>
          , G.:
          <article-title>Linda: distributed web-ofdata-scale entity matching</article-title>
          .
          <source>In: Proc. of the 21st ACM Int. Conf. on Information and knowledge management</source>
          . pp.
          <volume>2104</volume>
          {
          <issue>2108</issue>
          (
          <year>2012</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          6.
          <string-name>
            <surname>Brown</surname>
            ,
            <given-names>P.F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Della Pietra</surname>
            ,
            <given-names>S.A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Della Pietra</surname>
            ,
            <given-names>V.J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Mercer</surname>
            ,
            <given-names>R.L.</given-names>
          </string-name>
          :
          <article-title>The mathematics of statistical machine translation: Parameter estimation</article-title>
          .
          <source>Computational Linguistics</source>
          <volume>19</volume>
          (
          <issue>2</issue>
          ),
          <volume>263</volume>
          {
          <fpage>311</fpage>
          (
          <year>1993</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          7.
          <string-name>
            <surname>Burdick</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Drucker</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lunenfeld</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Presner</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Schnapp</surname>
            ,
            <given-names>J.: Digital</given-names>
          </string-name>
          <string-name>
            <surname>Humanities</surname>
          </string-name>
          . Mit Press (
          <year>2012</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          8.
          <string-name>
            <surname>Campos</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Dias</surname>
            ,
            <given-names>G.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Jorge</surname>
            ,
            <given-names>A.M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Jatowt</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          :
          <article-title>Survey of temporal information retrieval and related applications</article-title>
          .
          <source>ACM Computing Surveys</source>
          <volume>47</volume>
          (
          <issue>2</issue>
          ),
          <volume>1</volume>
          {
          <fpage>41</fpage>
          (
          <year>2014</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          9.
          <string-name>
            <surname>Christophides</surname>
            ,
            <given-names>V.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Efthymiou</surname>
            ,
            <given-names>V.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Stefanidis</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          :
          <article-title>Entity resolution in the web of data</article-title>
          .
          <source>Synthesis Lectures on the Semantic Web</source>
          <volume>5</volume>
          (
          <issue>3</issue>
          ),
          <volume>1</volume>
          {
          <fpage>122</fpage>
          (
          <year>2015</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref10">
        <mixed-citation>
          10.
          <string-name>
            <surname>Erekhinskaya</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tatu</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Balakrishna</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Patel</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Strebkov</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Moldovan</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          :
          <article-title>Ten ways of leveraging ontologies for rapid natural language processing customization for multiple use cases in disjoint domains</article-title>
          .
          <source>Open Journal of Semantic Web</source>
          <volume>7</volume>
          (
          <issue>1</issue>
          ),
          <volume>33</volume>
          {
          <fpage>51</fpage>
          (
          <year>2020</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref11">
        <mixed-citation>
          11.
          <string-name>
            <surname>Finkel</surname>
            ,
            <given-names>J.R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Grenager</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Manning</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          :
          <article-title>Incorporating non-local information into information extraction systems by Gibbs sampling</article-title>
          .
          <source>In: Proc. of the 43rd Annual Meeting of ACL</source>
          . p.
          <volume>363</volume>
          {
          <fpage>370</fpage>
          .
          <string-name>
            <surname>USA</surname>
          </string-name>
          (
          <year>2005</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref12">
        <mixed-citation>
          12.
          <string-name>
            <surname>Goldschmidt</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          :
          <article-title>El equipo argentino de arqueolog a forense, orgullo nacional (</article-title>
          <year>2015</year>
          ), https://www.revistacitrica.
          <article-title>com/el-eaaf-orgullo-nacional</article-title>
          .html
        </mixed-citation>
      </ref>
      <ref id="ref13">
        <mixed-citation>
          13.
          <string-name>
            <surname>Horrocks</surname>
            ,
            <given-names>I.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Patel-Schneider</surname>
            ,
            <given-names>P.F.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Boley</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Tabet</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Grosof</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Dean</surname>
            ,
            <given-names>M.:</given-names>
          </string-name>
          <article-title>SWRL: A Semantic Web Rule Language Combining OWL and RuleML</article-title>
          .
          <source>Tech. rep., W3C</source>
          (
          <year>2004</year>
          ), http://www.w3.org/Submission/SWRL
        </mixed-citation>
      </ref>
      <ref id="ref14">
        <mixed-citation>
          14. ICIJ:
          <article-title>The Panama Papers: Exposing the Rogue O shore Finance Industry (</article-title>
          <year>2016</year>
          ), https://www.icij.org/investigations/panama-papers/
        </mixed-citation>
      </ref>
      <ref id="ref15">
        <mixed-citation>
          15.
          <string-name>
            <surname>Jurafsky</surname>
            ,
            <given-names>D.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Martin</surname>
            ,
            <given-names>J.H.</given-names>
          </string-name>
          : Speech and
          <string-name>
            <given-names>Language</given-names>
            <surname>Processing</surname>
          </string-name>
          .
          <article-title>An Introduction to Natural Language Processing, Computational Linguistics, and Speech Recognition. 2nd Edition, chap. 4: N-grams</article-title>
          .
          <source>Prentice Hall</source>
          (
          <year>2008</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref16">
        <mixed-citation>
          16.
          <string-name>
            <surname>Klie</surname>
            ,
            <given-names>J.C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bugert</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Boullosa</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>de</surname>
            <given-names>Castilho</given-names>
          </string-name>
          ,
          <string-name>
            <given-names>R.E.</given-names>
            ,
            <surname>Gurevych</surname>
          </string-name>
          ,
          <string-name>
            <surname>I.</surname>
          </string-name>
          :
          <article-title>The INCEpTION Platform: Machine-Assisted and Knowledge-Oriented Interactive Annotation</article-title>
          .
          <source>In: Proc. of the 27th Int. Conf. on Computational Linguistics: System Demonstrations</source>
          . pp.
          <volume>5</volume>
          {
          <issue>9</issue>
          . Association for Computational Linguistics (
          <year>June 2018</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref17">
        <mixed-citation>
          17.
          <string-name>
            <surname>Koehn</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Hoang</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Birch</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Callison-Burch</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Federico</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bertoldi</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Cowan</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Shen</surname>
            ,
            <given-names>W.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Moran</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zens</surname>
            ,
            <given-names>R.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Dyer</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Bojar</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Constantin</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Herbst</surname>
          </string-name>
          , E.: Moses:
          <article-title>Open source toolkit for statistical machine translation</article-title>
          .
          <source>In: Proc. of the 45th meeting of the ACL</source>
          . pp.
          <volume>177</volume>
          {
          <fpage>180</fpage>
          . Prague, Czech Republic (
          <year>Jun 2007</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref18">
        <mixed-citation>
          18.
          <string-name>
            <surname>Levenshtein</surname>
            ,
            <given-names>V.I.</given-names>
          </string-name>
          :
          <article-title>Binary codes capable of correcting deletions, insertions, and reversals</article-title>
          .
          <source>Soviet Physics Doklady</source>
          <volume>10</volume>
          (
          <issue>8</issue>
          ),
          <volume>707</volume>
          {
          <fpage>710</fpage>
          (
          <year>1966</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref19">
        <mixed-citation>
          19.
          <string-name>
            <surname>Lin</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Michel</surname>
            ,
            <given-names>J.B.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Aiden Lieberman</surname>
            ,
            <given-names>E.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Orwant</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Brockman</surname>
            ,
            <given-names>W.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Petrov</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          :
          <article-title>Syntactic annotations for the Google Books NGram corpus</article-title>
          .
          <source>In: Proc. of the ACL 2012 System Demonstrations</source>
          . pp.
          <volume>169</volume>
          {
          <fpage>174</fpage>
          . Association for Computational Linguistics (
          <year>Jul 2012</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref20">
        <mixed-citation>
          20.
          <string-name>
            <surname>Padro</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Stanilovsky</surname>
          </string-name>
          , E.:
          <article-title>Freeling 3.0: Towards wider multilinguality</article-title>
          .
          <source>In: Proc. of the Language Resources and Evaluation Conf</source>
          . ELRA (May
          <year>2012</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref21">
        <mixed-citation>
          21.
          <string-name>
            <surname>Peters</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Neumann</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Iyyer</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Gardner</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Clark</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Lee</surname>
            ,
            <given-names>K.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Zettlemoyer</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          :
          <article-title>Deep contextualized word representations</article-title>
          .
          <source>In: Proc. of the 2018 Conf. of the North American Chapter of the ACL</source>
          . pp.
          <volume>2227</volume>
          {
          <fpage>2237</fpage>
          . Association for Computational Linguistics, New Orleans,
          <source>Louisiana (Jun</source>
          <year>2018</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref22">
        <mixed-citation>
          22.
          <string-name>
            <surname>Rico</surname>
            ,
            <given-names>A</given-names>
          </string-name>
          . (ed.): Investigacion
          <string-name>
            <surname>Historica sobre Detenidos Desaparecidos - Tomo</surname>
            <given-names>I</given-names>
          </string-name>
          ,
          <article-title>Investigacion Historica sobre Detenidos Desaparecidos</article-title>
          , vol.
          <volume>1</volume>
          . IMPO, Montevideo, Uruguay,
          <volume>1</volume>
          <fpage>edn</fpage>
          . (
          <year>2007</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref23">
        <mixed-citation>
          23.
          <string-name>
            <surname>Russell</surname>
            ,
            <given-names>B.C.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Torralba</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Murphy</surname>
            ,
            <given-names>K.P.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Freeman</surname>
          </string-name>
          , W.T.:
          <article-title>LabelMe: a database and web-based tool for image annotation</article-title>
          .
          <source>Int. journal of Computer Vision</source>
          <volume>77</volume>
          (
          <issue>1-3</issue>
          ),
          <volume>157</volume>
          {
          <fpage>173</fpage>
          (
          <year>2008</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref24">
        <mixed-citation>
          24.
          <string-name>
            <surname>Singh</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nejdl</surname>
            ,
            <given-names>W.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Anand</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          :
          <article-title>Expedition: a time-aware exploratory search system designed for scholars</article-title>
          .
          <source>In: Proc. of the 39th Int. ACM SIGIR Conf</source>
          . pp.
          <volume>1105</volume>
          {
          <issue>1108</issue>
          (
          <year>2016</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref25">
        <mixed-citation>
          25.
          <string-name>
            <surname>Singh</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Nejdl</surname>
            ,
            <given-names>W.</given-names>
          </string-name>
          ,
          <string-name>
            <surname>Anand</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          :
          <article-title>History by diversity: Helping historians search news archives</article-title>
          .
          <source>In: Proc. of the 2016 ACM on Conf. on Human Information Interaction and Retrieval</source>
          . pp.
          <volume>183</volume>
          {
          <issue>192</issue>
          (
          <year>2016</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref26">
        <mixed-citation>
          26. of Texas, U.:
          <article-title>Del silencio a la memoria: Revelaciones del archivo historico de la polic a nacional</article-title>
        </mixed-citation>
      </ref>
      <ref id="ref27">
        <mixed-citation>
          27. van Hage,
          <string-name>
            <given-names>W.R.</given-names>
            ,
            <surname>Malaise</surname>
          </string-name>
          ,
          <string-name>
            <given-names>V.</given-names>
            ,
            <surname>Segers</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            ,
            <surname>Hollink</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            ,
            <surname>Schreiber</surname>
          </string-name>
          ,
          <string-name>
            <surname>G.</surname>
          </string-name>
          :
          <article-title>Design and use of the simple event model (sem)</article-title>
          .
          <source>Journal of Web Semantics</source>
          <volume>9</volume>
          (
          <issue>2</issue>
          ),
          <volume>128</volume>
          {
          <fpage>136</fpage>
          (
          <year>2011</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref28">
        <mixed-citation>
          28. W3C:
          <article-title>Resource Description Framework (RDF 1.1) (</article-title>
          <year>2014</year>
          ), https://www.w3.org/ 2001/sw/wiki/RDF
        </mixed-citation>
      </ref>
      <ref id="ref29">
        <mixed-citation>
          29.
          <string-name>
            <surname>Wada</surname>
          </string-name>
          , K.:
          <article-title>labelme: Image Polygonal Annotation with Python</article-title>
          . https://github. com/wkentaro/labelme (
          <year>2016</year>
          )
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>