<?xml version="1.0" encoding="UTF-8"?>
<TEI xml:space="preserve" xmlns="http://www.tei-c.org/ns/1.0" 
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" 
xsi:schemaLocation="http://www.tei-c.org/ns/1.0 https://raw.githubusercontent.com/kermitt2/grobid/master/grobid-home/schemas/xsd/Grobid.xsd"
 xmlns:xlink="http://www.w3.org/1999/xlink">
	<teiHeader xml:lang="en">
		<fileDesc>
			<titleStmt>
				<title level="a" type="main">AliQAn, Spanish QA System at CLEF-2008 *</title>
			</titleStmt>
			<publicationStmt>
				<publisher/>
				<availability status="unknown"><licence/></availability>
			</publicationStmt>
			<sourceDesc>
				<biblStruct>
					<analytic>
						<author>
							<persName><forename type="first">S</forename><surname>Roger</surname></persName>
							<email>sroger@dlsi.ua.es</email>
							<affiliation key="aff0">
								<orgName type="department">Department of Software and Computing Systems</orgName>
								<orgName type="laboratory">Natural Language Processing and Information Systems Group</orgName>
								<orgName type="institution">University of Alicante</orgName>
								<address>
									<country key="ES">Spain</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">K</forename><surname>Vila</surname></persName>
							<email>kvila@dlsi.ua.es</email>
							<affiliation key="aff0">
								<orgName type="department">Department of Software and Computing Systems</orgName>
								<orgName type="laboratory">Natural Language Processing and Information Systems Group</orgName>
								<orgName type="institution">University of Alicante</orgName>
								<address>
									<country key="ES">Spain</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">A</forename><surname>Ferrández</surname></persName>
							<affiliation key="aff0">
								<orgName type="department">Department of Software and Computing Systems</orgName>
								<orgName type="laboratory">Natural Language Processing and Information Systems Group</orgName>
								<orgName type="institution">University of Alicante</orgName>
								<address>
									<country key="ES">Spain</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">M</forename><surname>Pardiño</surname></persName>
							<affiliation key="aff0">
								<orgName type="department">Department of Software and Computing Systems</orgName>
								<orgName type="laboratory">Natural Language Processing and Information Systems Group</orgName>
								<orgName type="institution">University of Alicante</orgName>
								<address>
									<country key="ES">Spain</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">J</forename><forename type="middle">M</forename><surname>Gómez</surname></persName>
							<email>jmgomez@dlsi.ua.es</email>
							<affiliation key="aff0">
								<orgName type="department">Department of Software and Computing Systems</orgName>
								<orgName type="laboratory">Natural Language Processing and Information Systems Group</orgName>
								<orgName type="institution">University of Alicante</orgName>
								<address>
									<country key="ES">Spain</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">M</forename><surname>Puchol-Blasco</surname></persName>
							<affiliation key="aff0">
								<orgName type="department">Department of Software and Computing Systems</orgName>
								<orgName type="laboratory">Natural Language Processing and Information Systems Group</orgName>
								<orgName type="institution">University of Alicante</orgName>
								<address>
									<country key="ES">Spain</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">J</forename><surname>Peral</surname></persName>
							<email>jperal@dlsi.ua.es</email>
							<affiliation key="aff0">
								<orgName type="department">Department of Software and Computing Systems</orgName>
								<orgName type="laboratory">Natural Language Processing and Information Systems Group</orgName>
								<orgName type="institution">University of Alicante</orgName>
								<address>
									<country key="ES">Spain</country>
								</address>
							</affiliation>
						</author>
						<title level="a" type="main">AliQAn, Spanish QA System at CLEF-2008 *</title>
					</analytic>
					<monogr>
						<imprint>
							<date/>
						</imprint>
					</monogr>
					<idno type="MD5">05D9200D4600EE158842246E5FAF2755</idno>
				</biblStruct>
			</sourceDesc>
		</fileDesc>
		<encodingDesc>
			<appInfo>
				<application version="0.7.2" ident="GROBID" when="2023-03-24T08:44+0000">
					<desc>GROBID - A machine learning software for extracting information from scholarly documents</desc>
					<ref target="https://github.com/kermitt2/grobid"/>
				</application>
			</appInfo>
		</encodingDesc>
		<profileDesc>
			<textClass>
				<keywords>
					<term>H.3 [Information Storage and Retrieval]: H.3.1 Content Analysis and Indexing</term>
					<term>H.3.3 Information Search and Retrieval</term>
					<term>H.3.4 Systems and Software</term>
					<term>H.3.7 Digital Libraries Measurement, Performance, Experimentation Question answering, Questions beyond factoids</term>
				</keywords>
			</textClass>
			<abstract>
<div xmlns="http://www.tei-c.org/ns/1.0"><p>This paper describes the participation of the system AliQAn, a monolingual opendomain Question Answering (QA) System developed in the Department of Language Processing and Information System at the University of Alicante, in the CLEF-2008 Spanish monolingual QA evaluation task. Here, we focus on explaining a couple of strong points of the current version of AliQAn: (i) our algorithm for dealing with topic-related questions, and (ii) our approach for decreasing the number of inexact answers. We have also explored the use of the Wikipedia corpora, which have proposed some new challenges for the QA task. Besides, the achieved results (overall accuracy of 19.50%) are shown and discussed in this paper.</p></div>
			</abstract>
		</profileDesc>
	</teiHeader>
	<text xml:lang="en">
		<body>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="1">Introduction</head><p>Our system has been developed during three participations <ref type="bibr">(2005, 2006 and 2008)</ref> in the Department of Language Processing and Information Systems at the University of Alicante. It is based on complex pattern matching using NLP tools. This year we have adapted our system to work on Wikipedia and we have proposed a method to work with inexact answers. None method for anaphora resolution has been done for the topic-related questions. A simple method has been proposed to treat this type of questions.</p><p>The rest of this paper is organized as follows: section two and three describes our algorithm for dealing with topic-related questions and the special treatment of inexact answers respectively. Afterwards, the handling of Wikipedia and related problems. Section five describes the obtained results. Finally, our conclusions and future work are presented.</p><p>2 Dealing with Topic-Related Questions Before QA@CLEF 2007, all questions could be answered in isolation without any reference to the context (previous questions or answers). From QA@CLEF 2007 run, topic-related questions are clusters of questions which are related to the same topic and possibly contain anaphoric references between one question and the other questions of the same cluster.</p><p>For example, the questions of QA@CLEF 2008: 008-"¿Dónde vivía la tribu de los Mojave?" (Where did the Mohave tribe live?) and 009: "¿Quiénes eran sus enemigos?" (Who were their enemies?) belong to the same cluster. On the other hand, the following example shows an anaphoric reference between second question and third question of the one cluster: 029: "¿Entre qué días fue la batalla de Brunete?" (Among which days was the battle of Brunete?), 030: "¿Dónde se publicó el reportaje de Gerda Taro sobre esta batalla?" (Where was the article of Gerda Taro about this battle published?) and 031: ¿A qué hospital fue trasladada tras su accidente? (Which hospital were she moved to after her accident?). In 2007, 30 questions out of 200 were topic-related and in this year this size was enlarged to 64 out of 200 for Spanish.</p><p>To treat such context-dependent questions, the underlying idea was very simple. It considers the enrichment of dependent questions by adding some noun phrases of the first question of each cluster and the noun of the answer for this question. By reasons of simplicity and for avoid introducing noise; we only considered the co-reference between the first question and other of the same cluster. The algorithm employed contained the following steps:</p><p>1. Answering the first question of one cluster without special treatment.</p><p>2. Extracting the set of noun phrases from question and answer.</p><p>3. Adding this set of noun phrases to all dependent questions. 4. Handling and extracting the answers from these expanded questions.</p><p>For instance, if we consider the question 008, the system returns the answer "Arizona"(step 1).</p><p>Step 2 produces the noun phrases "la tribu de los Mojave" (Mohave tribe) and "Arizona". Then (step 3), it obtains the noun phrases that correspond to the 009 question: "sus enemigos" (their enemies). Finally, in step 4, the previous extended set with all the noun phrases is used to find the answer of question 009.</p><p>For the final answers to topic-related questions we obtain the following criteria:</p><p>• If the answer to the extended question (following the steps previously described) was nil, then nil was returned as final answer.</p><p>• In the opposite case, we have ranked the answers (with or without extension) in a decreasing order, thus returning the first three ones of the ranking.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="3">Inexact Treatment</head><p>This year, the algorithm to treat the inexact answers has been modified only for the questions which expected answer type as group, person, first name, place, country and city. In theses cases, we considered that the answers are able to contain one or more noun phrases. Before explaining the general algorithm used for handling theses answers, we will define some variables:</p><p>Set N P = np 1 , np 2 , . . . , np n where np i ∀i = 1..n are consecutive proper names of the head of the different noun phrases (included embebed noun phrases). Set Ω = {N P |N P ∈ head of the noun phrases of the answer}. For example, let considere the sentence "la Sociedad Española de Vexilología" (the Spanish Society of Vexillology), in this case the cardinality of Ω (|Ω|) is 2 and its elements are: "Spanish Society" and "Vexillology". The algorithm begins by finding the set Ω for all noun phrases of the answer. If |Ω| &gt; 1 then the elements of Ω are ranking according to its weight. The weight is increased or decreased in accordance with many different criteria and whether it belongs to specific dictionary or it does not. Criteria and dictionary are defined according to expected answer type. After this, the algorithm selects the element with bigger score and it returns the head of the noun phrase corresponding to this element. On the other hand, if the |Ω| = 1, then it only returns the corresponding head of the noun phrase. In the previous example, we suppose that the weight of "Spanish Society" is N 1 and the weight of "Vexillology" is N 2 . If N 1 &gt; N 2 , then algorithm returns "the Spanish Society" else it returns "Vexillology".</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="4">Exploring Wikipedia</head><p>Compared to traditional CLEF corpora (based on articles from newspapers), Wikipedia is a very large document collection and has not enough redundancy. In spite of that fact, the articles from newspapers have a fair amount of redundancy because they are usually published, with pretty much relevance, in different days, by different people and using different expressions. Wikipedia collections use hyperlinks to avoid information repetition (i.e. data which is sensitive to be repeated is replaced by links to the original source).</p><p>An Information Retrieval (IR) system needs to be more precise in order to filter the fair amount of irrelevant information due to the size of the Wikipedia collections. At the same time, an IR system needs to have high coverage to deal with the low redundancy of these corpora. In addition, Wikipedia, unlike newspaper collections, is highly structured. This structure gives a lot of information about the article topic in the form of tables, references and links. Hence, an IR system needs to consider this structure to take advantage of this information.</p><p>Bearing these considerations in mind, we aim to adapt two IR systems (namely, IR-n [5] and JIRS <ref type="bibr" target="#b3">[4]</ref>) in order to (i) be able to use very large document collections, and (ii) face up to the above-commented new Wikipedia challenges. Specifically, in this paper, our effort has been directed towards solving the first goal.</p><p>In addition, we would like to point out that several problems derived from the codification of the no-latin characters in Wikipedia were solved from the viewpoint of our QA system. The source of these problems is that the Wikipedia collections was coded in UTF-8, while our QA system uses ISO encoding to perform the morpho-syntactic labelling of documents via MACO <ref type="bibr" target="#b0">[1]</ref> and SUPAR [2] NLP tools.</p><p>An example that illustrates this problem is shown in the following question from QA@CLEF 2008: 048-"¿Qué cargo ocupaba Hideki Tōjō antes del ataque a Pearl Harbor?" (What position did Hideki Tōjō hold before the Pearl Harbor attack?) the "ō" character was codified as "?" by our system.</p><p>Our proposed solution for our QA system consists of controlling the correspondences between the two encodings for non-latin characters. Even though it is a very simple solution, good results are obtained. Nevertheless, as future work we wish to adapt our system and its related tools to directly work the UTF-8 encoding.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="5">Results</head><p>This section describes the table related with the results and the evaluation of our system in CLEF-2008. The proposed system was applied to the set of 200 questions.</p><p>The University submitted four runs to QA@CLEF, two runs for Spanish-Spanish (spsp) and two runs for English-Spanish (ensp). We participate with the monolingual runs. Regrettably, there was an error in the submitting of the runs, one run of spsp was overwritten with one run of ensp. Therefore, we only look the good run.</p><p>Table <ref type="table" target="#tab_0">1</ref> shows the results for this run and the effect that the errors and problems produced our system performance. It is important to remark that it is our first participation with the new characteristic: Wikipedia and topic-related questions.</p><p>On the other hand, AliQAn system had a high percentage of inexact answers in previous years. This kind of answers has been improved in this participation: of 24 in the year 2005 <ref type="bibr" target="#b5">[6]</ref> and 15 in the year 2006 <ref type="bibr" target="#b2">[3]</ref> to 4 this year <ref type="bibr">(2008)</ref>, which all correspond a list questions<ref type="foot" target="#foot_0">1</ref> .</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="6">Conclusion and Future Work</head><p>This paper summarizes our participation in the CLEF-2008 monolingual task with our monolingual open-domain QA System (AliQAn). This year, the main contributions were:</p><p>• Algorithm for resolving topic-related questions. The essence of this algorithm is to extend every question (q i ) by adding some noun phrases and the noun of the answer of the first question of the same cluster which q i depends on.</p><p>• Approach for decreasing the number of inexact answers. This approach assigns certain weight (determined by using specific dictionaries) to the heads of each answer's noun phrase according to an expected answer type and it returns the head of the noun phrase with the greatest weight. We have obtained excellent results with a decrease of 20 inexact answers with regard to the year 2005.</p><p>• Using Wikipedia with our IR &amp; QA systems. On one hand, our IR system has been adapted for making possible the use Wikipedia with very large document collections. On the other hand, several problems derived from the codification of the non-latin characters in Wikipedia have been resolved in order to properly use it together with our QA system.</p><p>All questions given in this track, except the list questions, have been treated by our system and only one has been unsupported. Our paper only includes one run for the Spanish monolingual QA task and it has achieved an overall accuracy of 19.50%. Finally, we would like to point out that this is the first time we deal with Wikipedia and topic-related questions for our participation in the CLEF QA task.</p><p>Our future work is focused on the multilingual task, the adaptation of the NLP tools related to our system to directly work the UTF-8 encoding and the incorporation of knowledge to the phases that can be useful to increase the performance of our system.</p></div><figure xmlns="http://www.tei-c.org/ns/1.0" type="table" xml:id="tab_0"><head>Table 1 :</head><label>1</label><figDesc>General results obtained in the QA@CLEF-2008</figDesc><table><row><cell>Right</cell><cell>39</cell></row><row><cell>Inexact</cell><cell>4</cell></row><row><cell>Unsupported</cell><cell>1</cell></row><row><cell>Accuracy (overall)</cell><cell>19.50%</cell></row><row><cell>Factoid questions</cell><cell>20.49 %</cell></row><row><cell>Definition questions</cell><cell>31.57 %</cell></row><row><cell>List questions</cell><cell>0%</cell></row><row><cell cols="2">Temporally questions 14.28 %</cell></row></table></figure>
			<note xmlns="http://www.tei-c.org/ns/1.0" place="foot" n="1" xml:id="foot_0">It is important to say that list questions are not supported by our system.</note>
		</body>
		<back>

			<div type="funding">
<div xmlns="http://www.tei-c.org/ns/1.0"><p>* This work has been partially supported by the framework of the project QALL-ME (FP6-IST-033860), which is a 6th Framenwork Research Programme of the European Union (EU), the Spanish Government, project TEXT-MESS (TIN-2006-15265-C06-01), by the University of Comahue under the project 04/E062, by the Generalitat Valenciana throught the research grants BFPI/2008/093 and BFPI06/182 and University of Matanzas.</p></div>
			</div>

			<div type="references">

				<listBibl>

<biblStruct xml:id="b0">
	<analytic>
		<title level="a" type="main">MACO: Morphological Analyzer Corpus-Oriented</title>
		<author>
			<persName><forename type="first">S</forename><surname>Acebo</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Ageno</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><surname>Climent</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Farreres</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><surname>Padró</surname></persName>
		</author>
		<author>
			<persName><forename type="first">R</forename><surname>Placer</surname></persName>
		</author>
		<author>
			<persName><forename type="first">H</forename><surname>Rodriguez</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Taulé</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Turno</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">ESPRIT BRA-7315 Aquilex II</title>
				<imprint>
			<date type="published" when="1994">1994</date>
			<biblScope unit="volume">31</biblScope>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b1">
	<analytic>
		<title level="a" type="main">An Empirical Approach to Spanish Anaphora Resolution. Machine Translation</title>
		<author>
			<persName><forename type="first">A</forename><surname>Ferrández</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Palomar</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><surname>Moreno</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Special Issue on Anaphora Resolution In Machine Translation</title>
		<imprint>
			<biblScope unit="volume">14</biblScope>
			<biblScope unit="issue">3/4</biblScope>
			<biblScope unit="page" from="191" to="216" />
			<date type="published" when="1999-12">December 1999</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b2">
	<analytic>
		<title level="a" type="main">Monolingual and cross-lingual qa using aliqan and brili systems for clef</title>
		<author>
			<persName><forename type="first">S</forename><surname>Ferrández</surname></persName>
		</author>
		<author>
			<persName><forename type="first">P</forename><surname>López-Moreno</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><surname>Roger</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Ferrández</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Peral</surname></persName>
		</author>
		<author>
			<persName><forename type="first">X</forename><surname>Alvarado</surname></persName>
		</author>
		<author>
			<persName><forename type="first">E</forename><surname>Noguera</surname></persName>
		</author>
		<author>
			<persName><forename type="first">F</forename><surname>Llopis</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">CLEF</title>
				<imprint>
			<date type="published" when="2006">2006. 2006</date>
			<biblScope unit="page" from="450" to="453" />
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b3">
	<analytic>
		<title level="a" type="main">Language independent passage retrieval for question answering</title>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">M</forename><surname>Gómez</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Montes-Gómez</surname></persName>
		</author>
		<author>
			<persName><forename type="first">E</forename><surname>Sanchis</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><surname>Villaseor-Pineda</surname></persName>
		</author>
		<author>
			<persName><forename type="first">P</forename><surname>Rosso</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">Fourth Mexican International Conference on Artificial Intelligence MICAI 2005</title>
		<title level="s">Lecture Notes in Computer Science</title>
		<meeting><address><addrLine>Monterrey, Mexico</addrLine></address></meeting>
		<imprint>
			<publisher>Springer-Verlag GmbH</publisher>
			<date type="published" when="2005">2005</date>
			<biblScope unit="page" from="816" to="823" />
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b4">
	<analytic>
		<title level="a" type="main">Passage selection to improve question answering</title>
		<author>
			<persName><forename type="first">F</forename><surname>Llopis</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">L</forename><surname>Vicedo</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Ferrández</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">Proceedings of the COLING 2002 Workshop on Multilingual Summarization and Question Answering</title>
				<meeting>the COLING 2002 Workshop on Multilingual Summarization and Question Answering<address><addrLine>Taipei, Taiwan</addrLine></address></meeting>
		<imprint>
			<date type="published" when="2002">2002</date>
			<biblScope unit="page" from="1" to="6" />
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b5">
	<analytic>
		<title level="a" type="main">Aliqan, spanish qa system at clef-2005</title>
		<author>
			<persName><forename type="first">S</forename><surname>Roger</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><surname>Ferrández</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Ferrández</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Peral</surname></persName>
		</author>
		<author>
			<persName><forename type="first">F</forename><surname>Llopis</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Aguilar</surname></persName>
		</author>
		<author>
			<persName><forename type="first">D</forename><surname>Tomás</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">CLEF</title>
				<imprint>
			<date type="published" when="2005">2005</date>
			<biblScope unit="page" from="457" to="466" />
		</imprint>
	</monogr>
</biblStruct>

				</listBibl>
			</div>
		</back>
	</text>
</TEI>
