<?xml version="1.0" encoding="UTF-8"?>
<TEI xml:space="preserve" xmlns="http://www.tei-c.org/ns/1.0" 
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" 
xsi:schemaLocation="http://www.tei-c.org/ns/1.0 https://raw.githubusercontent.com/kermitt2/grobid/master/grobid-home/schemas/xsd/Grobid.xsd"
 xmlns:xlink="http://www.w3.org/1999/xlink">
	<teiHeader xml:lang="en">
		<fileDesc>
			<titleStmt>
				<title level="a" type="main">TUKE at MediaEval 2013 Spoken Web Search Task</title>
			</titleStmt>
			<publicationStmt>
				<publisher/>
				<availability status="unknown"><licence/></availability>
			</publicationStmt>
			<sourceDesc>
				<biblStruct>
					<analytic>
						<author>
							<persName><forename type="first">Jozef</forename><surname>Vavrek</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Matúš</forename><surname>Pleva</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Martin</forename><surname>Lojka</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Peter</forename><surname>Viszlay</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Eva</forename><surname>Kiktová</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Daniel</forename><surname>Hládek</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Jozef</forename><surname>Juhár</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">{jozef</forename><surname>Vavrek</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Matus</forename><surname>Pleva</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Eva</forename><surname>Kiktova</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Daniel</forename><surname>Hladek</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Jozef</forename><surname>Juhar}</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Tuke</forename><surname>Sk</surname></persName>
							<affiliation key="aff0">
								<orgName type="institution">Technical University of Kosice</orgName>
								<address>
									<addrLine>Letna 9</addrLine>
									<postCode>04200</postCode>
									<settlement>Košice</settlement>
									<country key="SK">Slovakia</country>
								</address>
							</affiliation>
						</author>
						<title level="a" type="main">TUKE at MediaEval 2013 Spoken Web Search Task</title>
					</analytic>
					<monogr>
						<imprint>
							<date/>
						</imprint>
					</monogr>
					<idno type="MD5">F1AD1E72A141DD330F4167406A922ABC</idno>
				</biblStruct>
			</sourceDesc>
		</fileDesc>
		<encodingDesc>
			<appInfo>
				<application version="0.7.2" ident="GROBID" when="2023-03-19T17:58+0000">
					<desc>GROBID - A machine learning software for extracting information from scholarly documents</desc>
					<ref target="https://github.com/kermitt2/grobid"/>
				</application>
			</appInfo>
		</encodingDesc>
		<profileDesc>
			<abstract>
<div xmlns="http://www.tei-c.org/ns/1.0"><p>This paper provides a rough description of zero resource Query-by-Example retrieving system for the MediaEval 2013 spoken web search task. The proposed solution firstly implements the voice activity detection (VAD) utilizing variance of acceleration MFCC (VAMFCC) rule-based approach. A PCA-based segmentation, K-means clustering and GMM training are then used in order to built the posteriorgrams. Finally, two searching architectures based on posteriorgram matching (SDTW) and GMM modeling (GMM-FST) are evaluated. Results show that none of our systems is able to achieve the positive Actual Term Weighted Value, because of high number of insertions. We suppose that chosen clustering scheme caused generation of too many false alarms. Only provided data were used and no other resources were examined in any system component during the development.</p></div>
			</abstract>
		</profileDesc>
	</teiHeader>
	<text xml:lang="en">
		<body>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="1.">MOTIVATION</head><p>The main purpose of our experiments was to check the proposed approaches for the language independent audio query detection and new speech feature analysis components. The mentioned approach is used in the MediaEval activity <ref type="bibr" target="#b1">[1]</ref> and could be also applied in various speech <ref type="bibr" target="#b4">[4]</ref> or non-speech <ref type="bibr">[5]</ref> Query-by-Example applications.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.">SYSTEM OVERVIEW</head><p>Proposed solution for SWS task uses posteriorgram term matching and audio segment GMM modeling. The overall architecture of proposed system is depicted on Fig. <ref type="figure" target="#fig_0">1</ref>.</p><p>At first, a training phase is carried out using available development utterances. The VAMFCC-based silence detector performs the initial discrimination of silent parts in audio stream. The block of feature extraction is implemented after VAD utilizing 13 MFCCs. The phase of segmentation and clustering creates the audio segment units (ASU). ASU is represented as a small audio part (phoneme for example) with some spectral and temporal characteristics, different for each ASU. Then the training of acoustic models is performed using these ASU, where each ASU represents one class. Labels for these classes (ASUs) are assigned according to the number of GMM. Only the process of voice activity detection and feature extraction is then performed in preprocessing stage within the retrieving phase. Each utterance and</p><p>Copyright is held by the author/owner(s).  Segmental Dynamic Time Warping (SDTW) algorithm is then used for comparing these posteriorgrams and finding a possible occurrence of a query in the test utterance. The other solution GMM-FST (GMM-Finite State Transducers) implements a Viterbi algorithm to create a model for each query and state sequence network to find occurrences in test utterance by using this model.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>MediaEval 2013 Workshop,</head></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.1">Segmentation and Clustering</head><p>In order to identify and to distinguish the speech segments in the i-th utterance, PCA (Principal Component Analysis) was applied as follows. Each 13-dimensional MFCC vector xj was reshaped to matrix Xj with row dimension nr, where j ∈ 1; ni is the number of vectors in i-th recording. In the next step, the covariance matrix C1 was computed from the first matrix X1 and its eigenvectors and eigenvalues were computed. The eigenvalue spectrum Λ1 = {λ1j} nr j=1 was used to determine the significance ∆(λ1 max ) of the dominant eigenvalue of C1 as ∆(λ1 max ) =</p><formula xml:id="formula_0">λ 1max nr j=1 λ 1j</formula><p>, where λ1j are the eigenvalues of C1. Then the matrix X1 was spliced together with X2 and the covariance matrix C12 and ∆(λ12 max ) were computed again. If ∆(λ12 max ) compared to ∆(λ1 max ) changed significantly, a new speech segment was created and PCA started from the current frame. In the other way, if ∆(λ12 max ) did not change significantly, the current matrix X12 was spliced together with X3 and the process was repeated automatically until a new segment was indicated. The created segments corresponded to ASUs. In the next phase, the segments with similar acoustic and statistical properties were grouped together into several speech clusters using k-means clustering with k = 50 clusters and squared Euclidean distance metrics. As the input data for clustering the means of the segments were used. Each mean vector obtained an index (label) of the specific cluster. This label was assigned to the original feature vectors corresponding to the specific mean vector.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.2">Searching techniques</head></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.2.1">GMM approach</head><p>A retrieving process uses Weighted Finite State Transducers (WFST) that allow us to find the most probable path (state sequence) in search network <ref type="bibr" target="#b2">[2]</ref>. The search of a query consists of two steps. At first, query alone is recognized using search network, created from the trained acoustic model so that all GMM states are arranged in parallel. The result is a sequence of states that model the particular query. The process of recognition is done repeatedly with different insertion penalties in order to obtain multiple states sequences with different lengths. It helps to improve the model representation of retrieving query. The sequences are labeled and added to the previous search network in parallel. The second step involves the recognition of a test utterance using Viterbi algorithm. The final score for decision is computed as a difference between modeled likelihoods of query and utterance using lambda acoustic model where score = (P (Ooccurence|λ) − P (Oquery|λ)) + Θ. Score is then shifted by predefined value Θ and then results with score below zero are removed.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.2.2">SDTW detection</head><p>A simplified SDTW searching algorithm was utilized in our system, similar to that used in <ref type="bibr" target="#b3">[3]</ref>. The adjustment window condition was set to |(i k − i1) − (j k − j1)| ≤ R, where i1 and j1 are starting coordinates of warping path in each segment, i k and j k define the k-th coordinates and R represents the constraint parameter, set to M/2, where M is the length of query. The range of starting coordinates was conditioned by the constraint parameter and length of each utterance:</p><formula xml:id="formula_1">((2R + 1)k + 1, 1), where 0 ≤ k ≤ N −1</formula><p>2R+1 . The process of finding the optimal local alignment between each utterance and query produces a set of local warp paths, equal to the number of diagonal regions. A score parameter was then set in the following form score</p><formula xml:id="formula_2">= 2n N +M n 1 warpDist n+1</formula><p>,where n is the number of steps in local alignment, N is the length of utterance and M the length of query, n 1 warpDist represents a summation of components in each warping path, where components are computed from Bhattacharyya distance matrix. </p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="3.">EXPERIMENTAL RESULTS AND CON-CLUSIONS</head><p>The official results for SWS task are listed in Tab. 1. Two metrics were used to asses the overall performance of GMM-FST and SDTW on dev and eval queries: the actual AT W V , normalized Cnxe and minimal cross-entropy C min nxe . The score normalization for both systems was performed only on development data. A minimum-cost alignment (MCA) for each segment was used as detection score at first level of search in case of SDTW. Final detection of retrieved query was then carried out utilizing score parameter defined in (2.2.2). A threshold for this parameter was set to 0.0819. A decision threshold for score parameter was set to 2.8 in case of GMM-FST based system, while Θ = 3 (2.2.1). Both systems produced a huge amount of false alarms (FA) during the evaluation. Regarding the evaluation results, the GMM-FST system is more appropriate solution for SWS task, because of its lower tendency to detect spurious terms.</p><p>All the experiments were mainly done using 2x IBM System x3650 servers, 2x Intel Xeon QuadCore E5530 CPU @ 2.4 GHz Hyper-threading enabled (16 threads), 28 GB RAM, 1TB SAS HDD (RAID5), running Debian OS.</p><p>The Speed Factors (the ratio of the total time employed in searching{indexing} the set of queries in{and} the set of audio documents to the product{sum} of their total durations) and Peak Memory Usage during Searching{Indexing} tasks are presented in Tab. 2. The Performance Load equals 0.9 • SSF • P M US + 0.1 • ISF • P M UI is derived from them.</p><p>In the future, an improved clustering and segmentation algorithm will be investigated in order to decrease the overlapping between individual ASUs. A minimal length of warping path algorithm will be integrated in SDTW approach, too.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="4.">ACKNOWLEDGMENTS</head><p>This research was supported by the ERDF funded projects ITMS-26220220155 (50%) &amp; ITMS-26220220182 (50%).</p></div><figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_0"><head>Figure 1 :</head><label>1</label><figDesc>Figure 1: SWS framework architecture</figDesc></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" type="table" xml:id="tab_0"><head></head><label></label><figDesc>October 18-19, 2013, Barcelona, Spain</figDesc><table><row><cell cols="2">TRAINING PHASE</cell><cell></cell><cell></cell><cell></cell><cell></cell><cell></cell></row><row><cell>Dev. utterances</cell><cell>VAD VAMFCC based</cell><cell>MFCC (13)</cell><cell>Segmentation PCA based</cell><cell>K-means clustering</cell><cell>GMM training</cell><cell>Acoustic models</cell></row><row><cell cols="2">RETRIEVING PHASE</cell><cell></cell><cell></cell><cell></cell><cell></cell><cell></cell></row><row><cell>Audio query Test utterance</cell><cell>VAD VAMFCC based</cell><cell>MFCC (39)</cell><cell>Scoring using Mah. distance</cell><cell>Posteriorgram Posteriorgram for utterances for query</cell><cell></cell><cell>SDTW</cell></row><row><cell></cell><cell></cell><cell></cell><cell></cell><cell>Transducer GMM based</cell><cell></cell><cell>Final match</cell></row></table></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" type="table" xml:id="tab_1"><head>Table 1 :</head><label>1</label><figDesc>Evaluation results of the tested algorithms</figDesc><table><row><cell>query</cell><cell>system</cell><cell>ATWV</cell><cell>Cnxe</cell><cell>C min nxe</cell></row><row><cell>dev</cell><cell cols="4">GMM-FST -0.1371 0.980745 0.976363</cell></row><row><cell>eval</cell><cell cols="2">GMM-FST -0.1372</cell><cell cols="2">0.98505 0.979883</cell></row><row><cell>dev</cell><cell>SDTW</cell><cell cols="3">-0.4176 0.998421 0.989649</cell></row><row><cell>eval</cell><cell>SDTW</cell><cell cols="3">-0.4252 0.998494 0.988404</cell></row></table></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" type="table" xml:id="tab_2"><head>Table 2 :</head><label>2</label><figDesc>Processing resources measures</figDesc><table><row><cell>system</cell><cell>ISF</cell><cell>SSF</cell><cell cols="2">P M UI P M US</cell><cell>PL</cell></row><row><cell cols="3">GMM-FST 0.0054 0.0048</cell><cell>2GB</cell><cell cols="2">1.8GB 0.009</cell></row><row><cell>SDTW</cell><cell cols="2">0.0054 0.0046</cell><cell>2GB</cell><cell>2.2GB</cell><cell>0.01</cell></row></table></figure>
		</body>
		<back>
			<div type="references">

				<listBibl>

<biblStruct xml:id="b0">
	<monogr>
		<title/>
		<author>
			<persName><surname>References</surname></persName>
		</author>
		<imprint/>
	</monogr>
</biblStruct>

<biblStruct xml:id="b1">
	<analytic>
		<title level="a" type="main">The Spoken Web Search Task</title>
		<author>
			<persName><forename type="first">X</forename><surname>Anguera</surname></persName>
		</author>
		<author>
			<persName><forename type="first">F</forename><surname>Metze</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Buzo</surname></persName>
		</author>
		<author>
			<persName><forename type="first">I</forename><surname>Szoke</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><forename type="middle">J</forename><surname>Rodriguez-Fuentes</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">MediaEval 2013 Workshop</title>
				<meeting><address><addrLine>Barcelona, Spain</addrLine></address></meeting>
		<imprint>
			<date type="published" when="2013-10-19">18-19 October 2013</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b2">
	<analytic>
		<title level="a" type="main">Finite-state transducers and speech recognition in Slovak language</title>
		<author>
			<persName><forename type="first">M</forename><surname>Lojka</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Juhár</surname></persName>
		</author>
		<idno>art. no. 5941305</idno>
	</analytic>
	<monogr>
		<title level="m">SPA 2009 Conference</title>
				<imprint>
			<date type="published" when="2009">2009</date>
			<biblScope unit="page" from="149" to="153" />
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b3">
	<analytic>
		<title level="a" type="main">Unsupervised pattern discovery in speech</title>
		<author>
			<persName><forename type="first">A</forename><surname>Park</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Glass</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">IEEE T Audio Speech</title>
		<imprint>
			<biblScope unit="volume">16</biblScope>
			<biblScope unit="issue">1</biblScope>
			<biblScope unit="page" from="186" to="197" />
			<date type="published" when="2008">2008</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b4">
	<analytic>
		<title level="a" type="main">TUKE MediaEval 2012: Spoken Web search using DTW and unsupervised SVM</title>
		<author>
			<persName><forename type="first">J</forename><surname>Vavrek</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Pleva</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Juhár</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">MediaEval 2012 Workshop</title>
		<title level="s">Pisa -CEUR Workshop Proceedings</title>
		<imprint>
			<date type="published" when="2012">2012</date>
			<biblScope unit="volume">927</biblScope>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b5">
	<analytic>
		<title level="a" type="main">Detection and classification of audio events in noisy environment</title>
		<author>
			<persName><forename type="first">E</forename><surname>Vozáriková</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Pleva</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><surname>Ondáš</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Vavrek</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Juhár</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Čižmár</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Journal of Computer Science and Control Systems</title>
		<imprint>
			<biblScope unit="volume">3</biblScope>
			<biblScope unit="issue">1</biblScope>
			<biblScope unit="page" from="253" to="258" />
			<date type="published" when="2010">2010</date>
		</imprint>
	</monogr>
</biblStruct>

				</listBibl>
			</div>
		</back>
	</text>
</TEI>
