<?xml version="1.0" encoding="UTF-8"?>
<TEI xml:space="preserve" xmlns="http://www.tei-c.org/ns/1.0" 
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" 
xsi:schemaLocation="http://www.tei-c.org/ns/1.0 https://raw.githubusercontent.com/kermitt2/grobid/master/grobid-home/schemas/xsd/Grobid.xsd"
 xmlns:xlink="http://www.w3.org/1999/xlink">
	<teiHeader xml:lang="en">
		<fileDesc>
			<titleStmt>
				<title level="a" type="main">An organizational environment for in silico experiments in molecular biology</title>
			</titleStmt>
			<publicationStmt>
				<publisher/>
				<availability status="unknown"><licence/></availability>
			</publicationStmt>
			<sourceDesc>
				<biblStruct>
					<analytic>
						<author>
							<persName><forename type="first">Yuan</forename><surname>Lin</surname></persName>
							<affiliation key="aff0">
								<orgName type="laboratory" key="lab1">LIRMM</orgName>
								<orgName type="laboratory" key="lab2">UMR5506 CNRS-UM2</orgName>
								<address>
									<addrLine>161, rue Ada</addrLine>
									<postCode>34095, Cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Marie-Angélique</forename><surname>Laporte</surname></persName>
							<affiliation key="aff0">
								<orgName type="laboratory" key="lab1">LIRMM</orgName>
								<orgName type="laboratory" key="lab2">UMR5506 CNRS-UM2</orgName>
								<address>
									<addrLine>161, rue Ada</addrLine>
									<postCode>34095, Cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
							<affiliation key="aff2">
								<orgName type="laboratory" key="lab1">Centre d&apos;Ecologie Fonctionnelle et Evolutive</orgName>
								<orgName type="laboratory" key="lab2">UMR5175 CNRS</orgName>
								<address>
									<addrLine>1919, route de Mende</addrLine>
									<postCode>34293, Cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Lucile</forename><surname>Soler</surname></persName>
							<affiliation key="aff3">
								<orgName type="department" key="dep1">CIRAD-PERSYST</orgName>
								<orgName type="department" key="dep2">Campus International de Baillarguet</orgName>
								<address>
									<postCode>34398, cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Isabelle</forename><surname>Mougenot</surname></persName>
							<affiliation key="aff0">
								<orgName type="laboratory" key="lab1">LIRMM</orgName>
								<orgName type="laboratory" key="lab2">UMR5506 CNRS-UM2</orgName>
								<address>
									<addrLine>161, rue Ada</addrLine>
									<postCode>34095, Cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
							<affiliation key="aff1">
								<orgName type="laboratory">UMR ESPACE DEV IRD-UM2</orgName>
								<address>
									<addrLine>500 rue J.F. Breton</addrLine>
									<postCode>34093, Cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
						</author>
						<author>
							<persName><forename type="first">Thérèse</forename><surname>Libourel</surname></persName>
							<affiliation key="aff0">
								<orgName type="laboratory" key="lab1">LIRMM</orgName>
								<orgName type="laboratory" key="lab2">UMR5506 CNRS-UM2</orgName>
								<address>
									<addrLine>161, rue Ada</addrLine>
									<postCode>34095, Cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
							<affiliation key="aff1">
								<orgName type="laboratory">UMR ESPACE DEV IRD-UM2</orgName>
								<address>
									<addrLine>500 rue J.F. Breton</addrLine>
									<postCode>34093, Cedex 5</postCode>
									<settlement>Montpellier</settlement>
									<country key="FR">France</country>
								</address>
							</affiliation>
						</author>
						<title level="a" type="main">An organizational environment for in silico experiments in molecular biology</title>
					</analytic>
					<monogr>
						<imprint>
							<date/>
						</imprint>
					</monogr>
					<idno type="MD5">01A170CF7ADFE51ED7AD0D322AB6284C</idno>
				</biblStruct>
			</sourceDesc>
		</fileDesc>
		<encodingDesc>
			<appInfo>
				<application version="0.7.2" ident="GROBID" when="2023-03-24T14:20+0000">
					<desc>GROBID - A machine learning software for extracting information from scholarly documents</desc>
					<ref target="https://github.com/kermitt2/grobid"/>
				</application>
			</appInfo>
		</encodingDesc>
		<profileDesc>
			<textClass>
				<keywords>
					<term>Scientific workflow</term>
					<term>analysis pipeline</term>
					<term>specification language</term>
					<term>validation aspects of service composition</term>
				</keywords>
			</textClass>
			<abstract>
<div xmlns="http://www.tei-c.org/ns/1.0"><p>Molecular biologists, just like geneticists, make use of various experimental mechanisms and devices to conduct research and to validate or invalidate their theories or initial hypotheses. Mechanisms powered by information technology, called in silico, put data and analysis tools at the centre of the experiments, and are thus different from in vivo, ex vivo and in vitro mechanisms. Multiple resources (data sources as well as analysis tools) are widely available and, very often, allow various modes of operation, requiring certain expertise for their optimal use. This is especially true when drawing up complex analysis scenarios based on the sequential use of appropriate processing tools. To facilitate the construction of these experimentation mechanisms, we propose a scientific workflow infrastructure which uses an organizational environment to allow abstract planning of the experimentation, followed by its concretization. The concretization phase includes a verification of the conformity of the planned process chains composition to avoid any error during execution.</p></div>
			</abstract>
		</profileDesc>
	</teiHeader>
	<text xml:lang="en">
		<body>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="1">Introduction</head><p>Life sciences often rely on the chaining of data and application resources to express the experimentation process. Valuable resources for biology, while available in ever-increasing quantities, remain, for the most part, cost-expensive and time-consuming to acquire and thus their reuse becomes almost a necessity.</p><p>To design these complex experiments, scientists often need to locate suitable resources and then to organize or reorganize them. In addition, each experiment deserves to be saved so that it can be re-executed several times, either in various different configurations or with diverse test data. In such a context, the use of a scientific workflow proves to be an invaluable help. Several dedicated software applications for this purpose now exist, most notably in the financial sector, and research in the field is relatively advanced. A first study <ref type="bibr" target="#b6">[7]</ref> presented our approach based on the concept of the scientific workflow environment. Its objective is to help the user to:</p><p>design experimentation process chains (in as abstract a manner as possible), better organize resources (data and processes) which will be elements in the concretization of these process chains, capitalize on the existing by constructing new processes from previously devised experimentation plans.</p><p>This article develops our research advances in terms of resource organization and semi-automatic verification of validity of workflows designed within a prototype.</p><p>This article is structured as follows: section 2 presents a brief state of the art, section 3 proposes an architecture for implementing a scientific workflow and section 4 provides a glimpse of the organization brought about. Section 5 covers the proposed verification of conformity, section 6 illustrates with an example the validation of conformity of a concrete process chain, and section 7 presents perspectives in progress.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2">State of the art</head><p>A study was conducted based on characteristics we deemed relevant <ref type="bibr" target="#b7">[8]</ref>:</p><p>-The existence of a meta level for describing and creating process chains. In fact, the generic aspect conferred by meta-modelling appears to be fundamental for all of us. -Taking the experimental aspect into account. The unique characteristics of scientific data and processes should show through at the formalism level.</p><p>We present here only two representative projects, Kepler <ref type="bibr" target="#b0">[1]</ref> and Taverna <ref type="bibr" target="#b5">[6]</ref>, which gain a certain amount of popularity among workflow scientists.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.1">KEPLER</head><p>KEPLER 5 is a complete scientific workflow environment based on the Ptolemy II platform of the University of Berkeley. As far as process chains are concerned, KEPLER adopts a human organization metaphor. It is Actor-Based and considers all components of a process chain as actors. Actors (services) are accessed via a structure corresponding to the business ontology of the concerned domain.</p><p>The workflow is represented using a graphical language in the form of a graph linking ports (input/output parameters) of actors via channels. One or more actors in charge, Directors, plan tasks for other actors of the organization; they do so based on the available ontology. The execution plan of a process chain (or a portion of a process chain) is therefore created by a Director of the system. Any necessary adaptations are achieved by intermediary sender and receiver programs, which ensure the compatibility of data transferred over a channel. The process chain is saved in the form of MoML (Modelling Markup Language) files. (MoML is an XML-based language.) At the environment-interface level, a specific zoom feature is associated with the concept of an opaque actor (cf. figure <ref type="figure" target="#fig_0">1</ref>). An opaque actor appearing in a process chain can be opened, thus revealing its constituent details. </p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.2">Taverna</head><p>Taverna is a workflow project created by the my Grid team in England and used mainly in the life sciences. A workflow in Taverna is considered as a process graph in which processes are connected by data links or control links. Processes used are essentially web services (which can be supplemented by local libraries, manuscript scripts, etc.). During process composition, the user manually couples input/output parameters of web services or invokes shim services, specific adaptors existing from couplings constructed and tested for experiments. In addition, the process chain is saved in the form of a SCUFL (Simple Conceptual Unified Flow Language) file. (SCUFL is an XML-based language.) Fig. <ref type="figure" target="#fig_6">2</ref>. A concrete workflow in Taverna (taken from the myExperiment Taverna sharing site)</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="2.3">Other related works of interest</head><p>The Taverna and Kepler projects both provide generic models for instantiation and composition of services. Additionally, some other approaches are also highly relevant to scientific workflow management:</p><p>-The project BioMoby <ref type="bibr" target="#b16">[17]</ref>, as a first attempt to assist process chaining by using scientific resources, which are described and classified in the MOBY Central. -PISE ans its revised system Mobyle <ref type="bibr" target="#b17">[18]</ref> that provides a web environment (a Web Portal) to define and execute bioinformatics analyses. Registered analysis programs are pre-classified in a hierarchy, as well as some frequentlyused workflows. Experts can easily find them by using the search function panel that is integrated in the web site. -The project ProtocolDB <ref type="bibr" target="#b18">[19]</ref> proposed to model scientific workflows at two different layers (design protocol/ implementation protocol). An implementation protocol for a given design protocol is realized by mapping design tasks to different implementation tasks (scientific resources like database queries/ tools), and by connecting them together.</p><p>-In <ref type="bibr" target="#b19">[20]</ref><ref type="bibr" target="#b20">[21]</ref><ref type="bibr" target="#b21">[22]</ref>, scientific workflow modeling is supported by resource discovery approaches.</p><p>In this manuscript we focus mainly on scientific workflows and the way they are modeled and implemented. Our proposal introduces an additional level of abstraction, whose purpose is to describe the business domain prior to creating the process chains. This additional modelling level is predicted to facilitate the construction of process chains by allowing biologists to use their expertise of their domain, but without requiring them to have expert and often precise knowledge of the underlying resources and their locations. It also plays the role of a prescription model, to which instantiation and service composition models have to conform.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="3">Workflow architecture</head><p>Our efforts have been guided by the business point of view, that of the experimenters. Designing an experimental protocol corresponds to general model with three stages: 1) Definition: abstract definition of a process chain corresponding to an experimentation sequence (planning the experiments), 2) Instantiation: a more specific definition after identifying the various elements of the chain (data/processes), 3) Execution: customized execution (according to strategies corresponding to the requirements).</p><p>Based on this experimental life cycle, and inspired by the architectural styles proposed by OMG <ref type="bibr" target="#b10">[11]</ref>, we propose the following 3-level architectural vision (cf. figure <ref type="figure" target="#fig_2">3</ref> </p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>Conforms</head></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>Business description of the process chain</head></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>Instance of</head></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>Model instantiated from a business model</head></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>Choice of the execution strategy</head><p>Language used to define a workflow business model The static level concerns the design phase. It is a matter of constructing (abstract) business-process models using a simple language. The intermediate level represents an instantiation and pre-verification phase. Using the business process model, the user constructs the real process chain by selecting and locating the processes and data most appropriate to the planned experimentation. The pre-verification is semi-automatized (cf. section 4). The dynamic level concerns the actual execution phase. It takes place based on the various strategies defined by both the user and the operational configurations.</p><p>The static level has been studied in some detail in our <ref type="bibr" target="#b6">[7,</ref><ref type="bibr" target="#b7">8]</ref>. We have analyzed various language standards such as UML (activity diagram) <ref type="bibr" target="#b8">[9]</ref> and SPEM <ref type="bibr" target="#b9">[10]</ref>, as also various existing projects such as BioSide <ref type="bibr" target="#b4">[5]</ref>, Meta-model WDO-It! <ref type="bibr" target="#b11">[12]</ref> and CIMFlow <ref type="bibr" target="#b3">[4]</ref>. Following this study, we proposed a simple but complete language. It is based on a language defined by a meta-model whose abstract elements, tasks or processes, are connected by unidirectional links and by the intermediary of ports. To facilitate the manipulation of abstract process chains, a corresponding graphical language was created within a prototype (cf. the top part of the figure <ref type="figure">4</ref>). By using this workflow definition language, a simple example is modelled and shown in the lower part of the figure <ref type="figure">4</ref>  </p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>Abstract model</head><p>Fig. <ref type="figure">4</ref>. Some essential elements of our graphical language and a simple example</p><p>We currently focus on the intermediate level, which consists of two essential stages:</p><p>instantiation of the abstract model with existing resources (data/processes); validation of the concrete model instantiated from the organizational environment.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="4">Organizational environment</head><p>To carry out the experimental protocols, the abstract model instantiation stage consists of finding and reusing existing resources. To facilitate this search, we base ourselves on the concept of organizational environment. This environment relies on the description of resources (data and processes) in the form of metadata (expressed in XML schema format). The resource descriptions are hierarchized in resource categories and in concrete resources. As shown in figure <ref type="figure">5</ref>, it consists of:</p><p>an organization relating to processes. It manages the hierarchy of descriptions of process categories and of concrete processes. The concept of Converter corresponds to the concept of a specific process responsible for adapting data between different formats of the same data category.</p><p>an organization relating to data. It manages a hierarchy of descriptions of data categories, of concrete data and of the various associated data formats 7 . To illustrate this concept of the environment, we take an example from the world of molecular biology (cf. figure <ref type="figure" target="#fig_4">6</ref>). The upper part of each hierarchy (processes and data) represent a set of categories (shown as ovals) sorted according to the generalization/specialization relationship. The descriptions of concrete resources (data or processes) are then associated to their category.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head>Environment</head><p>The description of a concrete data describes its format, whereas that of a concrete process corresponds to its signature, which we formalize thus: Definition 1. Formalized signature of a concrete process Name (Input parameter list) : (Output parameter list), where each parameter is described by the doublet (Data category : data format).</p><p>A set of data formats (Fasta, xml, MultiFasta, Clustal, Newick, Jpeg) is also presented. Figure <ref type="figure" target="#fig_4">6</ref> is therefore complemented by the description of signatures of some example concrete processes: </p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="5.1">The problem</head><p>As already mentioned, the second important stage of the intermediate level consists of validating the concrete model instantiated from the abstract model. Let us take an example described by using the workflow language, corresponding to an abstract process chain model that a biologist designs with the intention of characterizing a protein sequence which interests him in the context of his putative functional domains.</p><p>At the concrete level, the idea is to begin by using the Blast similarity-search tool to compare the protein sequence under consideration with a data bank of protein sequences and to thus identify segments with high similarity shared both by the protein sequence under consideration and by various sequences in the sequence data bank. These similar segments indicate the possible presence of functional domains. The biologist then continues his study by reusing the results output from the Blast tool <ref type="bibr" target="#b1">[2]</ref>, either to construct a phylogenetic tree and retrace the evolutionary history of the sequence via the PhyML tool <ref type="bibr" target="#b2">[3]</ref> or to display the preserved positions common to all the similar segments via the Logo tool <ref type="bibr" target="#b12">[13]</ref>. This simplified example of a process chain in molecular biology allows us to highlight the difficulties encountered by the biologist in using the results output by one tool as input to another tool. The difficulties relate, at the same time, to the nature of the data (here characterized as data category), to the format of this data, and, finally, to the biologists expertise. In the example, we make willing use of the discrepancy which arises between the Blast tool, which outputs a collection of simple alignments, and the PhyML and Logo tools, which require multiple alignments to run. In fact, Blast leads to multiple discrepancies two-by-two, involving the sequence under consideration and one of the sequences from the sequence data bank which is similar to it; whereas PhyML and Logo use the shared similarity by a set of sequences which includes the sequence under consideration. This example highlights what we will subsequently term semantic incompatibility.</p><p>In its upper part, the figure <ref type="figure" target="#fig_5">7</ref> shows the abstract process chain and in the lower the concrete chain obtained after locating data descriptions S1 and adapted processes Blastp and PhyML. The problem which we designate as one of validation of the instantiated (concrete) model consists of verifying the compatibility of each composition. A composition corresponds to the link between an output parameter p1 of a process T and an input parameter p2 of the process following T; we denote it (p1 → p2).</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="5.2">Identifying situations of compatibility</head><p>Verification is undertaken by analyzing the signatures of linked processes. To do so, we have to take two important aspects into account:</p><p>-The syntactic aspect : relating to the data formats used by the parameters. -The semantic aspect : relating to the processs functionality. It not only depends on the processs name but also on the signification of the input/output parameters.</p><p>For two processes T1(dc1:fo1) : (dc2:fo2, dc3:fo3) and T2(dc4:fo4) : (dc5:fo5), let us suppose that there exists a composition, denoted p1→p2, between the p1 (dc3:fo3) output parameter of process T1 and the p2 (dc4:fo4) input parameter of process T2.</p><p>Syntactic and semantic compatibilities are defined as follows: The verification of a compositions compatibility is thus done at two levels: syntactic and semantic. Three types of situations can arise:</p><formula xml:id="formula_0">-Situation 1 (p1 Sem → p2) ∧ (p1</formula><p>Syn → p2): p1 and p2 are compatible at the semantic and syntactic levels. This is the ideal situation in our context; we designate it as valid.</p><p>-Situation 2 (p1 Sem → p2) ∧ (p1 Syn p2): p1 and p2 are compatible at the semantic level but not at the syntactic level. The composition is syntactically adaptable. An adaptation between the two data formats will be necessary (cf. converters).</p><p>-Situation 3 p1</p><p>Sem p2 : The two parameters are not semantically compatible.</p><p>In such a case, it is pointless to proceed to verify their syntactic compatibility (in fact, for us, two parameters with different significations cannot be paired).</p><p>The composition is semantically adaptable.</p><p>From these definitions, we develop our proposed approach for resolving the incompatibilities.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="6">Validation of the experimental chain</head><p>Of the three compatibility situations identified, the latter two require an adaptation stage before going on to the execution phase. It is a matter of finding one or more intermediate processes which can overcome the compositions incompatibility. For situations 2 and 3, two types of adaptations are proposed:</p><p>semantic adaptation (for situation 3). The incompatibility of situation 3 represents the case where the two parameters of a composition use incompatible data categories. The adaptation here consists of finding a possible intermediate process chain between these two categories. syntactic adaptation (for situation 2). In situation 2, where the composition is already semantically compatible, the problem can be expressed as a divergence between the data formats used by the two connected parameters.</p><p>All that is required is to find converters to convert one data format into the other.</p><p>These adaptations are based on the organizational environment. The search for intermediate processes can be equated to a search for itineraries between two incompatible data categories or formats. We will illustrate this using the example and the organizational environment constructed earlier (cf. figure <ref type="figure" target="#fig_4">6</ref>).</p><p>Let us consider again the previous example. The verification conducted on the instantiation of the abstract model detects a semantic incompatibility in the composition between Blastp and Logo or between Blastp and PhyML due to difference in categories Pairs of sequences and Multiple Alignment (Incompatibility situation 3 ). The (semantic) adaptation will be applied; it consists of finding in what we call the (semantic) resource graph the path allowing the conversion of categories.</p><p>The construction of the (semantic) resource graph consists of extracting, from the organizational environment, the descriptions of processes and of data categories referenced by their parameters. Such a (semantic) resource graph generated from the environment described in the figure <ref type="figure" target="#fig_4">6</ref> is shown in the figure <ref type="figure" target="#fig_7">8</ref>.</p><p>A graph traversal algorithm is used to find all the possible paths between the two concerned data categories (Pairs of sequences and Multiple Alignment). A single path is found in the graph: Pairs of sequences → InteractiveSelection → ProteinDataBank → ClustalW → Multiple Alignment. The two processes, InteractiveSelection and ClustalW, will therefore be added to the incompatible chain (cf. figure <ref type="figure" target="#fig_9">9</ref>).   Once this adaptation is done, there still remains the existing syntactic incompatibility of the composition between the InteractiveSelection and ClustalW processes because even though InteractiveSelection outputs the same data category that is accepted for input by ClustalW, their data formats are different (xml and MultiFasta). Syntactic adaptation consists of finding specific converters, or compositions of converters, necessary for these conversions. We will not cover this stage in detail; it is simply enough to understand that converters (or their composition) can be added to obtain the required validity.</p></div>
<div xmlns="http://www.tei-c.org/ns/1.0"><head n="7">Conclusion and perspectives</head><p>A prototype (http://www.lirmm.fr/ lin/project/) illustrating the key aspects of our approach for designing and validating scientific process chains is currently being developed. This prototype serves as a basis for an inductive experimental approach using data of BAC and EST nucleic sequences as well as physical and genetic maps for identifying and characterizing genetic markers relating to sex of the Nile tilapia (Oreochromis niloticus). Over a longer term, we intend to integrate the current prototype into a platform with a search engine based on resource descriptions to be able to undertake the execution using real re-sources, after requisite validation of experimentation chain. It will eventually also use open-source controlled vocabularies such as PFO (Protein Feature Ontology) <ref type="bibr" target="#b13">[14]</ref>, SO (Sequence Ontology) <ref type="bibr" target="#b14">[15]</ref>, and GO (Gene Ontology) <ref type="bibr" target="#b15">[16]</ref> to enrich data categories by additional representations and thus extend the descriptive capacities of the organizational environment.</p></div><figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_0"><head>Fig. 1 .</head><label>1</label><figDesc>Fig. 1. Overview of a process chain in the KEPLER environment</figDesc><graphic coords="3,149.55,315.85,316.25,141.90" type="bitmap" /></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_2"><head>Fig. 3 .</head><label>3</label><figDesc>Fig. 3. 3-level architecture of a workflow component</figDesc></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_3"><head>1 Fig. 5 .</head><label>15</label><figDesc>Fig. 5. Organizational environment</figDesc></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_4"><head>7Fig. 6 .</head><label>6</label><figDesc>Fig. 6. Illustration of an organizational environment in a biological context</figDesc></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_5"><head>Fig. 7 .</head><label>7</label><figDesc>Fig. 7. Problem at hand</figDesc><graphic coords="10,134.77,116.83,368.50,113.39" type="bitmap" /></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_6"><head>Definition 2 .</head><label>2</label><figDesc>Syntactic compatibility p1 → p2 is syntactically compatible if (fo3 = fo4) ∨ (fo3 is a sub-format of fo4), denoted p1 Syn → p2. Two parameters are syntactically compatible if they use the same data format or if they use an output format which is a sub-format of the input format. Else p1 Syn p2. Definition 3. Semantic compatibility p1 → p2 is semantically compatible if (dc3 = dc4) ∨ (dc3 is a sub-category of dc4), denoted p1 Sem → p2. Two parameters are semantically compatible if they use the same category, or if they use an output category which is a sub-category of the input category. Else p1 Sem p2.</figDesc></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_7"><head>Fig. 8 .</head><label>8</label><figDesc>Fig. 8. (Semantic) resource graph generated from the organizational environment of the figure 6</figDesc><graphic coords="12,134.77,116.83,368.51,56.69" type="bitmap" /></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" xml:id="fig_9"><head>Fig. 9 .</head><label>9</label><figDesc>Fig. 9. Semantic adaptation</figDesc></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0"><head></head><label></label><figDesc></figDesc><graphic coords="4,173.68,115.83,268.00,214.80" type="bitmap" /></figure>
<figure xmlns="http://www.tei-c.org/ns/1.0" type="table" xml:id="tab_0"><head></head><label></label><figDesc>6 .</figDesc><table><row><cell>Atomic task</cell><cell>Role</cell><cell>Data</cell><cell cols="2">Port (parameter)</cell><cell>Data link</cell></row><row><cell>Task</cell><cell>Role</cell><cell>page page Data</cell><cell></cell><cell></cell><cell>data</cell></row><row><cell></cell><cell></cell><cell></cell><cell>1</cell><cell>Visualization</cell><cell>page page Image</cell></row><row><cell>page page Protein sequence</cell><cell>Similarity search</cell><cell>page page Alignment</cell><cell>2</cell><cell>Tree reconstruction</cell><cell>page page Tree</cell></row></table></figure>
			<note xmlns="http://www.tei-c.org/ns/1.0" place="foot" n="5" xml:id="foot_0">http://kepler-project.org/</note>
			<note xmlns="http://www.tei-c.org/ns/1.0" place="foot" n="6" xml:id="foot_1">This example is also used in the later sections, we will explain it in detail during the following sections.</note>
		</body>
		<back>
			<div type="references">

				<listBibl>

<biblStruct xml:id="b0">
	<analytic>
		<title level="a" type="main">S04 -introduction to scientific workflow management and the kepler system</title>
		<author>
			<persName><forename type="first">I</forename><surname>Altintas</surname></persName>
		</author>
		<author>
			<persName><forename type="first">B</forename><surname>Ludäscher</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><surname>Klasky</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><forename type="middle">A</forename><surname>Vouk</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">SC</title>
		<imprint>
			<biblScope unit="page">205</biblScope>
			<date type="published" when="2006">2006</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b1">
	<analytic>
		<title level="a" type="main">Basic local alignment search tool</title>
		<author>
			<persName><forename type="first">S</forename><surname>Altschul</surname></persName>
		</author>
		<author>
			<persName><forename type="first">W</forename><surname>Gish</surname></persName>
		</author>
		<author>
			<persName><forename type="first">W</forename><surname>Miller</surname></persName>
		</author>
		<author>
			<persName><forename type="first">E</forename><surname>Myers</surname></persName>
		</author>
		<author>
			<persName><forename type="first">D</forename><surname>Lipman</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Journal of Molecular Biology</title>
		<imprint>
			<biblScope unit="volume">215</biblScope>
			<biblScope unit="page" from="403" to="410" />
			<date type="published" when="1990">1990</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b2">
	<analytic>
		<title level="a" type="main">A simple, fast, and accurate algorithm to estimate large phylogenies by maximum likelihood</title>
		<author>
			<persName><forename type="first">S</forename><surname>Guindon</surname></persName>
		</author>
		<author>
			<persName><forename type="first">O</forename><surname>Gascuel</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Systematic Biology</title>
		<imprint>
			<biblScope unit="volume">52</biblScope>
			<biblScope unit="page" from="696" to="704" />
			<date type="published" when="2003">2003</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b3">
	<analytic>
		<title level="a" type="main">CIMFlow: A Workflow Management System Based on Integration Platform Environment</title>
		<author>
			<persName><forename type="first">L</forename><surname>Haibin</surname></persName>
		</author>
		<author>
			<persName><forename type="first">F</forename><surname>Yushun</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">Proceedings of 7th IEEE International Conference on Emerging Technologies and Factory Automation</title>
				<meeting>7th IEEE International Conference on Emerging Technologies and Factory Automation<address><addrLine>Barcelona</addrLine></address></meeting>
		<imprint>
			<publisher>ETFA</publisher>
			<date type="published" when="1999">1999</date>
			<biblScope unit="page" from="187" to="193" />
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b4">
	<analytic>
		<title level="a" type="main">Bioside : faciliter l&apos;acceès des biologistes aux ressources bioinformatiques</title>
		<author>
			<persName><forename type="first">M</forename><surname>Hallard</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">JOBIM</title>
		<imprint>
			<biblScope unit="page">64</biblScope>
			<date type="published" when="2004">2004</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b5">
	<analytic>
		<title level="a" type="main">Taverna: a tool for building and running workflows of services</title>
		<author>
			<persName><forename type="first">D</forename><surname>Hull</surname></persName>
		</author>
		<author>
			<persName><forename type="first">K</forename><surname>Wolstencroft</surname></persName>
		</author>
		<author>
			<persName><forename type="first">R</forename><surname>Stevens</surname></persName>
		</author>
		<author>
			<persName><forename type="first">C</forename><forename type="middle">A</forename><surname>Goble</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><forename type="middle">R</forename><surname>Pocock</surname></persName>
		</author>
		<author>
			<persName><forename type="first">P</forename><surname>Li</surname></persName>
		</author>
		<author>
			<persName><forename type="first">T</forename><surname>Oinn</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Nucleic Acids Research</title>
		<imprint>
			<biblScope unit="volume">34</biblScope>
			<biblScope unit="page">729732</biblScope>
			<date type="published" when="2006">2006</date>
		</imprint>
	</monogr>
	<note>Web-Server-Issue</note>
</biblStruct>

<biblStruct xml:id="b6">
	<analytic>
		<title level="a" type="main">A Platform Dedicated to Share and Mutualize Environmental Applications</title>
		<author>
			<persName><forename type="first">T</forename><surname>Libourel</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Y</forename><surname>Lin</surname></persName>
		</author>
		<author>
			<persName><forename type="first">I</forename><surname>Mougenot</surname></persName>
		</author>
		<author>
			<persName><forename type="first">C</forename><surname>Pierkot</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">C</forename><surname>Desconnets</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">Proceedings of 12th International Conference on Enterprise Information Systems</title>
				<meeting>12th International Conference on Enterprise Information Systems</meeting>
		<imprint>
			<publisher>Madere</publisher>
			<date type="published" when="2010">2010</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b7">
	<analytic>
		<title level="a" type="main">A Workflow Language for the Experimental Sciences</title>
		<author>
			<persName><forename type="first">Y</forename><surname>Lin</surname></persName>
		</author>
		<author>
			<persName><forename type="first">T</forename><surname>Libourel</surname></persName>
		</author>
		<author>
			<persName><forename type="first">I</forename><surname>Mougenot</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="m">Proceedings of 11th International Conference on Enterprise Information Systems</title>
				<meeting>11th International Conference on Enterprise Information Systems<address><addrLine>Milan</addrLine></address></meeting>
		<imprint>
			<date type="published" when="2009">2009</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b8">
	<monogr>
		<idno>formal/2010-05-03</idno>
		<title level="m">Object Management Group (OMG), OMG Unified Modeling LanguageTM (OMG UML), Infrastructure Version 2.3</title>
				<imprint/>
	</monogr>
	<note>OMG Document Number</note>
</biblStruct>

<biblStruct xml:id="b9">
	<monogr>
		<idno>formal/2008-04-01</idno>
		<title level="m">Object Management Group (OMG), SPEM -Software &amp; Systems Process Engineering Meta-Model Specification</title>
				<imprint/>
	</monogr>
	<note>Version 2.0. OMG Document Number</note>
</biblStruct>

<biblStruct xml:id="b10">
	<monogr>
		<idno>formal/06-</idno>
		<title level="m">Object Management Group (OMG), Meta Object Facility (MOF) Core Specification OMG Available Specification Version 2.0, OMG Document Number</title>
				<imprint>
			<biblScope unit="page" from="1" to="01" />
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b11">
	<monogr>
		<author>
			<persName><forename type="first">P</forename><forename type="middle">Pinheiro</forename><surname>Da Silva</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><surname>Salayandia</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><forename type="middle">Q</forename><surname>Gates</surname></persName>
		</author>
		<title level="m">WDO-It! A Tool for Building Scientific Workflows from Ontologies</title>
				<imprint>
			<date type="published" when="2007">2007</date>
			<biblScope unit="volume">201</biblScope>
		</imprint>
	</monogr>
	<note type="report_type">Departmental Technical Reports</note>
	<note>CS). Paper</note>
</biblStruct>

<biblStruct xml:id="b12">
	<analytic>
		<title level="a" type="main">Sequence Logos: A New Way to Display Consensus Sequences</title>
		<author>
			<persName><forename type="first">T</forename><forename type="middle">D</forename><surname>Schneider</surname></persName>
		</author>
		<author>
			<persName><forename type="first">R</forename><forename type="middle">M</forename><surname>Stephens</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Nucleic Acids Res</title>
		<imprint>
			<biblScope unit="volume">18</biblScope>
			<biblScope unit="page" from="6097" to="6100" />
			<date type="published" when="1990">1990</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b13">
	<analytic>
		<title level="a" type="main">The Protein Feature Ontology: a tool for the unification of protein feature annotations</title>
		<author>
			<persName><forename type="first">G</forename><forename type="middle">A</forename><surname>Reeves</surname></persName>
		</author>
		<author>
			<persName><forename type="first">K</forename><surname>Eilbeck</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Magrane</surname></persName>
		</author>
		<author>
			<persName><forename type="first">C</forename><surname>O'donovan</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><surname>Montecchi-Palazzi</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><forename type="middle">A</forename><surname>Harris</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><forename type="middle">E</forename><surname>Orchard</surname></persName>
		</author>
		<author>
			<persName><forename type="first">R</forename><forename type="middle">C</forename><surname>Jimenez</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Prlic</surname></persName>
		</author>
		<author>
			<persName><forename type="first">T</forename><forename type="middle">J P</forename><surname>Hubbard</surname></persName>
		</author>
		<author>
			<persName><forename type="first">H</forename><surname>Hermjakob</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">M</forename><surname>Thornton</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Bioinformatics</title>
		<imprint>
			<biblScope unit="volume">24</biblScope>
			<biblScope unit="page" from="2767" to="2772" />
			<date type="published" when="2008">2008</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b14">
	<analytic>
		<title level="a" type="main">The Sequence Ontology: a tool for the unification of genome annotations</title>
		<author>
			<persName><forename type="first">K</forename><surname>Eilbeck</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><surname>Lewis</surname></persName>
		</author>
		<author>
			<persName><forename type="first">C</forename><surname>Mungall</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Yandell</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><surname>Stein</surname></persName>
		</author>
		<author>
			<persName><forename type="first">R</forename><surname>Durbin</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Ashburner</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Genome Biology</title>
		<imprint>
			<biblScope unit="volume">6</biblScope>
			<biblScope unit="page">R44</biblScope>
			<date type="published" when="2005">2005</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b15">
	<analytic>
		<title level="a" type="main">Gene ontology: tool for the unification of biology. The Gene Ontology Consortium</title>
		<author>
			<persName><forename type="first">M</forename><surname>Ashburner</surname></persName>
		</author>
		<author>
			<persName><forename type="first">C</forename><forename type="middle">A</forename><surname>Ball</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">A</forename><surname>Blake</surname></persName>
		</author>
		<author>
			<persName><forename type="first">D</forename><surname>Botstein</surname></persName>
		</author>
		<author>
			<persName><forename type="first">H</forename><surname>Butler</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><surname>Michael</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><forename type="middle">P</forename><surname>Cherry</surname></persName>
		</author>
		<author>
			<persName><forename type="first">K</forename><surname>Davis</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><forename type="middle">S</forename><surname>Dolinski</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">T</forename><surname>Dwight</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><forename type="middle">A</forename><surname>Eppig</surname></persName>
		</author>
		<author>
			<persName><forename type="first">D</forename><forename type="middle">P</forename><surname>Harris</surname></persName>
		</author>
		<author>
			<persName><forename type="first">L</forename><surname>Hill</surname></persName>
		</author>
		<author>
			<persName><forename type="first">A</forename><surname>Issel-Tarver</surname></persName>
		</author>
		<author>
			<persName><forename type="first">S</forename><surname>Kasarskis</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">C</forename><surname>Lewis</surname></persName>
		</author>
		<author>
			<persName><forename type="first">J</forename><forename type="middle">E</forename><surname>Matese</surname></persName>
		</author>
		<author>
			<persName><forename type="first">M</forename><surname>Richardson</surname></persName>
		</author>
		<author>
			<persName><forename type="first">G</forename><forename type="middle">M</forename><surname>Ringwald</surname></persName>
		</author>
		<author>
			<persName><forename type="first">G</forename><surname>Rubin</surname></persName>
		</author>
		<author>
			<persName><surname>Sherlock</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Nature Genetics</title>
		<imprint>
			<biblScope unit="volume">25</biblScope>
			<biblScope unit="page" from="25" to="29" />
			<date type="published" when="2000">2000</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b16">
	<analytic>
		<title level="a" type="main">Semi-automatic web service composition for the life sciences using the BioMoby semantic web framework</title>
		<author>
			<persName><forename type="first">Michael</forename><surname>Dibernardo</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Rachel</forename><surname>Pottinger</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Mark</forename><surname>Wilkinson</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Journal of Biomedical Informatics</title>
		<imprint>
			<biblScope unit="volume">41</biblScope>
			<biblScope unit="issue">5</biblScope>
			<biblScope unit="page" from="837" to="847" />
			<date type="published" when="2008">2008</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b17">
	<analytic>
		<title level="a" type="main">Mobyle: a new full web bioinformatics framework</title>
		<author>
			<persName><forename type="first">Bertrand</forename><surname>Néron</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Hervé</forename><surname>Ménager</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Corinne</forename><surname>Maufrais</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Nicolas</forename><surname>Joly</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Julien</forename><surname>Maupetit</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Sébastien</forename><surname>Letort</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Sébastien</forename><surname>Carrère</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Pierre</forename><surname>Tufféry</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Catherine</forename><surname>Letondal</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Bioinformatics</title>
		<imprint>
			<biblScope unit="volume">25</biblScope>
			<biblScope unit="issue">22</biblScope>
			<biblScope unit="page" from="3005" to="3011" />
			<date type="published" when="2009">2009</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b18">
	<analytic>
		<title level="a" type="main">ProtocolDB: Storing Scientific Protocols with a Domain Ontology</title>
		<author>
			<persName><forename type="first">Michel</forename><surname>Kinsy</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Zoé</forename><surname>Lacroix</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Christophe</forename><surname>Legendre</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Piotr</forename><surname>Wlodarczyk</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Nadia</forename><forename type="middle">Yacoubi</forename><surname>Ayadi</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">WISE Workshops</title>
		<imprint>
			<biblScope unit="page" from="17" to="28" />
			<date type="published" when="2007">2007</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b19">
	<monogr>
		<author>
			<persName><forename type="first">Zoé</forename><surname>Lacroix</surname></persName>
		</author>
		<title level="m">Resource Discovery, Second International Workshop, RED 2009</title>
				<meeting><address><addrLine>Lyon, France</addrLine></address></meeting>
		<imprint>
			<date type="published" when="2009-08-28">August 28, 2009. 2010</date>
		</imprint>
	</monogr>
	<note>Revised Papers Springer</note>
</biblStruct>

<biblStruct xml:id="b20">
	<analytic>
		<title level="a" type="main">Biological Resource Discovery</title>
		<author>
			<persName><forename type="first">Zoé</forename><surname>Lacroix</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Cartik</forename><forename type="middle">R</forename><surname>Kothari</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Peter</forename><surname>Mork</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Rami</forename><surname>Rifaieh</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Mark</forename><surname>Wilkinson</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Juliana</forename><surname>Freire</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Sarah</forename><forename type="middle">Cohen</forename><surname>Boulakia</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">Encyclopedia of Database Systems</title>
		<imprint>
			<biblScope unit="page" from="220" to="223" />
			<date type="published" when="2009">2009</date>
		</imprint>
	</monogr>
</biblStruct>

<biblStruct xml:id="b21">
	<analytic>
		<title level="a" type="main">A Deductive Approach for Resource Interoperability and Well-Defined Workflows</title>
		<author>
			<persName><forename type="first">Nadia</forename><surname>Yacoubi Ayadi</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Zoé</forename><surname>Lacroix</surname></persName>
		</author>
		<author>
			<persName><forename type="first">Maria-Esther</forename><surname>Vidal</surname></persName>
		</author>
	</analytic>
	<monogr>
		<title level="j">OTM Workshops</title>
		<imprint>
			<biblScope unit="page" from="998" to="1009" />
			<date type="published" when="2008">2008</date>
		</imprint>
	</monogr>
</biblStruct>

				</listBibl>
			</div>
		</back>
	</text>
</TEI>
