<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>October</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <title-group>
        <article-title>TOWARDS RUSSIAN NATIONAL DATA LAKE PROTOTYPE</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>A. Alekseev</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>S. Campana</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>X. Espinal</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>S. Jezequel</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>A. Kiryanov</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>A. Klimentov</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>V. Mitsyn</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>A. Zarochentsev</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>CERN.</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Espl. des Particules</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Geneva</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Switzerland</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Joliot-Curie st.</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Dubna</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Russia</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Universidad Andrés Bello</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Santiago</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Chile</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Annecy-le-Veus</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>France</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Aleksandr Alekseev</institution>
          ,
          <addr-line>Simone Campana, Xavier Espinal, Stephane Jezequel, Andrey Kiryanov, Alexei Klimentov, Valery Mitsyn, Andrey Zarochentsev</addr-line>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Brookhaven National Laboratory</institution>
          ,
          <addr-line>Upton, NY</addr-line>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Institute of System Programming, Russian Academy of Science</institution>
          ,
          <addr-line>Moscow</addr-line>
          ,
          <country country="RU">Russia</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Petersburg Nuclear Physics Institute of NRC “KI”. 1 Orlova roshcha</institution>
          ,
          <addr-line>Gatchina</addr-line>
          ,
          <country country="RU">Russia</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Plekhanov Russian University of Economics.</institution>
          <addr-line>36 Stremyanny lane, Moscow</addr-line>
          ,
          <country country="RU">Russia</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>St. Petersburg State University. 13B Universitetskaya emb.</institution>
          ,
          <addr-line>Saint Petersburg</addr-line>
          ,
          <country country="RU">Russia</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2019</year>
      </pub-date>
      <volume>4</volume>
      <issue>2019</issue>
      <fpage>44</fpage>
      <lpage>50</lpage>
      <abstract>
        <p>The evolution of the computing facilities and the way storage will be organized and consolidated will play a key role in how this possible shortage of resources will be addressed by the LHC experiments. The need for an effective distributed data storage has been identified as fundamental from the beginning of LHC, and this topic has become particularly vital in the light of the preparation for the HL-LHC run. WLCG has started an R&amp;D within DOMA project and in this contribution we will report the recent results related to the Russian federated data storage systems configuration and testing. We will describe different system configurations and various approaches to test data storage federation. We are considering EOS and dCache storage systems as a backbone software for data federation and xCache for data caching. We'll also report about synthetic tests and experiments specific tests developed by ATLAS and ALICE for federated storage prototype in Russia. Data Lake project has been launched in Russian Federation in 2019 to set up a National Data Lake prototype for HENP and to consolidate geographically distributed data storage systems connected by fast network with low latency, we will report the project objectives and status.</p>
      </abstract>
      <kwd-group>
        <kwd>HL-LHC</kwd>
        <kwd>WLCG</kwd>
        <kwd>Data Lake</kwd>
        <kwd>DOMA</kwd>
        <kwd>Distributed Storage</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>1. Introduction</title>
      <p>The High Luminosity LHC (HL-LHC) will be a multi-Exabyte challenge where the envisaged
Storage and Compute needs are a factor 10–100 above the expected technology evolution and flat
funding (fig.1).</p>
      <p>The WLCG community needs to evolve current computing and data organization models in
order to introduce changes in the way it uses and manages the infrastructure, focused on optimizations
to bring performance and efficiency not forgetting simplification of operations. These are the
ingredients that will allow to drive down costs and be able to satisfy HL-LHC requirements.
Technologies that will address the HL-LHC computing challenges may be applicable for other
scientific communities (SKA, DUNE, LSST, BELLE-II, JUNO, NICA, etc.) to manage large-scale
data volumes. The evolution of the computing facilities and the way storage will be organized and
consolidated will play a key role in how this possible shortage of resources will be addressed by the
LHC experiments. The need for an effective distributed data storage has been identified as
fundamental from the beginning of LHC, and this topic has become particularly vital in the light of the
preparation for the HL-LHC run. WLCG has started several R&amp;Ds within Data Organization and
Management Project (DOMA), one of which is a Data Lake project.</p>
      <p>Data Lake is a set of sites, associated by proximity, providing together storage services,
possibly accompanied by compute nodes to an identified set of user communities, capable to carry out
independently well-defined tasks. Proximity could be defined by geography, connectivity, funding or a
shared user community. This requires that their combined storage capacity and network bandwidth can
meet the demands of the designated task and that usage of the different sites is transparent to the users,
which, in turn, implies some form of trust relationship between the sites and a way to locate data,
ranging from a simple file catalogue to a full-fledged namespace.</p>
      <p>While access for users is transparent, the population and management of the storages within
the Data Lake is a planned and managed activity. This includes the transitions between
Quality-ofService (QoS) levels. These operations are done on the granularity of the Data Lake. Data is moved to
or from the Data Lake as a whole, not to or from a specific site. Resource management within the Data
Lake is the responsibility of the Data Lake.</p>
      <p>Taking the aforementioned into account we can come up with some basic but crucial
requirements for the future WLCG sites data storage infrastructure:
 Common namespace and interoperability.
 Coexistence of different QoS.
 Geo-awareness.
 Fault tolerance through redundancy of key components.



</p>
      <p>Scalability, with the ability to change the topology without stopping the entire system.
Security with mutual authentication and authorization for data and metadata access.
Optimal data transfer routing, providing the user direct access to the closest data location.
Universality, which implies validity for a wide range of research projects of various sizes,
including, but not limited to the LHC experiments.</p>
    </sec>
    <sec id="sec-2">
      <title>2. Data Lake. Data Storage and Data Handling R&amp;D Project</title>
      <p>
        In 2015 in the framework of the Laboratory "BigData Technologies for mega-science class
projects" at NRC "Kurchatov Institute" a work has begun on the creation of a united disk resource
federation for geographically distributed data centres, located in Moscow, Saint Petersburg, Dubna,
Gatchina (all above centres are part of the Russian Data Intensive Grid (RDIG) of WLCG) and
Geneva, its integration with existing computing resources and provision of access to these resources
for applications running on both supercomputers and high throughput distributed computing systems
(Grid) [
        <xref ref-type="bibr" rid="ref1">1</xref>
        ]. The objective of these studies was to create a federated storage system with a single access
endpoint and an integrated internal management system. With such an architecture, the system looks
like a single entity for the end user, while in fact being put together from geographically distributed
resources. This work was continued as a part of award granted by the Russian Science Foundation to
the Laboratory of Cloud Computing of Plekhanov University. The concept of Russian Data Lake for
Scientific Data is described in [
        <xref ref-type="bibr" rid="ref2">2</xref>
        ]. The resources used for RF data lake prototype are located at PNPI
(Gatchina), JINR (Dubna), SPbSU (Saint Petersburg) and MEPhI (Moscow). Project milestones to be
addressed in the context of the Data Lake prototype in Russia in the next few years:
 Deploy a working Data Lake prototype
 Develop and deploy a monitoring infrastructure
 Validate data access patterns
 Develop of a testing methodology for sites and data handling, conduct and automate a
functional test suite:
o Synthetic tests including files transfer/replication to have a realistic benchmark and to
measure a performance of metadata operations;
o Experiment-specific tests for I/O-intensive (derivation data production) and
CPUintensive (Monte-Carlo simulation) payloads.
 Create a data distribution model
 Connect Data Lake to the WLCG production infrastructure in Russia
 Use Data Lake for processing of a real experiments’ data (LHC experiments + NICA)
      </p>
      <p>Data Lake for Scientific Data project supported by the Russian Science Foundation award has
been launched in Russian Federation in 2019 to set up a National Data Lake prototype for HENP and
to consolidate geographically distributed data storage systems connected by fast networks with low
latency. JINR, SPbSU, PNPI, and MEPhI groups are participating at the first stage of the project and it
is anticipated that more centres will be involved during the subsequent stages. The short-term plans for
building a distributed Data Lake system in Russian Federation are shown on fig. 2.</p>
      <p>We are considering EOS and dCache storage systems as a backbone software for data
federation and xCache for data caching. Synthetic tests and experiments specific tests have been
developed by ATLAS and ALICE for federated storage prototype in Russian Federation.
site CE
xCache
site CE
xCache
site CE</p>
      <p>xCache
JINR SE
dCache</p>
      <p>EOS pools
site CE
xCache
site CE
xCache
site CE
xCache</p>
      <p>The Computing Element (CE) + xCache computing infrastructure has been set up at PNPI and
access to it has been configured from JINR. Sites’ technical characteristics are shown below:
 Worker Node @ JINR: 8 cores, Xeon E5420, 16GB RAM, 8.74 HEP-SPEC06 per Core
 Worker Node @ PNPI: 8 cores, Xeon E5-2680, 32GB RAM (VM), ~11 HEP-SPEC06 per</p>
      <p>Core
 Local network @ JINR (SE ⇄ CE) 1 Gbps
 Local network @ PNPI (SE ⇄ CE) 10 Gbps
 Network IPv4,6 JINR → PNPI: Latency ~5 ms
 Network IPv4,6 PNPI → JINR: Latency ~10 ms
 Network IPv4,6 JINR → PNPI: Throughput ~1 Gbps
 Network IPv4,6 PNPI → JINR: Throughput ~1,5 Gbps




</p>
      <p>
        The following authorization parameters were configured and tested:
PNPI xCache → JINR SE: GSI authorization by local gridmapfile on JINR SE
PNPI WN → PNPI xCache: GSI authorization by VOMS (ALICE &amp; ATLAS)
PNPI UI → JINR CE, PNPI CE (for local tests): GSI authorization by VOMS (ALICE &amp;
ATLAS)
Hammer Cloud → ALL: GSI authorization by VOMS (ATLAS)
An external library for VOMS authorization in xCache [
        <xref ref-type="bibr" rid="ref3">3</xref>
        ]
      </p>
      <p>The following tests were conducted during the infrastructure set-up phase:
1. Synthetic tests from Worker Nodes and through Cream-CE
2. ATLAS HammerCloud tests:
a. Copy2scratch: data copy from WN to scratch area
b. Directaccess: remote data access</p>
      <p>Tests were conducted on 3 configurations:
1. Direct access from PNPI WN to JINR SE: “PNPI direct”
2. Access through xCache from PNPI WN to JINR SE: “PNPI xCache”
3. Local access from JINR WN to JINR SE: “JINR local”</p>
    </sec>
    <sec id="sec-3">
      <title>3. Synthetic tests</title>
      <p>Synthetic tests are composed of sequential transfers of identical files accessing the storage in
three different ways: directly from the remote site, through xCache at the remote site (PNPI) and
directly from the site local to the storage (JINR). Synthetic test results are as follows:
 For “PNPI direct” the speed was 650±40 Mbps (fig. 3-1).</p>
      <p>EOS pools</p>
      <p>JINR SE
EOS mgm
EOS pools
site CE</p>
      <p>xCache</p>
      <p>EOS pools
For “PNPI xCache” the speed was 6700±700 Mbps, but if we remove the first hit (cache
warm-up) we will get the same speed with a much lower deviation. As a result, with 100 hits
we have a 95% transfer speed gain and 92% transfer time gain (fig. 3-2). For xCache transfers
the first warm-up access to a new file is highlighted with a red circle.</p>
      <p>For “JINR local” with a 1 Gbps internal network tests show an expected transfer speed of
670±220 Mbps with a pretty high deviation (fig. 3-3).</p>
    </sec>
    <sec id="sec-4">
      <title>4. ATLAS HammerCloud tests</title>
      <p>The tests have been conducted for two days and results are presented in table 1. We can
estimate a file transfer speed for Copy2Sctartch tests by “Download input files” metric, and for
directaccess tests by “Athena Run Time” metric. Results of HammerClouds tests demonstrate 30%
gain in time for copy2scratch and 12% gain in time for directaccess.</p>
    </sec>
    <sec id="sec-5">
      <title>5. Monitoring</title>
      <p>
        All of the components in the aforementioned tests were monitored by various monitoring
systems. Synthetic tests were run through WLCG middleware (CREAM-CE) and tasks execution
progress was monitored directly by CREAM-CE logs and monitoring tools. Tasks that were launched
using HammerCloud were tracked using HammerCloud and BigPanda monitoring [
        <xref ref-type="bibr" rid="ref6">6</xref>
        ]. The state of
network links was monitored using the mesh of perfSONAR [
        <xref ref-type="bibr" rid="ref7">7</xref>
        ] systems deployed at all participating
sites sites. Separately, the work of the xCache service was monitored by Kibana monitoring [
        <xref ref-type="bibr" rid="ref8">8</xref>
        ] (fig.
4). All virtual nodes at PNPI were monitored by the Zabbix [
        <xref ref-type="bibr" rid="ref9">9</xref>
        ] suite (fig. 5).
      </p>
    </sec>
    <sec id="sec-6">
      <title>6. Summary</title>
      <p>Data Lake for Scientific Data is a three years R&amp;D project started in Russian Federation in
2019 as a continuation of the successful Federated Data Storage project. Production-grade computing
resources and 10–100 Gbps network connectivity with low latency is used to prototype a production
Data Lake. The project will have two phases:
1. Four Russian WLCG centres (JINR, PNPI, SPbSU and MEPhI) will participate and they will
be configured to test an initial prototype. The test methodology and consolidated monitoring
tools should be developed, as well as we need a better understanding of xCache data access
control scenarios.
2. More RDIG centres will participate to have a realistic set of production resources and
scattered storages. This prototype will be tested for real LHC use-cases, monitoring tools will
be provided to sites and experiments.</p>
    </sec>
    <sec id="sec-7">
      <title>7. Acknowledgements</title>
      <p>The Russian Data Lake R&amp;D project is funded by the Russian Science Foundation under
contract No.19-71-30008 (research is conducted in Plekhanov Russian University of Economics). The
work at PNPI is supported by the NRC "Kurchatov Institute" order No. 1571.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [1]
          <string-name>
            <given-names>A.</given-names>
            <surname>Kiryanov</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Klimentov</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Krasnopevtsev</surname>
          </string-name>
          ,
          <string-name>
            <given-names>E.</given-names>
            <surname>Ryabinkin</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Zarochentsev</surname>
          </string-name>
          ,
          <article-title>Federated data storage system prototype for LHC experiments and data intensive science // 2017</article-title>
          <string-name>
            <given-names>J.</given-names>
            <surname>Phys</surname>
          </string-name>
          .:
          <source>Conf. Ser. 898 062016</source>
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [2]
          <string-name>
            <surname>Kiryanov</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Klimentov</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Zarochentsev</surname>
          </string-name>
          .
          <source>Russian scientific data lake // Open Systems Journal, issue 4</source>
          ,
          <year>2018</year>
          . Available at: https://www.osp.ru/os/2018/04/13054563/ (accessed 13.11.
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          <article-title>[3] XRootD repository</article-title>
          . https://github.com/opensciencegrid/xrootd-lcmaps
          <source>(accessed 13.11</source>
          .
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [4]
          <string-name>
            <given-names>HammerCloud</given-names>
            <surname>Distributed</surname>
          </string-name>
          <article-title>Analysis testing system</article-title>
          . Available at: http://hammercloud.cern.ch/hc/ (accessed 13.11.
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          [5]
          <string-name>
            <given-names>Johannes</given-names>
            <surname>Elmsheuser</surname>
          </string-name>
          et al,
          <article-title>Improving ATLAS grid site reliability with functional tests using HammerCloud// 2012</article-title>
          <string-name>
            <given-names>J.</given-names>
            <surname>Phys</surname>
          </string-name>
          .:
          <source>Conf. Ser. 396 032066</source>
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          [6]
          <string-name>
            <given-names>A.</given-names>
            <surname>Alekseev</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Klimentov</surname>
          </string-name>
          ,
          <string-name>
            <given-names>T.</given-names>
            <surname>Korchuganova</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Padolski</surname>
          </string-name>
          , T. Wenaus, ATLAS BigPanDA monitoring// 2018 J.
          <source>Phys: Conf. Ser. 1085. 032043</source>
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          <article-title>[7] PerfSONAR official site</article-title>
          . Available at: https://www.perfsonar.
          <source>net/ (accessed 13.11</source>
          .
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref8">
        <mixed-citation>
          [8]
          <string-name>
            <given-names>ATLAS</given-names>
            <surname>Kibana</surname>
          </string-name>
          in Chicago. Available at: https://atlas-kibana.
          <year>mwt2</year>
          .org:5601/s/xcache/app/kibana (accessed
          <volume>13</volume>
          .11.
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
      <ref id="ref9">
        <mixed-citation>
          <article-title>[9] Zabbix official site</article-title>
          . Available at: https://www.zabbix.
          <source>com/ (accessed 13.11</source>
          .
          <year>2019</year>
          )
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>