<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE raweb PUBLIC "-//INRIA//DTD " "raweb2.dtd">
<raweb xmlns:xlink="http://www.w3.org/1999/xlink" xml:lang="en" year="2012">
  <identification id="kerdata" isproject="true">
    <shortname>KERDATA</shortname>
    <projectName>Scalable Storage for Clouds and Beyond</projectName>
    <theme-de-recherche>Distributed and High Performance Computing</theme-de-recherche>
    <domaine-de-recherche>Networks, Systems and Services, Distributed Computing</domaine-de-recherche>
    <urlTeam>http://www.irisa.fr/kerdata/</urlTeam>
    <datecreation type="Project-Team">July 01, 2012 </datecreation>
    <structure_exterieure type="Labs">
      <libelle>Institut de recherche en informatique et systèmes aléatoires (IRISA)</libelle>
    </structure_exterieure>
    <structure_exterieure type="Organism">
      <libelle>Institut national des sciences appliquées de Rennes</libelle>
    </structure_exterieure>
    <structure_exterieure type="Organism">
      <libelle>Université Rennes 1</libelle>
    </structure_exterieure>
    <structure_exterieure type="Organism">
      <libelle>Ecole normale supérieure de Cachan</libelle>
    </structure_exterieure>
    <UR name="Rennes"/>
    <keywords>
      <term>High Performance Computing</term>
      <term>Big Data</term>
      <term>Cloud Computing</term>
      <term>Middleware</term>
      <term>Data Management</term>
      <term>Data Storage</term>
    </keywords>
    <moreinfo>
      <p>The KerData project-team is associated with the IRISA CNRS joint laboratory, University Rennes 1,
INSA Rennes and ENS Cachan/Rennes.</p>
    </moreinfo>
  </identification>
  <team id="uid1">
    <person key="paris-2006-idm124332495696">
      <firstname>Gabriel</firstname>
      <lastname>Antoniu</lastname>
      <affiliation>INRIA</affiliation>
      <categoryPro>Chercheur</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Team leader, Senior
Researcher (DR2) Inria.</moreinfo>
      <hdr>oui</hdr>
    </person>
    <person key="paris-2006-idm124332467968">
      <firstname>Luc</firstname>
      <lastname>Bougé</lastname>
      <affiliation>UnivFr</affiliation>
      <categoryPro>Enseignant</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Professor, ENS
Cachan/Rennes.</moreinfo>
      <hdr>oui</hdr>
    </person>
    <person key="kerdata-2009-idm140027562000">
      <firstname>Alexandru</firstname>
      <lastname>Costan</lastname>
      <affiliation>UnivFr</affiliation>
      <categoryPro>Enseignant</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Associate Professor,
INSA Rennes, since September 1, 2012. Before this date, he was an
Inria Post-Doctoral Fellow until August 31, 2012, supported by the
MapReduce ANR project.</moreinfo>
    </person>
    <person key="algorille-2007-idm186081019872">
      <firstname>Louis-Claude</firstname>
      <lastname>Canon</lastname>
      <affiliation>INRIA</affiliation>
      <categoryPro>PostDoc</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Post-Doctoral Fellow,
funded by the Microsoft Research-Inria A-Brain project since October
2011.</moreinfo>
    </person>
    <person key="kerdata-2011-idm367107193856">
      <firstname>Shadi</firstname>
      <lastname>Ibrahim</lastname>
      <affiliation>INRIA</affiliation>
      <categoryPro>PostDoc</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Inria Post-Doctoral Fellow,
since November 2012. Before this date, he was supported by
the Hemera Large Wingspan Project since October 2011.</moreinfo>
    </person>
    <person key="kerdata-2009-idm140027572976">
      <firstname>Viet-Trung</firstname>
      <lastname>Tran</lastname>
      <affiliation>UnivFr</affiliation>
      <categoryPro>PhD</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>MESR Grant until September
2012, and the ACET Inria. PhD to be defended on January 21,
2013. PhD thesis started in October 2009.</moreinfo>
    </person>
    <person key="kerdata-2010-idm58934898256">
      <firstname>Houssem-Eddine</firstname>
      <lastname>Chihoub</lastname>
      <affiliation>INRIA</affiliation>
      <categoryPro>PhD</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>SCALUS FP7
Marie-Curie Initial Training Network Grant. PhD thesis started in
October 2010.</moreinfo>
    </person>
    <person key="kerdata-2011-idm367107176512">
      <firstname>Radu</firstname>
      <lastname>Tudoran</lastname>
      <affiliation>UnivFr</affiliation>
      <categoryPro>PhD</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>MESR Grant from University
Rennes 1. PhD thesis started in October 2011</moreinfo>
    </person>
    <person key="kerdata-2009-idm140027550560">
      <firstname>Matthieu</firstname>
      <lastname>Dorier</lastname>
      <affiliation>UnivFr</affiliation>
      <categoryPro>PhD</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>MESR Grant from ENS Cachan. PhD
thesis started in October 2011</moreinfo>
    </person>
    <person key="kerdata-2012-idm485253319088">
      <firstname>Zhe</firstname>
      <lastname>Li</lastname>
      <affiliation>INRIA</affiliation>
      <categoryPro>Technique</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Research Engineer, fixed-term
contract, funded through the BlobSeer Technology Development Action
(ADT) since November 2012.</moreinfo>
    </person>
    <person key="kerdata-2011-idm367107170400">
      <firstname>Bunjamin</firstname>
      <lastname>Memishi</lastname>
      <affiliation>UnivEtrangere</affiliation>
      <categoryPro>Visiteur</categoryPro>
      <research-centre>Rennes</research-centre>
    </person>
    <person key="kerdata-2011-idm367107157248">
      <firstname>Elena</firstname>
      <lastname>Apostol</lastname>
      <affiliation>UnivEtrangere</affiliation>
      <categoryPro>Visiteur</categoryPro>
      <research-centre>Rennes</research-centre>
    </person>
    <person key="kerdata-2012-idm485253310656">
      <firstname>Bharath</firstname>
      <lastname>Vissapragada</lastname>
      <affiliation>UnivEtrangere</affiliation>
      <categoryPro>Visiteur</categoryPro>
      <research-centre>Rennes</research-centre>
    </person>
    <person key="bunraku-2010-idm189804545168">
      <firstname>Céline</firstname>
      <lastname>Gharsalli</lastname>
      <affiliation>CNRS</affiliation>
      <categoryPro>Assistant</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Team Administrative
Assistant, CNRS, until September 2012.</moreinfo>
    </person>
    <person key="texmex-2011-idm349566503888">
      <firstname>Élodie</firstname>
      <lastname>Lequoc</lastname>
      <affiliation>UnivFr</affiliation>
      <categoryPro>Assistant</categoryPro>
      <research-centre>Rennes</research-centre>
      <moreinfo>Team Administrative
Assistant, University Rennes 1, since September 2012.</moreinfo>
    </person>
  </team>
  <presentation id="uid2">
    <bodyTitle>Overall Objectives</bodyTitle>
    <subsection id="uid3" level="1">
      <bodyTitle>Context: the need
for scalable data management</bodyTitle>
      <p>We are witnessing a rapidly increasing number of application areas
generating and processing very large volumes of data on a regular
basis. Such applications are called
<i>data-intensive</i>. Governmental and commercial statistics,
climate modeling, cosmology, genetics, bio-informatics, high-energy
physics are just a few examples. In these fields, it becomes crucial
to efficiently store and manipulate massive data, which are
typically <i>shared</i> at a large scale and <i>concurrently
accessed</i>. In all these examples, the overall application
performance is highly dependent on the properties of the underlying
data management service. With the emergence of recent
infrastructures such as cloud computing platforms and post-Petascale
architectures, achieving highly scalable data management has become
a critical challenge.</p>
      <p>The KerData project-team is namely focusing on <i>scalable data
storage and processing on clouds and post-Petascale platforms</i>,
according to the current needs and requirements of data-intensive
applications. We are especially concerned by the applications of
major international and industrial players in Cloud Computing and
post-Petascale High-Performance Computing (HPC), which shape the
longer-term agenda of the Cloud Computing and Exascale HPC research
communities.</p>
      <p>Our research activities focus on data-intensive high-performance
applications that exhibit the need to handle:</p>
      <simplelist>
        <li id="uid4">
          <p noindent="true">massive data BLOBs (Binary Large OBjects), in the order of
Terabytes,</p>
        </li>
        <li id="uid5">
          <p noindent="true">stored in a large number of nodes, thousands to tens of
thousands,</p>
        </li>
        <li id="uid6">
          <p noindent="true">accessed under heavy concurrency by a large number of
processes, thousands to tens of thousands at a time,</p>
        </li>
        <li id="uid7">
          <p noindent="true">with a relatively fine access grain, in the order of
Megabytes.</p>
        </li>
      </simplelist>
      <p>Examples of such applications are:</p>
      <simplelist>
        <li id="uid8">
          <p noindent="true">Massively parallel cloud data-mining applications (e.g.,
MapReduce-based data analysis).</p>
        </li>
        <li id="uid9">
          <p noindent="true">Advanced Platform-as-a-Service (PaaS) cloud data services
requiring efficient data sharing under heavy concurrency.</p>
        </li>
        <li id="uid10">
          <p noindent="true">Advanced concurrency-optimized, versioning-oriented cloud
services for virtual machine image storage and management at IaaS
(Infrastructure-as-a-Service) level.</p>
        </li>
        <li id="uid11">
          <p noindent="true">Scalable storage solutions for I/O-intensive HPC simulations for
post-Petascale architectures.</p>
        </li>
        <li id="uid12">
          <p noindent="true">Storage and I/O stacks for big data analysis in applications
that manipulate structured scientific data (e.g. very large
multi-dimensional arrays).</p>
        </li>
      </simplelist>
    </subsection>
    <subsection id="uid13" level="1">
      <bodyTitle>Highlights of the Year</bodyTitle>
      <descriptionlist>
        <li id="uid14">
          <p noindent="true">The KerData project-team has been officially created on
July 1st 2012 as a Joint Project-Team with ENS Cachan/Brittany and INSA Rennes, for a 4-year term.</p>
        </li>
        <li id="uid15">
          <p noindent="true">Alexandru Costan, a former Post-Doc fellow at the KerData
project-team, has been hired on a permanent position at INSA. Alexandru got
his PhD in Valentin Cristea's NCIT group at Polytechnic University
of Bucharest (Romania), our partner in the <i>DataCloud@work</i>
Inria Associate Team.</p>
        </li>
        <li id="uid16">
          <p noindent="true">The KerData project-team organized the 7th Workshop of the
Inria-Illinois Joint Laboratory on Petascale Computing, June
13-15, 2012 <ref xlink:href="http://jointlab.ncsa.illinois.edu/events/workshop7/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>jointlab.<allowbreak/>ncsa.<allowbreak/>illinois.<allowbreak/>edu/<allowbreak/>events/<allowbreak/>workshop7/</ref>.</p>
        </li>
        <li id="uid17">
          <p noindent="true">After successful experiments with up to 9000 cores on the Kraken Cray XT5 machine (NICS) in 2011, Damaris scaled up to 16000 cores on Oak Ridge's leadership supercomputer Titan (now first in the Top500), providing in-situ analysis to the CM1 tornado simulation.</p>
        </li>
      </descriptionlist>
    </subsection>
  </presentation>
  <fondements id="uid18">
    <bodyTitle>Scientific Foundations</bodyTitle>
    <subsection id="uid19" level="1">
      <bodyTitle>Our goals and
methodology</bodyTitle>
      <p><i>Data-intensive applications</i> demonstrate common
requirements with respect to the need for data storage and I/O
processing. These requirements lead to several core challenges
discussed below.</p>
      <descriptionlist>
        <label>Challenges related to cloud storage.</label>
        <li id="uid20">
          <p noindent="true">In the area of cloud data
management, a significant milestone is the emergence of the
Map-Reduce  <ref xlink:href="#kerdata-2012-bid0" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> parallel programming paradigm,
currently used on most cloud platforms, following the trend set up
by Amazon  <ref xlink:href="#kerdata-2012-bid1" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. At the core of the Map-Reduce
frameworks stays a key component, which must meet a series of
specific requirements that have not fully been met yet by existing
solutions: the ability to provide efficient <i>fine-grain
access</i> to the files, while sustaining a <i>high throughput</i>
in spite of <i>heavy access concurrency</i>. Additionally, as
thousands of clients simultaneously access shared data, it is
critical to preserve <i>fault-tolerance</i> and <i>security</i>
requirements.</p>
        </li>
        <label>Challenges related to data-intensive HPC applications.</label>
        <li id="uid21">
          <p noindent="true">The
requirements exhibited by climate simulations
specifically highlights a major, more general research topic. It
has been clearly identified by international panels of experts
like IESP  <ref xlink:href="#kerdata-2012-bid2" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> and EESI  <ref xlink:href="#kerdata-2012-bid3" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, in the context of HPC
simulations running on post-Petascale supercomputers. A jump of
one order of magnitude in the size of numerical simulations is
required to address some of the fundamental questions in several
communities such as climate modeling, solid earth sciences or
astrophysics. In this context, the lack of data-intensive
infrastructure and methodology to analyze huge simulations is a
growing limiting factor. The challenge is to find new ways to
store and analyze massive outputs of data during and after the
simulation without impacting the overall performance.</p>
        </li>
      </descriptionlist>
      <p>The overall goal of the KerData project-team is to bring a substantial
contribution to the effort of the research community to address the
above challenges. KerData aims to design and implement distributed
algorithms for scalable data storage and input/output management for
efficient large-scale data processing. We target two main execution
infrastructures: cloud platforms and post-Petascale HPC
supercomputers. We are also looking at other kinds of
infrastructures (that we are considering as secondary), e.g. hybrid
platforms combining enterprise desktop grids extended to cloud
platforms. Our collaboration porfolio includes international teams that are
active in this area both in Academia (e.g., Argonne National Lab,
University of Illinois at Urbana-Champaign, University of Tsukuba)
and Industry (Microsoft, IBM).</p>
      <p>The highly experimental nature of our research validation
methodology should be stressed. Our approach relies on building
prototypes and on their large-scale experimental validation on real
testbeds and experimental platforms. We strongly rely on the
ALADDIN-Grid'5000 platform. Moreover, thanks to our projects and
partnerships, we have access to reference software and physical
infrastructures in the cloud area (Microsoft Azure, Amazon clouds,
Nimbus clouds); in the post-Petascale HPC area we have access to the
Jaguar and Kraken supercomputers (ranked 3rd and 11th respectively
in the Top 500 supercomputer list) and, hopefully soon, to the Blue
Waters supercomputer). This provides us with excellent opportunities
to validate our results on realistic platforms.</p>
      <p>Moreover, the consortiums of our current projects include
application partners in the areas of Bio-Chemistry, Neurology and
Genetics, and Climate Simulations. This is an additional asset, it
enables us to take into account application requirements in the
early design phase of our solutions, and to validate those solutions
with real applications. We intend to continue increasing our
collaborations with application communities, as we believe that this
a key to perform effective research with a high potential impact.</p>
    </subsection>
    <subsection id="uid22" level="1">
      <bodyTitle>Our research agenda</bodyTitle>
      <p>Three typical application scenarios are described in
Section <ref xlink:href="#uid27" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>:</p>
      <simplelist>
        <li id="uid23">
          <p noindent="true">Joint genetic and neuroimaging data analysis on Azure clouds</p>
        </li>
        <li id="uid24">
          <p noindent="true">Structural protein analysis on Nimbus clouds</p>
        </li>
        <li id="uid25">
          <p noindent="true">I/O intensive climate simulations for the Blue
Waters post-Petascale machine</p>
        </li>
      </simplelist>
      <p>They illustrate the above challenges in some specific ways. They all
exhibit a common scheme: massively concurrent processes which access
massive data at a fine granularity, where data is shared and
distributed at a large scale. To efficiently address the
aforementioned challenges we have started to work out an approach
called BlobSeer, which stands today at the center of our research
efforts. This approach relies on the design and implementation of
<i>scalable</i> distributed algorithms for data storage and
access. They combine advanced techniques for decentralized metadata
and data management, with versioning-based concurrency control to
optimize the performance of applications under heavy access
concurrency.</p>
      <p>Preliminary experiments with our BlobSeer BLOB management system
within today's cloud software infrastructures proved very
promising. Recently, we used the BlobSeer approach as a starting point
to address more in depth two usage scenarios, which led to two more
specific approaches: 1) Pyramid (which borrows many concepts from
BlobSeer), with a specific focus on array-oriented storage; and 2)
Damaris (totally independent of BlobSeer), which exploits multicore
parallelism in post-Petascale supercomputers. All these directions
are described below.</p>
      <p>Our short- and medium-term research plan is devoted to storage
challenges in two main contexts: clouds and post-Petascale HPC
architectures. Consequently, our research plan is split in two main
themes, which correspond to their respective challenges. For each of
those themes, we have initiated several actions through collaborative
projects coordinated by KerData, which define our agenda for the next
4 years.</p>
      <p>Based on very promising results demonstrated by this approach in
preliminary experiments  <ref xlink:href="#kerdata-2012-bid4" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, we
have initiated several collaborative projects led by KerData in the
area of cloud data management, e.g., the MapReduce ANR project, the
A-Brain Microsoft-Inria project. Such frameworks are for us concrete
and efficient means to work in close connection with strong partners
already well positioned in the area of cloud computing research.
Thanks to those projects, we have already started to enjoy a visible
scientific positioning at the international level.</p>
      <p>The particularly active DataCloud@work Associate Team creates the
framework for an enlarged research activity involving a large number
of young researchers and students. It serves as a basis for extended
research activities based on our approaches, carried out beyond the
frontiers of our team. In the HPC area, our presence in the research
activities of the Joint UIUC-Inria Lab for Petascale Computing at
Urbana-Champaign is a very exciting opportunity that we have started
to leverage. It facilitates high-quality collaborations and access to
some of the most powerful supercomputers, an important asset which
already helped us produce and transfer some results, as described in
Section <ref xlink:href="#uid51" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
    </subsection>
  </fondements>
  <domaine id="uid26">
    <bodyTitle>Application Domains</bodyTitle>
    <subsection id="uid27" level="1">
      <bodyTitle>Application Domains</bodyTitle>
      <p>Below are three examples which illustrate the needs of large-scale
data-intensive applications with respect to storage, I/O and data
analysis. They illustrate the classes of applications that can
benefit from our research activities.</p>
      <subsection id="idp140392514074384" level="2">
        <bodyTitle>Joint genetic and neuroimaging data analysis on
Azure clouds</bodyTitle>
        <p>Joint acquisition of neuroimaging and genetic data on large cohorts
of subjects is a new approach used to assess and understand the
variability that exists between individuals, and that has remained
poorly understood so far. As both neuroimaging- and genetic-domain
observations represent a huge amount of variables (of the order of
millions), performing statistically rigorous analyses on such
amounts of data is a major computational challenge that cannot be
addressed with conventional computational techniques only. On the
one hand, sophisticated regression techniques need to be used in
order to perform significant analysis on these large datasets; on
the other hand, the cost entailed by parameter optimization and
statistical validation procedures (e.g. permutation tests) is very
high.</p>
        <p>The A-Brain (AzureBrain) Project started in October 2010 within the
Microsoft Research-Inria Joint Research Center. It is co-led by the
KerData (Rennes) and Parietal (Saclay) Inria teams. They jointly
address this computational problem using cloud related techniques on
Microsoft Azure cloud infrastructure. The two teams bring together
their complementary expertise: KerData in the area of scalable cloud
data management, and Parietal in the field of neuroimaging and
genetics data analysis.</p>
        <p>In particular, KerData brings its expertise in designing solutions
for optimized data storage and management for the Map-Reduce
programming model. This model has recently arisen as a very
effective approach to develop high-performance applications over
very large distributed systems such as grids and now clouds. The
computations involved in the statistical analysis designed by the
Parietal team fit particularly well with this model.</p>
      </subsection>
      <subsection id="idp140392514077984" level="2">
        <bodyTitle>Structural protein analysis on Nimbus clouds</bodyTitle>
        <p>Proteins are major components of the life. They are involved in lots
of biochemical reactions and vital mechanisms for the living
organisms. The three-dimensional (3D) structure of a protein is
essential for its function and for its participation to the whole
metabolism of a living organism. However, due to experimental
limitations, only few protein structures (roughly, 60,000) have been
experimentally determined, compared to the millions of proteins
sequences which are known. In the case of structural genomics, the
knowledge of the 3D structure may be not sufficient to infer the
function. Thus, an usual way to make a structural analysis of a
protein or to infer its function is to compare its known, or
potential, structure to the whole set of structures referenced in
the <i>Protein Data Bank</i> (PDB).</p>
        <p>In the framework of the MapReduce ANR project led by KerData, we
focus on the SuMo application (<i>Surf the Molecules</i>) proposed
by Institute for Biology and Chemistry of the Proteins from Lyon
(IBCP, a partner in the MapReduce project). This application
performs structural protein analysis by comparing a set of protein
structures against a very large set of structures stored in a huge
database. This is a typical data-intensive application that can
leverage the Map-Reduce model for a scalable execution on
large-scale distributed platforms. Our goal is to explore
storage-level concurrency-oriented optimizations to make the SuMo
application scalable for large-scale experiments of protein
structures comparison on cloud infrastructures managed using the
Nimbus IaaS toolkit developed at Argonne National Lab (USA).</p>
        <p>If the results are convincing, then they can immediately be applied
to the derived version of this application for drug design in an
industrial context, called MED-SuMo, a software managed by the MEDIT
SME (also a partner in this project). For pharmaceutical and biotech
industries, such an implementation run over a cloud computing
facility opens several new applications for drug design. Rather than
searching for 3D similarity into biostructural data, it will become
possible to classify the entire biostructural space and to
periodically update all derivative predictive models with new
experimental data. The applications in that complete chemo-proteomic
vision concern the identification of new druggable protein targets
and thereby the generation of new drug candidates.</p>
      </subsection>
      <subsection id="idp140392514081392" level="2">
        <bodyTitle>I/O intensive climate simulations for the Blue
Waters post-Petascale machine</bodyTitle>
        <p>A major research topic in the context of HPC simulations running on
post-Petascale supercomputers is to explore how to efficiently
record and visualize data during the simulation without impacting
the performance of the computation generating that data.
Conventional practice consists in storing data on disk, moving it
off-site, reading it into a workflow, and analyzing it. It becomes
increasingly harder to use because of the large data volumes
generated at fast rates, in contrast to limited back-end
speeds. Scalable approaches to deal with these I/O limitations are
thus of utmost importance. This is one of the main challenges
explicitly stated in the roadmap of the Blue Waters Project
(<ref xlink:href="http://www.ncsa.illinois.edu/BlueWaters/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>www.<allowbreak/>ncsa.<allowbreak/>illinois.<allowbreak/>edu/<allowbreak/>BlueWaters/</ref>), which aims to
build one of the most powerful supercomputers in the world when it
comes online in 2012.</p>
        <p>In this context, the KerData project-team started to explore ways to
remove the limitations mentioned above through a collaborative work
in the framework of the Joint Inria-UIUC Lab for Petascale Computing
(JLPC, Urbana-Champaign, Illinois, USA), whose research activity
focuses on the Blue Waters project. As a starting point, we are
focusing on a particular tornado simulation code called CM1 (Cloud
Model 1), which is intended to be run on the Blue Waters
machine. Preliminary investigation demonstrated the inefficiency of
the current I/O approaches, which typically consists in periodically
writing a very large number of small files. This causes burst of I/O
in the parallel file system, leading to poor performance and extreme
variability (jitter) compared to what could be expected from the
underlying hardware. The challenge here is to investigate how to
make an efficient use of the underlying file system by avoiding
synchronization and contention as much as possible. In collaboration
with the JLPC, we started to address those challenges through an
approach based on dedicated I/O cores.</p>
      </subsection>
    </subsection>
  </domaine>
  <logiciels id="uid28">
    <bodyTitle>Software</bodyTitle>
    <subsection id="uid29" level="1">
      <bodyTitle>BlobSeer</bodyTitle>
      <participants>
        <person key="kerdata-2009-idm140027572976">
          <firstname>Viet-Trung</firstname>
          <lastname>Tran</lastname>
        </person>
        <person key="kerdata-2012-idm485253319088">
          <firstname>Zhe</firstname>
          <lastname>Li</lastname>
        </person>
        <person key="kerdata-2009-idm140027562000">
          <firstname>Alexandru</firstname>
          <lastname>Costan</lastname>
        </person>
        <person key="paris-2006-idm124332495696">
          <firstname>Gabriel</firstname>
          <lastname>Antoniu</lastname>
        </person>
        <person key="paris-2006-idm124332467968">
          <firstname>Luc</firstname>
          <lastname>Bougé</lastname>
        </person>
      </participants>
      <descriptionlist>
        <label>Contact:</label>
        <li id="uid30">
          <p noindent="true">Gabriel Antoniu.</p>
        </li>
        <label>Presentation:</label>
        <li id="uid31">
          <p noindent="true">BlobSeer is the core software platform for
most current projects of the KerData team. It is a data storage
service specifically designed to deal with the requirements of
large-scale data-intensive distributed applications that abstract
data as huge sequences of bytes, called BLOBs (Binary Large
OBjects). It provides a versatile versioning interface for
manipulating BLOBs that enables reading, writing and appending to
them.</p>
          <p>BlobSeer offers both scalability and performance with respect to
a series of issues typically associated with the data-intensive
context: <i>scalable aggregation of storage space</i> from the
participating nodes with minimal overhead, ability to store
<i>huge data objects</i>, <i>efficient fine-grain access</i> to
data subsets, <i>high throughput in spite of heavy access
concurrency</i>, as well as <i>fault-tolerance</i>.</p>
        </li>
        <label>Users:</label>
        <li id="uid32">
          <p noindent="true">Work is currently in progress in several formalized
projects (see previous section) to integrate and leverage
BlobSeer as a data storage back-end in the reference cloud
environments: a) Microsoft Azure; b) the Nimbus cloud toolkit
developed at Argonne National Lab (USA); and c) in the OpenNebula
IaaS cloud environment developed at UCM (Madrid).</p>
        </li>
        <label>URL:</label>
        <li id="uid33">
          <p noindent="true">
            <ref xlink:href="http://blobseer.gforge.inria.fr/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>blobseer.<allowbreak/>gforge.<allowbreak/>inria.<allowbreak/>fr/</ref>
          </p>
        </li>
        <label>License:</label>
        <li id="uid34">
          <p noindent="true">GNU Lesser General Public License (LGPL) version 3.</p>
        </li>
        <label>Status:</label>
        <li id="uid35">
          <p noindent="true">This software is available on Inria's forge.
Version 1.0 (released late 2010) registered with APP:
IDDN.FR.001.310009.000.S.P.000.10700.</p>
        </li>
      </descriptionlist>
      <p>A new <i>Technology Research Action</i> (ADT, <i>Action de
recherche technologique</i>) has been launched in Septembre 2012 for
one year, with a possible 1-year renewal, to robustify the BlobSeer
software and and make it a safeky distributable product. This
project is funded by Inria <i>Technological Development
Office</i> (D2T, <i>Direction du Développement
Technologique</i>). Zhe Li has been hired as a senior (PhD) engineer
for this task.</p>
    </subsection>
    <subsection id="uid36" level="1">
      <bodyTitle>Damaris</bodyTitle>
      <participants>
        <person key="kerdata-2009-idm140027550560">
          <firstname>Matthieu</firstname>
          <lastname>Dorier</lastname>
        </person>
        <person key="paris-2006-idm124332495696">
          <firstname>Gabriel</firstname>
          <lastname>Antoniu</lastname>
        </person>
      </participants>
      <descriptionlist>
        <label>Contact:</label>
        <li id="uid37">
          <p noindent="true">Gabriel Antoniu.</p>
        </li>
        <label>Presentation:</label>
        <li id="uid38">
          <p noindent="true">Damaris is a middleware for multicore SMP nodes
enabling them to efficiently handle data transfers for storage and
visualization. The key idea is to dedicate one or a few cores of
each SMP node to the application I/O. It is developed within the
framework of a collaboration between KerData and the Joint
Laboratory for Petascale Computing (JLPC). The current version
enables efficient asynchronous I/O, hiding all I/O related
overheads such as data compression and post-processing. On-going
work is targeting fast direct access to the data from running
simulations, and efficient I/O scheduling.</p>
        </li>
        <label>Users:</label>
        <li id="uid39">
          <p noindent="true">Damaris has been preliminarily evaluated at NCSA
(Urbana-Champaign) with the CM1 tornado simulation code. CM1 is
one of the target applications of the Blue Waters supercomputer
developed by at NCSA/UIUC (USA), in the framework of the
Inria-UIUC-ANL Joint Lab (JLPC). Damaris now has external users,
including (to our knowledge) visualization specialists from NCSA
and researchers from the France/Brazil Associated research team on
Parallel Computing (joint team between Inria/LIG Grenoble and the
UFRGS in Brazil). Damaris has been successfully integrated into
three large-scale simulations (CM1, OLAM, Nek5000). Works are in
progress to evaluate it in the context of several other
simulations including HACC (cosmology code) and GTC (fusion).</p>
        </li>
        <label>URL:</label>
        <li id="uid40">
          <p noindent="true">
            <ref xlink:href="http://damaris.gforge.inria.fr/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>damaris.<allowbreak/>gforge.<allowbreak/>inria.<allowbreak/>fr/</ref>
          </p>
        </li>
        <label>License:</label>
        <li id="uid41">
          <p noindent="true">GNU Lesser General Public License (LGPL) version 3.</p>
        </li>
        <label>Status:</label>
        <li id="uid42">
          <p noindent="true">This software is available on Inria's
forge. Registration with APP is in progress.</p>
        </li>
      </descriptionlist>
    </subsection>
    <subsection id="uid43" level="1">
      <bodyTitle>Derived software</bodyTitle>
      <p>Derived from BlobSeer, two additional platforms are currently being
developed within KerData: 1) Pyramid, a software service for
array-oriented active storage developed within the framework of the
PhD thesis of Viet-Trung Tran; and 2) BlobSeer-WAN, a data
management service specifically optimized for geographically
distributed environments. It is also developed within the framework
of the PhD thesis of Viet-Trung Tran in relation to the FP3C
project. These platforms have not been publicly released yet.</p>
    </subsection>
  </logiciels>
  <resultats id="uid44">
    <bodyTitle>New Results</bodyTitle>
    <subsection id="uid45" level="1">
      <bodyTitle>Optimizing MapReduce processing</bodyTitle>
      <subsection id="idp140392514142960" level="2">
        <bodyTitle>Hybrid infrastructures</bodyTitle>
        <participants>
          <person key="kerdata-2009-idm140027562000">
            <firstname>Alexandru</firstname>
            <lastname>Costan</lastname>
          </person>
          <person key="kerdata-2012-idm485253310656">
            <firstname>Bharath</firstname>
            <lastname>Vissapragada</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>As Map-Reduce emerges as a leading programming paradigm for
data-intensive computing, today's frameworks which support it still
have substantial shortcomings that limit its potential scalability.
At the core of Map-Reduce frameworks stays a key component with a huge
impact on their performance: the storage layer. To enable scalable
parallel data processing, this layer must meet a series of specific
requirements. An important challenge regards the target execution
infrastructures. While the Map-Reduce programming model has become
very visible in the cloud computing area, it is also subject to active
research efforts on other kinds of large-scale infrastructures, such
as desktop grids. We claim that it is worth investigating how such
efforts (currently done in parallel) could converge, in a context
where large-scale distributed platforms become more and more connected
together.</p>
        <p>In 2012 we investigated several directions where
there is room for such progress: they concern storage efficiency under
massive data access concurrency, scheduling, volatility and
fault-tolerance. We placed our discussion in the perspective of the
current evolution towards an increasing integration of large-scale
distributed platforms (clouds, cloud federations, enterprise desktop
grids, etc.) (<ref xlink:href="#kerdata-2012-bid5" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>). We proposed an approach which aims to overcome the
current limitations of existing Map-Reduce frameworks, in order to
achieve scalable, concurrency-optimized, fault-tolerant Map-Reduce
data processing on hybrid infrastructures. We are designing and
implementing our approach through an original architecture for
scalable data processing: it combines two approaches, BlobSeer and
BitDew, which have shown their benefits separately (on clouds and
desktop grids respectively) into a unified system. The global goal is
to improve the behavior of Map-Reduce-based applications on the target
large-scale infrastructures. The internship of Bharath Vissapragada was dedicated to this topic.</p>
        <p>This approach will be evaluated with real-life bio-informatics
applications on existing Nimbus-powered cloud testbeds interconnected
with desktop grids.</p>
      </subsection>
      <subsection id="idp140392514150704" level="2">
        <bodyTitle>Scheduling: Maestro</bodyTitle>
        <participants>
          <person key="kerdata-2011-idm367107193856">
            <firstname>Shadi</firstname>
            <lastname>Ibrahim</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>As data-intensive applications became popular in the cloud,
data-intensive cloud systems call for empirical evaluations and
technical innovations. We have investigated some performance
limits in current MapReduce frameworks (Hadoop in particular). Our
studies reveal that the current Hadoop's scheduler for map tasks is
inadequate, as it disregards replicas distributions. It causes
performance degradation due to a high number of non-local map tasks,
which in turn causes too many needless speculative map tasks and leads
to imbalanced execution of map tasks among data nodes. We addressed
these problems by developing a new map task scheduler called
Maestro.</p>
        <p>In  <ref xlink:href="#kerdata-2012-bid6" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, we developed a scheduling algorithm
(Maestro) to alleviate the nonlocal map tasks executions problem of
MapReduce. Maestro is conducive to improving the locality of map tasks
executions efficiency by virtue of the finer-grained replica aware
execution of map tasks, thereby having one additional factor for the
chunk hosting status: the expected number of map tasks executions to
be launched. Maestro keeps track of the chunks' locations along with
their replicas' locations and the number of other chunks hosted by
each node. In doing so, Maestro can efficiently schedule the map task
to the node with minimal impacts on other nodes' local map tasks
executions. Maestro schedules the map tasks in two waves: first, it
fills the empty slots of each data node based on the number of hosted
map tasks and on the replication scheme for their input data; second,
runtime scheduling takes into account the probability of scheduling a
map task on a given machine depending on the replicas of the task's
input data. These two waves lead to a higher locality in the execution
of map tasks and to a more balanced intermediate data distribution for
the shuffling phase.</p>
        <p>We evaluated Maestro through a set of experiments on the
Grid'5000  <ref xlink:href="#kerdata-2012-bid7" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> testbed. Preliminary
results <ref xlink:href="#kerdata-2012-bid6" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> show the efficiency and
scalability of our proposals, as well as additional benefits brought
forward by our approach.</p>
      </subsection>
      <subsection id="idp140392514158688" level="2">
        <bodyTitle>Fault tolerance</bodyTitle>
        <participants>
          <person key="kerdata-2011-idm367107170400">
            <firstname>Bunjamin</firstname>
            <lastname>Memishi</lastname>
          </person>
          <person key="kerdata-2011-idm367107193856">
            <firstname>Shadi</firstname>
            <lastname>Ibrahim</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>The simple philosophy of MapReduce has made huge community interest
for its exploration, especially in environments where data-intensive
applications are primary concern. Fault tolerance is one of the key
features of the MapReduce system. MapReduce tasks are re-executed in
case of failure, and a potential failure of a single master causes an
additional bottleneck. It is observed that the detection of the failed
worker tasks in Hadoop have a certain delay, yet not solved. Willing
to improve the applications performance and optimal resource
utilization, both of this concerns were more than a motivation so that
we show in <ref xlink:href="#kerdata-2012-bid8" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> that a little attention has
been devoted to the failure detection in Hadoop's MapReduce which
currently uses a timeout based mechanism for detecting failed tasks.</p>
        <p>We have performed an in-depth analysis of MapReduce's failure detection, and
these preliminary studies have revealed that the current static
timeout value (600 seconds) is not adequate and demonstrate
significant variations in the application's response time with
different timeout value. Moreover, in the presence of single machine
failure, the applications latencies vary not only in accordance to the
occupancy time of the failure, similar to
<ref xlink:href="#kerdata-2012-bid9" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, but also vary with the job
length (short or long).</p>
        <p>Based on our aforementioned micro-analysis of failure detection in
MapReduce, we are currently investigating an adaptive failure
detection mechanism for Hadoop, which basically addresses the timeout
adjustment in real-time for different jobs and applications, so that
finally to adjust this model into a Shared Hadoop Cluster. Another
work should discuss in details different failures types in MapReduce
system and survey the different mechanisms used in MapReduce for
detecting, handling and recovering from these failures and their
inherited pros and cons; additionally, to a particular interest will
be the analyzing of different execution environments including
Cluster, Cloud and Desktop Grid on the efficiency of fault-tolerance
in MapReduce. This work will soon be published.</p>
      </subsection>
    </subsection>
    <subsection id="uid46" level="1">
      <bodyTitle>A-Brain and TomusBlobs</bodyTitle>
      <subsection id="idp140392514167664" level="2">
        <bodyTitle>TomusBlobs</bodyTitle>
        <participants>
          <person key="kerdata-2011-idm367107176512">
            <firstname>Radu</firstname>
            <lastname>Tudoran</lastname>
          </person>
          <person key="kerdata-2009-idm140027562000">
            <firstname>Alexandru</firstname>
            <lastname>Costan</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>Enabling high-throughput massive data processing on cloud data becomes
a critical issue, as it impacts the overall application
performance. In the framework of the MSR-Inria A-Brain co-led by Gabriel Antoniu (KerData) and Bertrand Thirion (PARIETAL), the TomusBlobs<ref xlink:href="#kerdata-2012-bid10" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> system was
designed and implemented by KerData to address such challenges at the level of the cloud
storage. The system we introduce is a concurrency-optimized data
storage system which federates the virtual disks associated to VMs. As
TomusBlobs does not require modifications to the cloud middleware, it
can serve as a high-throughput globally-shared data storage for the
cloud applications that require data passing among computation nodes.</p>
        <p>We leveraged the performance of this solution to enable efficient
data-intensive processing on commercial clouds by building an
optimized prototype MapReduce framework for Azure. The system,
deployed on 350 cores in Azure, was used to execute a real-life
application, A-Brain with the goal of searching for significant
associations between brain locations and genes.</p>
        <p>The achieved throughput increased with an order of 2 for reading,
respectively 3 for writing compared to the remote storage. With our
approach for MapReduce data processing, the computation time is
reduced to 50 % compared to the existing solutions, while the cost is
reduced up to 30 %.</p>
      </subsection>
      <subsection id="idp140392514175008" level="2">
        <bodyTitle>Iterative MapReduce</bodyTitle>
        <participants>
          <person key="kerdata-2011-idm367107176512">
            <firstname>Radu</firstname>
            <lastname>Tudoran</lastname>
          </person>
          <person key="kerdata-2009-idm140027562000">
            <firstname>Alexandru</firstname>
            <lastname>Costan</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
          <person key="algorille-2007-idm186081019872">
            <firstname>Louis-Claude</firstname>
            <lastname>Canon</lastname>
          </person>
        </participants>
        <p>While MapReduce has arisen as a major programming model for data
analysis on clouds, there are many scientific applications that
require processing patterns different from this paradigm. As such,
reduce-intensive algorithms are becoming increasingly useful in
applications such as data clustering, classification and mining. These
algorithms have a common pattern: data are processed iteratively and
aggregated into a single final result. While in the initial MapReduce
proposal the reduce phase was a simple aggregation function, recently
an increasing number of applications relying on MapReduce exhibit a
reduce-intensive pattern, that is, an important part of the
computations are done during the reduce phase. However, platforms like
MapReduce or Dryad lack built-in support for reduce-intensive
workloads.</p>
        <p>To overcome these issues, we introduced
MapIterativeReduce <ref xlink:href="#kerdata-2012-bid11" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, a framework which: 1)
extends the MapReduce programming model to better support
reduce-intensive applications by exploiting the inherent parallelism
of the reduce tasks which have an associative and/or commutative
operation; and 2) substantially improves their efficiency by
eliminating the implicit barrier between the Map and the Reduce
phase. We showed how to leverage this architecture for scientific
applications by enhancing the fault tolerance support in Azure and
TomusBlobs, the underlying storage system, with a light checkpointing
scheme and without any centralized control.</p>
        <p>We evaluated MapIterativeReduce on the Microsoft Azure cloud with
synthetic benchmarks and with a real-life application. Compared to
state-of-art solutions, our approach enables faster data processing,
by reducing the execution times by up to 75 %.</p>
      </subsection>
      <subsection id="idp140392514182864" level="2">
        <bodyTitle>Adaptive file management for clouds</bodyTitle>
        <participants>
          <person key="kerdata-2011-idm367107176512">
            <firstname>Radu</firstname>
            <lastname>Tudoran</lastname>
          </person>
          <person key="kerdata-2009-idm140027562000">
            <firstname>Alexandru</firstname>
            <lastname>Costan</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>Recently, there is an increasing interest to execute general data
processing schemas in clouds, as it would allow many scientific
applications to migrate to this computing infrastructures. The natural
way to do this is to designe and adopt Workflow Processing engines
built for clouds. Such workflow processing in clouds would involve
data propagation on the computation nodes based on well defined data
access patterns. Having an efficient file management backend for a
workflow engines is thus essential as we move to the world of BigData.</p>
        <p>We proposed a new approach for a transfer-optimized file management in
clouds On the one hand, our solution manages files within the
deployment leveraging data locality. On the other hand, we envision an
adaptive system that adopts the transfer method most suited based on
the data transfer context.</p>
        <p>The performance evaluation showed significant gains in terms of
transfer throughput and computation time. File transfer times are
reduced up to a factor of 5 with respect to the remote storage, while
the timespan of running applications is reduced by more than 25%
compared with other frameworks like Hadoop on Azure. This work was
done in the context of a 3-month internship of Radu Tudoran hosted by the Advance Technology
Lab from Microsoft Europe, Germany, Aachen.</p>
      </subsection>
    </subsection>
    <subsection id="uid47" level="1">
      <bodyTitle>Autonomic Cloud data storage management</bodyTitle>
      <participants>
        <person key="paris-2006-idm124332495696">
          <firstname>Gabriel</firstname>
          <lastname>Antoniu</lastname>
        </person>
        <person key="kerdata-2009-idm140027562000">
          <firstname>Alexandru</firstname>
          <lastname>Costan</lastname>
        </person>
      </participants>
      <p>Providing the users with the possibility to store and process data on
externalized, virtual resources from the cloud requires simultaneously
investigating important aspects related to security, efficiency and
quality of service. To this purpose, it clearly becomes necessary to
create mechanisms able to provide feedback about the state of the
storage system along with the underlying physical infrastructure. This
information thus monitored, can further be fed back into the storage
system and used by self-managing engines, in order to enable an
autonomic behavior, possibly with several goals such as
self-configuration, self-optimization, or self-healing. Within the DataCloud@work
Associate Team in partnership with Politehnica University of Bucharest, our goal was
to bring substantial contributions in this direction by leveraging
previous efforts materialized through the BlobSeer data-sharing
platform and several large-scale applications.</p>
      <subsection id="uid48" level="2">
        <bodyTitle>Evaluating BlobSeer for sharing application data on IaaS cloud infrastructures</bodyTitle>
        <p>.
We showed how several types of large scale applications
(e.g. scientific data aggregation, context-aware data management,
video and image processing) rely on BlobSeer's support for high
concurrency and increased data access throughput in order to achieve
their goals. Several building blocks were implemented to address all
the applications' requirements (new meta-data management, extended
clients). An illustrative class of applications is represented by the
context-aware ones. Our goal was to provide a cloud-based storage
layer for sensitive context data, collected from a vast amount of
sources: from smartphones to sensors located in the environment. We
developed a layer on top of BlobSeer to allow two major things:
efficient access to data based on meta-information (a catalogue of
context data), and the support from mobility in the form of
distributed caches able to support the movement of people and give
support for fast access to real-time event of interest (dissemination
of events of interest). The system as a whole was evaluated in
extensive experiments, involving thousands of simulated clients, and
the results proved its valuable contribution to advance the current
state-of-the-art in the area of interested (middlewares to support
context-aware apps).</p>
      </subsection>
      <subsection id="uid49" level="2">
        <bodyTitle>Fault-tolerant VM management in Clouds, using BlobSeer</bodyTitle>
        <p>.
We were also concerned about the fault tolerance support for the aforementioned
applications on the cloud. A first step towards this goal consisted in
exploring ways to deploy, boot and terminate VMs very quickly,
enabling cloud users to exploit elasticity to find the optimal
trade-off between the computational needs (number of resources, usage
time) and budget constraints. We built a VM management system based on
the FUSE interface leveraging the high throughput under increased
concurrency of BlobSeer. We integrated it within the Nimbus cloud to
allow fast VM deployment / snapshotting/ live migration. An adaptive
prefetching mechanism is used to reduce the time required to
simultaneously boot a large number of VM instances on clouds from the
same initial VM image (multi-deployment). This proposal does not
require any foreknowledge of the exact access pattern. It dynamically
adapts to it at run time, enabling the slower instances to learn from
the experience of the faster ones. Since all booting instances
typically access only a small part of the virtual image along almost
the same pattern, the required data can be pre-fetched in the
background. In parallel, we investigated ways to ensure the anonimity
of the data management layer, a requirement for HPC applications
deployed into the clouds.</p>
      </subsection>
    </subsection>
    <subsection id="uid50" level="1">
      <bodyTitle>Advanced techniques for scalable cloud storage</bodyTitle>
      <subsection id="idp140392514195504" level="2">
        <bodyTitle>Adaptive consistency</bodyTitle>
        <participants>
          <person key="kerdata-2010-idm58934898256">
            <firstname>Houssem-Eddine</firstname>
            <lastname>Chihoub</lastname>
          </person>
          <person key="kerdata-2011-idm367107193856">
            <firstname>Shadi</firstname>
            <lastname>Ibrahim</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>In just a few years cloud computing has become a very popular paradigm
and a business success story, with storage being one of the key
features. To achieve high data availability, cloud storage services
rely on replication. In this context, one major challenge is data
consistency. In contrast to traditional approaches that are mostly
based on strong consistency, many cloud storage services opt for
weaker consistency models in order to achieve better availability and
performance. This comes at the cost of a high probability of stale
data being read, as the replicas involved in the reads may not always
have the most recent write. In <ref xlink:href="#kerdata-2012-bid12" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, we propose
a novel approach, named Harmony, which adaptively tunes the
consistency level at run-time according to the application
requirements. The key idea behind Harmony is an intelligent estimation
model of stale reads, allowing to elastically scale up or down the
number of replicas involved in read operations to maintain a low
(possibly zero) tolerable fraction of stale reads. As a result,
Harmony can meet the desired consistency of the applications while
achieving good performance. We have implemented Harmony and performed
extensive evaluations with the Cassandra cloud storage on Grid'5000
testbed and on Amazon EC2. The results show that Harmony can achieve
good performance without exceeding the tolerated number of stale
reads. For instance, in contrast to the static eventual consistency
used in Cassandra, Harmony reduces the stale data being read by almost
80%. Meanwhile, it improves the throughput of the system by 45%
while maintaining the desired consistency requirements of the
applications when compared to the strong consistency model in
Cassandra.</p>
        <p>While most optimizations efforts for consistency management in the
cloud focus on how to provide adequate trade-offs between consistency
guarantees and performance, a little work has been investigating the
impact of consistency on monetary cost. However, and since strict
strong consistency is not always required for large class of
applications, in <ref xlink:href="#kerdata-2012-bid13" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> we argue that monetary
cost should be taken into consideration when evaluating or selecting a
consistency level in the cloud. Accordingly, we define a new metric
called consistency-cost efficiency. Based on this metric, we present a
simple, yet efficient economical consistency model, called Bismar,
that adaptively tunes the consistency level at run-time in order to
reduce the monetary cost while simultaneously maintaining a low
fraction of stale reads. Experimental evaluations with the Cassandra
cloud storage on a Grid'5000 testbed show the validity of the metric
and demonstrate the effectiveness of the proposed consistency model
allowing up to 31 % of money saving while tolerating a very small
fraction of stale reads.</p>
      </subsection>
      <subsection id="idp140392514203088" level="2">
        <bodyTitle>In-memory data management</bodyTitle>
        <participants>
          <person key="kerdata-2009-idm140027572976">
            <firstname>Viet-Trung</firstname>
            <lastname>Tran</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
          <person key="paris-2006-idm124332467968">
            <firstname>Luc</firstname>
            <lastname>Bougé</lastname>
          </person>
        </participants>
        <p>As a result of continuous innovation in hardware technology, computers
are made more and more powerful than their prior models. Modern
servers nowadays can possess large main memory capability that can
size up to 1 Terabytes (TB) and more. As memory accesses are at least
100 times faster than disk, keeping data in main memory becomes an
interesting design principle to increase the performance of data
management systems. We design DStore <ref xlink:href="#kerdata-2012-bid14" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, a
document-oriented store residing in main memory to fully exploit
high-speed memory accesses for high performance. DStore is able to
scale up by increasing memory capability and the number of CPU-cores
rather than scaling horizontally as in distributed data-management
systems. This design decision favors DStore in supporting fast and
atomic complex transactions, while maintaining high throughput for
analytical processing (read-only accesses). This goal is (to our best
knowledge) not easy to achieve with high performance in distributed
environments.</p>
        <p>To achieve its goals, DStore is built with several design
principles. DStore follows a single threaded execution model to
execute update transactions sequentially by one <i>master thread</i>
while relying on a versioning concurrency control to enable multiple
<i>reader threads</i> running simultaneously. DStore builds indexes
for fast document lookups. Those indexes are built using the
<i>delta-indexing</i> and <i>bulk updating</i> mechanisms for faster
indexes maintenance and for atomicity guarantees of complex
queries. Moreover, DStore is designed to favor stale reads that only
need to access isolated snapshots of the indexes. Thus, it can
eliminate interference between transactional processing and analytical
processing.</p>
        <p>We conducted multiple synthetic benchmarks on the Grid'5000 to
evaluate the DStore prototype. Our preliminary results demonstrated
that DStore achieved high performance even in scenarios where
<i>Read</i>, <i>Insert</i> and <i>Delete</i> queries were performed
simultaneously. In fact, the processing rate measured was about
600,000 operations per second for each concurrent process.</p>
      </subsection>
      <subsection id="idp140392514214144" level="2">
        <bodyTitle>Scalable geographically distributed storage systems</bodyTitle>
        <participants>
          <person key="kerdata-2009-idm140027572976">
            <firstname>Viet-Trung</firstname>
            <lastname>Tran</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
          <person key="paris-2006-idm124332467968">
            <firstname>Luc</firstname>
            <lastname>Bougé</lastname>
          </person>
        </participants>
        <p>To build a globally scalable distributed file system that spreads over
a wide area network (WAN), we propose an integrated architecture for a
storage system relying on a distributed metadata-management system and
BlobSeer, a large-scale data-management service. Since BlobSeer was
initially designed to run on cluster environments, it is necessary to
extend BlobSeer in order to take into account the latency hierarchy on
geographically distributed environments.</p>
        <p>We proposed BlobSeer-WAN, an extension of BlobSeer optimized for
geographically distributed environments. First, in order to keep
metadata I/O local to each site as much as possible, we proposed an
asynchronous metadata replication scheme at the level of metadata
providers. As metadata replication is asynchronous, we guarantee a
minimal impact on the writing clients that generate metadata. Second,
we introduced a distributed version management in BlobSeer-WAN by
leveraging an implementation of multiple version managers and using
vector clocks for detection and resolution of collision. This
extension to BlobSeer keeps BLOBs consistent while they are globally
shared among distributed sites under high concurrency.</p>
        <p>Several experiments were performed on the Grid'5000 testbed
demonstrated that BlobSeer-WAN can offer scalable aggregated
throughput when concurrent clients append to one BLOB. The aggregated
throughput reached to 1400 MB/s for 20 concurrent clients. We also
compared BlobSeer-WAN and the original BlobSeer in local site
accesses. The experiments shown that the overhead of the multiple
version managers implementation and the metadata replication scheme in
BlobSeer-WAN is minimal, thanks to our asynchronous replication
scheme.</p>
      </subsection>
    </subsection>
    <subsection id="uid51" level="1">
      <bodyTitle>Scalable I/O for HPC</bodyTitle>
      <subsection id="idp140392514220368" level="2">
        <bodyTitle>Damaris and HPC visualization</bodyTitle>
        <participants>
          <person key="kerdata-2009-idm140027550560">
            <firstname>Matthieu</firstname>
            <lastname>Dorier</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>In the context of the Joint Inria/UIUC/ANL Laboratory for Petascale
computing (JLCP), have proposed the Damaris approach to enable
efficient I/O, data analysis and visualization at ver large scale from
SMP machines. The I/O bottlenecks already present on current
petascale systems as well as the amount of data written by HPC
applications force to consider new approaches to get insights from
running simulations. Trying to bypass the storage or drastically
reducing the amount of data generated will be of outmost importance
for exascale. In-situ visualization has therefor been proposed to run
analysis and visualization tasks closer to the simulation, as it runs.</p>
        <p>The first results obtained with Damaris in achieving scalable,
jitter-free I/O, were published this year <ref xlink:href="#kerdata-2012-bid15" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.
In order to achieve efficient in-situ visualization at extreme scale,
we investigated the limitations of existing in-situ visualization
software and proposed to fill the gaps of these software by providing
in-situ visualization support to Damaris. The use of Damaris on top
of existing visualization packages allows us to:</p>
        <simplelist>
          <li id="uid52">
            <p noindent="true">Reduce code instrumentation to a minimum in existing
simulations,</p>
          </li>
          <li id="uid53">
            <p noindent="true">Gather the capabilities of several visualization tools to offer
adaptability under a unified data management interface,</p>
          </li>
          <li id="uid54">
            <p noindent="true">Use dedicated cores to hide the run time impact of in-situ
visualization and</p>
          </li>
          <li id="uid55">
            <p noindent="true">Efficiently use memory through a shared-memory-based
communication model.</p>
          </li>
        </simplelist>
        <p>Experiments are now being conducted on BlueWaters (Cray XK6 at NCSA),
Intrepid (BlueGene/P at ANL) and Grid5000 with representative
visualization scenarios for the CM1  <ref xlink:href="#kerdata-2012-bid16" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> atmospheric
simulation and the Nek5000  <ref xlink:href="#kerdata-2012-bid17" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> CFD solver.</p>
        <p>Results will be submitted to a conference in early 2013. We plan to
further investigate the role that Damaris can take in performing
efficient and self-adaptive data analysis in HPC simulations.</p>
      </subsection>
      <subsection id="idp140392514235168" level="2">
        <bodyTitle>Advanced I/O and Storage</bodyTitle>
        <participants>
          <person key="kerdata-2009-idm140027550560">
            <firstname>Matthieu</firstname>
            <lastname>Dorier</lastname>
          </person>
          <person key="kerdata-2009-idm140027562000">
            <firstname>Alexandru</firstname>
            <lastname>Costan</lastname>
          </person>
          <person key="paris-2006-idm124332495696">
            <firstname>Gabriel</firstname>
            <lastname>Antoniu</lastname>
          </person>
        </participants>
        <p>The recent extension of the JLPC to Argonne National Lab (ANL) has
opened new research directions in the field of advanced I/O and
storage for HPC, in collaboration with Robert Ross's team at ANL's
Mathematics and Computer Science Division (MCS). A founding from the
FACCTS program (France And Chicago CollaboraTing in Science) allowed
multiple visits (see Section <ref xlink:href="#uid89" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>) of students and researchers
from both sides to initiate this new collaboration and explore
potential research directions.</p>
        <p>One outcome of these visits has been the adaptation of Damaris to work
on BlueGene/P and BlueGene/Q machines installed at ANL. Several
exchanges led to the design of new I/O scheduling algorithms
leveraging Damaris for efficient asynchronous I/O and storage. These
algorithms are currently being evaluated, and expected to be published
in early 2013.</p>
        <p>During these exchanges we also investigated new storage architectures
for Exascale systems leveraging BLOB-based large-scale storage able to
cope with complex data models. We will explore how we can combine the
benefits of the approaches to Big Data storage currently developed by
the partners: the BlobSeer approach (KerData), which provides support
for multi- versioning and efficient fine-grain access to huge data
under heavy concurrency and the Triton approach (ANL), which
introduces new object storage semantics. The final goal of the
resulting architecture will be to propose efficient solutions to
data-related bottlenecks in Exascale HPC systems.</p>
      </subsection>
    </subsection>
  </resultats>
  <contrats id="uid56">
    <bodyTitle>Bilateral Contracts and Grants with Industry</bodyTitle>
    <subsection id="uid57" level="1">
      <bodyTitle>Bilateral Contracts with Industry</bodyTitle>
      <descriptionlist>
        <label>Microsoft: A-Brain (2010–2013).</label>
        <li id="uid58">
          <p noindent="true">In the framework of the
Joint Inria-Microsoft Research Center. See details in
Section <ref xlink:href="#uid27" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. To support this project,
Microsoft provides 2 million computation hours on the Azure
platform and 10 TB of storage per year. The project is
funding Louis-Claude Canon as a postdoc fellow (18 months
since September 2011) and to complete the PhD MESR grant of
Radu Tudoran (<i>Mission complémentaire d'expertise</i>, 3 years,
started in October 2011).</p>
        </li>
        <label>IBM: MapReduce ANR Project (2010–2014).</label>
        <li id="uid59">
          <p noindent="true">IBM is a partner of the MapReduce
ANR Project: see Section <ref xlink:href="#uid61" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
        </li>
      </descriptionlist>
    </subsection>
  </contrats>
  <partenariat id="uid60">
    <bodyTitle>Partnerships and Cooperations</bodyTitle>
    <subsection id="uid61" level="1">
      <bodyTitle>National
Initiatives</bodyTitle>
      <subsection id="uid62" level="2">
        <bodyTitle>ANR</bodyTitle>
        <descriptionlist>
          <label>MapReduce (2010–2014).</label>
          <li id="uid63">
            <p noindent="true">An ANR project (ARPEGE 2010) with
international partners on optimized Map-Reduce data processing on
cloud platforms. This project started in October 2010 in
collaboration with Argonne National Lab, the University of Illinois
at Urbana Champaign, the UIUC/Inria Joint Lab on Petascale
Computing, IBM, IBCP, MEDIT and the GRAAL Inria Project-Team. URL:
<ref xlink:href="http://mapreduce.inria.fr/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>mapreduce.<allowbreak/>inria.<allowbreak/>fr/</ref></p>
          </li>
        </descriptionlist>
      </subsection>
      <subsection id="uid64" level="2">
        <bodyTitle>Other National projects</bodyTitle>
        <descriptionlist>
          <label>HEMERA (2010–2014).</label>
          <li id="uid65">
            <p noindent="true">An Inria Large Wingspan Project, started
in 2010. Within Hemera, G. Antoniu (KerData Inria Team) and Gilles
Fedak (GRAAL Inria Project-Team) co-lead the Map-Reduce scientific
challenge. KerData also co-initiated a working group called
“Efficient management of very large volumes of information for
data-intensive applications”, co-led by G. Antoniu and Jean-Marc
Pierson (IRIT, Toulouse).</p>
          </li>
          <label>Grid'5000.</label>
          <li id="uid66">
            <p noindent="true">We are members of the Grid'5000 community: we make
experiments on the Grid'5000 platform on an everyday basis.</p>
          </li>
        </descriptionlist>
      </subsection>
    </subsection>
    <subsection id="uid67" level="1">
      <bodyTitle>European Initiatives</bodyTitle>
      <subsection id="uid68" level="2">
        <bodyTitle>FP7 Projects</bodyTitle>
        <descriptionlist>
          <label>The SCALUS FP7 Marie Curie Initial Training Network</label>
          <li id="uid69">
            <p noindent="true">(2009–2013). Partners: Universidad Politécnica de Madrid (UPM),
Barcelona Supercomputing Center, University of Paderborn,
Ruprecht-Karls-Universität Heidelberg, Durham University, FORTH,
École des Mines de Nantes, XLAB, CERN, NEC, Microsoft Research,
Fujitsu, Sun Microsystems. Topic: scalable distributed storage. We
mainly collaborate with UPM (2 co-advised PhD theses).</p>
          </li>
        </descriptionlist>
      </subsection>
      <subsection id="uid70" level="2">
        <bodyTitle>Collaborations in European Programs, except FP7</bodyTitle>
        <sanspuceslist>
          <li id="uid71">
            <p noindent="true">CoreGRID ERCIM Working Group, since 2009. The CoreGRID Symposium held in Las Palmas de Gran
Canaria, Spain, 25-26 August 2008 marked the end of the
ERCIM-managed CoreGRID Network of Excellence funded by the
European Commission. There, it was decided to re-launch
CoreGRID as a self-sustained ERCIM Working Group covering
research activities on both Grid and Service Computing while
maintaining the momentum of the European collaboration on
Grid research.</p>
          </li>
        </sanspuceslist>
      </subsection>
    </subsection>
    <subsection id="uid72" level="1">
      <bodyTitle>International
Initiatives</bodyTitle>
      <subsection id="uid73" level="2">
        <bodyTitle>Inria Associate Teams</bodyTitle>
        <subsection id="uid74" level="3">
          <bodyTitle>
            <ref xlink:href="http://www.irisa.fr/kerdata/doku.php?id=cloud_at_work:start" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">DATACLOUD</ref>
          </bodyTitle>
          <sanspuceslist>
            <li id="uid75">
              <p noindent="true">Title: Distributed data management for cloud services</p>
            </li>
            <li id="uid76">
              <p noindent="true">Inria principal investigator: Gabriel Antoniu</p>
            </li>
            <li id="uid77">
              <p noindent="true">International Partner (Institution - Laboratory - Researcher):</p>
              <sanspuceslist>
                <li id="uid78">
                  <p noindent="true">Politechnica University of Bucharest (Romania) - NCIT -
Valentin Cristea</p>
                </li>
              </sanspuceslist>
            </li>
            <li id="uid79">
              <p noindent="true">Duration: 2010 - 2012</p>
            </li>
            <li id="uid80">
              <p noindent="true">See also:
<ref xlink:href="http://www.irisa.fr/kerdata/doku.php?id=cloud_at_work:start" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>www.<allowbreak/>irisa.<allowbreak/>fr/<allowbreak/>kerdata/<allowbreak/>doku.<allowbreak/>php?id=cloud_at_work:start</ref></p>
            </li>
            <li id="uid81">
              <p noindent="true">Our research topics address the area of distributed data
management for cloud services. We aim at investigating several
open issues related to autonomic storage in the context of cloud
services. The goal is explore how to build an efficient, secure
and reliable storage IaaS for data-intensive distributed
applications running in cloud environments by enabling an
autonomic behavior, while leveraging the advantages of the grid
operating system approach.</p>
              <p>Our research activities involve the design and implementation of
experimental prototypes based on the following software platforms:</p>
              <sanspuceslist>
                <li id="uid82">
                  <p noindent="true">The BlobSeer data-sharing platform (designed by the KerData
Team)</p>
                </li>
                <li id="uid83">
                  <p noindent="true">The XtreemOS grid operation system (designed under the
leadership of the Myriads Team)</p>
                </li>
                <li id="uid84">
                  <p noindent="true">The MonALISA monitoring framework (using the expertise of the
PUB Team).</p>
                </li>
              </sanspuceslist>
            </li>
          </sanspuceslist>
          <p>The main results obtained in 2012 are described in Section <ref xlink:href="#uid50" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
        </subsection>
      </subsection>
      <subsection id="uid85" level="2">
        <bodyTitle>Inria International Partners</bodyTitle>
        <p>Politehnica University of Bucharest</p>
      </subsection>
      <subsection id="uid86" level="2">
        <bodyTitle>Participation In International Programs</bodyTitle>
        <descriptionlist>
          <label>Joint Inria-UIUC Lab for Petascale Computing (JLPC),</label>
          <li id="uid87">
            <p noindent="true">since
2009. Collaboration on concurrency-optimized I/O for post-Petascale
platforms (see details inw
Section <ref xlink:href="#uid27" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>). A joint project
proposal with the team of Rob Ross (Argonne National Lab) has been
accepted in 2012 at the FACCTS call for projects. It served to prepare the preparation of a project for an Associate Team with ANL and UIUC. The project, called Data@Exascale has been accepted for 2013-2015.</p>
          </li>
        </descriptionlist>
        <descriptionlist>
          <label>FP3C ANR-JST project (2010–2014).</label>
          <li id="uid88">
            <p noindent="true">This project co-funded by ANR and by JST (Japan Science and Technology Agency) started in October 2010 for 42 months. It focuses on programming issues for Post-Petascale architectures. In this framework, KerData collaborates with the University of Tsukuba on data management issues.</p>
          </li>
        </descriptionlist>
      </subsection>
    </subsection>
    <subsection id="uid89" level="1">
      <bodyTitle>International
Research Visitors</bodyTitle>
      <subsection id="uid90" level="2">
        <bodyTitle>Visits of International Scientists</bodyTitle>
        <simplelist>
          <li id="uid91">
            <p noindent="true">Robert Ross and Dried Kimpe (Argonne National Lab) visited the
KerData team for a week (June 2012) within the framework of our
FACCTS project.</p>
          </li>
          <li id="uid92">
            <p noindent="true">Florin Pop and Ciprian Dobre (Politehnica University of
Bucharest) visited the KerData team for a week (June 2012) within
the framework of our DataCloud@work Associate Team.</p>
          </li>
        </simplelist>
      </subsection>
      <subsection id="uid93" level="2">
        <bodyTitle>Internships</bodyTitle>
        <sanspuceslist>
          <li id="uid94">
            <p noindent="true">Elena Burceanu (from February 2012 until June 2012)</p>
            <sanspuceslist>
              <li id="uid95">
                <p noindent="true">Subject: Distributed data storage for context-aware
applications</p>
              </li>
              <li id="uid96">
                <p noindent="true">Institution: Politehnica University of Bucharest (Romania)</p>
              </li>
            </sanspuceslist>
          </li>
        </sanspuceslist>
        <sanspuceslist>
          <li id="uid97">
            <p noindent="true">Vlad Nicolae Serbanescu (from February 2012 until June 2012)</p>
            <sanspuceslist>
              <li id="uid98">
                <p noindent="true">Subject: Distributed data aggregation using the BlobSeer
cloud storage service</p>
              </li>
              <li id="uid99">
                <p noindent="true">Institution: Politehnica University of Bucharest (Romania)</p>
              </li>
            </sanspuceslist>
          </li>
        </sanspuceslist>
        <sanspuceslist>
          <li id="uid100">
            <p noindent="true">Bharath Vissapragada (from February 2012 until June 2012)</p>
            <sanspuceslist>
              <li id="uid101">
                <p noindent="true">Subject: MapReduce data processing on hybrid (cloud/desktop
grid) infrastructures</p>
              </li>
              <li id="uid102">
                <p noindent="true">Institution: University of Hyderabad (India)</p>
              </li>
            </sanspuceslist>
          </li>
        </sanspuceslist>
        <sanspuceslist>
          <li id="uid103">
            <p noindent="true">Mauricio De Oliveira de Diana (June 2012)</p>
            <sanspuceslist>
              <li id="uid104">
                <p noindent="true">Subject: Performance modeling for the BlobSeer storage system</p>
              </li>
              <li id="uid105">
                <p noindent="true">Institution: Master student from Brazil</p>
              </li>
            </sanspuceslist>
          </li>
        </sanspuceslist>
        <sanspuceslist>
          <li id="uid106">
            <p noindent="true">Sergiu Vicol (June–August 2012)</p>
            <sanspuceslist>
              <li id="uid107">
                <p noindent="true">Subject: Optimizing memory management in Damaris</p>
              </li>
              <li id="uid108">
                <p noindent="true">Institution: Bachelor student from Oxford University. Former awardee of the ENS-Inria
Excellence Award for the Laureates of the Romanian Olympiad in
Informatics.</p>
              </li>
            </sanspuceslist>
          </li>
        </sanspuceslist>
        <sanspuceslist>
          <li id="uid109">
            <p noindent="true">Alexandru Farcasanu(June–August 2012)</p>
            <sanspuceslist>
              <li id="uid110">
                <p noindent="true">Subject: Optimizing the DStore in-memory storage system</p>
              </li>
              <li id="uid111">
                <p noindent="true">Institution: Bachelor students from Politehnica University of Bucharest. Former awardee of the ENS-Inria
Excellence Award for the Laureates of the Romanian Olympiad in
Informatics.</p>
              </li>
            </sanspuceslist>
          </li>
        </sanspuceslist>
      </subsection>
      <subsection id="uid112" level="2">
        <bodyTitle>Visits to International Teams</bodyTitle>
        <simplelist>
          <li id="uid113">
            <p noindent="true">Viet-Trung Tran visited Microsoft Research Cambridge (Dushyanth
Narayanan) for a 3-month internship, funded by MSR.</p>
          </li>
          <li id="uid114">
            <p noindent="true">Houssem-Eddine Chihoub visited the Polytechnical University of
Madrid (Maria Perez) for 3 months, funded by the FP7 SCALUS MCITN
project.</p>
          </li>
          <li id="uid115">
            <p noindent="true">Radu Tudoran visited the ATL Lab at European Microsoft
Innovation Center (Aaachen Germany) for 3 months, funded by
Microsoft.</p>
          </li>
          <li id="uid116">
            <p noindent="true">Matthieu Dorier visited ANL (Rob Ross, Tom Peterka, Phil Carns)
and UIUC (Franck Cappello) for one month, funded by our FACCTS
grant.</p>
          </li>
        </simplelist>
      </subsection>
    </subsection>
  </partenariat>
  <diffusion id="uid117">
    <bodyTitle>Dissemination</bodyTitle>
    <subsection id="uid118" level="1">
      <bodyTitle>Scientific Animation</bodyTitle>
      <p>Gabriel Antoniu:</p>
      <simplelist>
        <li id="uid119">
          <p noindent="true">General Co-Chair of the ScienceCloud 2012 International workshop held in conjunction with the ACM HPDC 2012 conference.</p>
        </li>
        <li id="uid120">
          <p noindent="true">Local Chair of the 7th International Workshop of the Joint Inria-UIUC-ANL Lab for Petascale Computing, Rennes, June 2012.</p>
        </li>
        <li id="uid121">
          <p noindent="true">Track chair at IEEE CloudCom 2012 international conference.</p>
        </li>
        <li id="uid122">
          <p noindent="true">Editor for a Special Issue of Concurrency and Computation: Practice and Experience Journal on Cloud Computing for Data-driven Science and Engineering, 2012.</p>
        </li>
        <li id="uid123">
          <p noindent="true">Program Committee member (selection): ACM HPDC 2012, ACM/IEEE SC 2013, IEEE/ACM CCGRID 2013, ICCCN 2012, IEEE HPCC 2012, IEEE AINA 2012, IEEE CloudCom 2012, ICPADS 2012.</p>
        </li>
        <li id="uid124">
          <p noindent="true">Coordinator for the MapReduce ANR project (see
Section <ref xlink:href="#uid61" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>).</p>
        </li>
        <li id="uid125">
          <p noindent="true">G. Antoniu and B. Thirion
(PARIETAL Project-Team, <span class="smallcap" align="left">Inria Saclay –
Île-de-France</span>) co-lead the
AzureBrain Microsoft-Inria Project (2010-2013).</p>
        </li>
        <li id="uid126">
          <p noindent="true">Coordinator for the DataCloud@work Associate Team, a project
involving the KerData and MYRIADS Inria Teams in Rennes and the
Distributed Systems Group from Politehnica University of
Bucharest (2010–2012).</p>
        </li>
        <li id="uid127">
          <p noindent="true">Local coordinator for Inria Rennes – Bretagne Atlantique Research Center in the SCALUS Project of the
Marie-Curie Initial Training Networks Programme (ITN), call
FP7-PEOPLE-ITN-2008 (2009-2013).</p>
        </li>
      </simplelist>
      <p>Alexandru Costan:</p>
      <simplelist>
        <li id="uid128">
          <p noindent="true">Organizer of the 1st Workshop on Big Data Management in Clouds
BDMC2012 in conjunction with EuroPar 2012, see
<ref xlink:href="http://www.irisa.fr/kerdata/bdmc/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>www.<allowbreak/>irisa.<allowbreak/>fr/<allowbreak/>kerdata/<allowbreak/>bdmc/</ref></p>
        </li>
        <li id="uid129">
          <p noindent="true">Program Committee member: CloudCom 2012, ScienceClouds 2012,
EIDWT 2012</p>
        </li>
        <li id="uid130">
          <p noindent="true">Reviewer: J. of Parallel and Distributed Computing, IEEE
Internet Computing, Concurrency and Computation: Practice and
Experience, Intl. J. of Grid and Utility Computing, Innovative
Studies Intl. J., Scalable Computing: Practice and Experience,
Simulation Modelling Practice and Theory, HPDC 2012, AINA 2012,
RenPar 2012</p>
        </li>
      </simplelist>
      <p>Luc Bougé:</p>
      <simplelist>
        <li id="uid131">
          <p noindent="true">Since September 2012: Scientific coordinator for the Information &amp; Communication Science &amp; Technology Department of the National Research Agency (ANR).</p>
        </li>
      </simplelist>
    </subsection>
    <subsection id="uid132" level="1">
      <bodyTitle>Teaching - Supervision -
Juries</bodyTitle>
      <subsection id="uid133" level="2">
        <bodyTitle>Teaching</bodyTitle>
        <p>Gabriel Antoniu:</p>
        <sanspuceslist>
          <li id="uid134">
            <p noindent="true">Master (Engineering Degree, 5th year): Grid and cloud computing, 18 hours (lectures), M2 level, Ecole Supérieure d'Informatique, Electronique, Automatique, Paris, France.</p>
          </li>
          <li id="uid135">
            <p noindent="true">Master: Grid, P2P and cloud data management, 18 hours (lectures), M2 level, University of Nantes, ALMA Master, Distributed Architectures module, France.</p>
          </li>
          <li id="uid136">
            <p noindent="true">Master: Peer-to-Peer Applications and Systems, 10 hours (lectures), M2 level, ENS Cachan - Brittany, M2RI Master Program, PAP Module, France.</p>
          </li>
        </sanspuceslist>
        <p>Alexandru Costan</p>
        <simplelist>
          <li id="uid137">
            <p noindent="true">Object-oriented programming, 18h, L3, ENS Cachan - Antenne de
Bretagne</p>
          </li>
          <li id="uid138">
            <p noindent="true">Databases, 28h, L2, INSA Rennes, France</p>
          </li>
          <li id="uid139">
            <p noindent="true">Object-oriented design, 28h, M1, INSA de
Rennes</p>
          </li>
          <li id="uid140">
            <p noindent="true">Practical case studies, 16h, L3, INSA de Rennes</p>
          </li>
        </simplelist>
        <p>Matthieu Dorier:</p>
        <simplelist>
          <li id="uid141">
            <p noindent="true">Java programming (lectures, seminars, practical sessions), L1 level, INSA de Rennes, 56h.</p>
          </li>
          <li id="uid142">
            <p noindent="true">Programming techniques (seminars, practical sessions), L3 level, ENS Cachan - Antenne de Bretagne 42h</p>
          </li>
        </simplelist>
      </subsection>
      <subsection id="uid143" level="2">
        <bodyTitle>Supervision</bodyTitle>
        <p>PhD &amp; HdR :</p>
        <sanspuceslist>
          <li id="uid144">
            <p noindent="true">PhD: Viet-Trung Tran, Scalable data-management systems for Big Data, thesis started in October 2009 co-advised by Gabriel Antoniu and Luc Bougé. Date of defense: 21 January 2013.</p>
          </li>
          <li id="uid145">
            <p noindent="true">PhD in progress : Houssem Chihoub, Consistency issues in cloud storage systems, thesis started in October 2010 co-advised by Maria Pérez (UPM - Madrid) and Gabriel Antoniu.</p>
          </li>
          <li id="uid146">
            <p noindent="true">PhD in progress : Matthieu Dorier, Scalable I/O for postpetascale HPC systems, thesis started in October 2011 co-advised by Gabriel Antoniu and Luc Bougé.</p>
          </li>
          <li id="uid147">
            <p noindent="true">PhD in progress : Radu Tudoran, Scalable data sharing for Azure clouds, thesis started in October 2011 co-advised by Gabriel Antoniu and Luc Bougé.</p>
          </li>
        </sanspuceslist>
      </subsection>
      <subsection id="uid148" level="2">
        <bodyTitle>Juries</bodyTitle>
        <sanspuceslist>
          <li id="uid149">
            <p noindent="true">Gabriel Antoniu served as a member of Inria's national Jury for hiring researchers (junior positions, confirmed positions and starting research positions).</p>
          </li>
          <li id="uid150">
            <p noindent="true">Gabriel Antoniu served as a Referee and as a Chair for a PhD Jury at the University of Bordeaux; as a Chair for a PhD Jury at the University of Nantes.</p>
          </li>
        </sanspuceslist>
      </subsection>
      <subsection id="uid151" level="2">
        <bodyTitle>Miscellaneous</bodyTitle>
        <sanspuceslist>
          <li id="uid152">
            <p noindent="true">Gabriel Antoniu serves as a member of Inria's Evaluation Committee.</p>
          </li>
          <li id="uid153">
            <p noindent="true">Luc Bougé serves as Head of the Computer Science Department of ENS Cachan - Brittany.</p>
          </li>
        </sanspuceslist>
      </subsection>
    </subsection>
  </diffusion>
  <biblio id="bibliography" html="bibliography" numero="10" titre="Bibliography">
    <biblStruct id="kerdata-2012-bid32" type="inproceedings" rend="refer" n="refercite:ANTONIU:2007:Inria-00178653:1">
      <identifiant type="hal" value="inria-00178653"/>
      <analytic>
        <title level="a">Performance scalability of the JXTA P2P framework</title>
        <author>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2006-idm124331463632">
            <foreName>Loïc</foreName>
            <surname>Cudennec</surname>
            <initial>L.</initial>
          </persName>
          <persName key="paris-2006-idm124331446512">
            <foreName>Mathieu</foreName>
            <surname>Jan</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Mike</foreName>
            <surname>Duigou</surname>
            <initial>M.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">Proc. IEEE International Parallel and Distributed Processing Symposium (IPDPS 2007)</title>
        <loc>Long Beach, USA</loc>
        <imprint>
          <dateStruct>
            <year>2007</year>
          </dateStruct>
          <biblScope type="pages">108</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00178653/en/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00178653/<allowbreak/>en/</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid31" type="article" rend="refer" n="refercite:ANTONIU:2006:Inria-00000987:2">
      <identifiant type="hal" value="inria-00000987"/>
      <analytic>
        <title level="a">How to bring together fault tolerance and data consistency to enable grid data sharing</title>
        <author>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="caps-2006-idm215852169472">
            <foreName>Jean-François</foreName>
            <surname>Deverge</surname>
            <initial>J.-F.</initial>
          </persName>
          <persName key="paris-2006-idm124331438528">
            <foreName>Sébastien</foreName>
            <surname>Monnet</surname>
            <initial>S.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-editorial-board="yes" x-international-audience="yes">
        <title level="j">Concurrency and Computation: Practice and Experience</title>
        <imprint>
          <biblScope type="number">17</biblScope>
          <dateStruct>
            <year>2006</year>
          </dateStruct>
          <biblScope type="pages">1-19</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00000987/en/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00000987/<allowbreak/>en/</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid34" type="article" rend="refer" n="refercite:costan:hal-00767034">
      <identifiant type="hal" value="hal-00767034"/>
      <analytic>
        <title level="a">TomusBlobs: Scalable Data-intensive Processing on Azure Clouds</title>
        <author>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="kerdata-2011-idm367107176512">
            <foreName>Radu</foreName>
            <surname>Tudoran</surname>
            <initial>R.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Goetz</foreName>
            <surname>Brasche</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Concurrency and Computation Practice and Experience</title>
        <imprint>
          <dateStruct>
            <year>2013</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00767034" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00767034</ref>
        </imprint>
      </monogr>
      <note type="bnote">To appear</note>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid36" type="inproceedings" rend="refer" n="refercite:dorier:hal-00715252">
      <identifiant type="hal" value="hal-00715252"/>
      <analytic>
        <title level="a">Damaris: How to Efficiently Leverage Multicore Parallelism to Achieve Scalable, Jitter-free I/O</title>
        <author>
          <persName key="kerdata-2009-idm140027550560">
            <foreName>Matthieu</foreName>
            <surname>Dorier</surname>
            <initial>M.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610725168">
            <foreName>Franck</foreName>
            <surname>Cappello</surname>
            <initial>F.</initial>
          </persName>
          <persName>
            <foreName>Marc</foreName>
            <surname>Snir</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Leigh</foreName>
            <surname>Orf</surname>
            <initial>L.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="m">CLUSTER - IEEE International Conference on Cluster Computing</title>
        <loc>Beijing, China</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>September</month>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00715252" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00715252</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid28" type="article" rend="refer" n="refercite:nicolae:2010:inria-00511414:1">
      <analytic>
        <title level="a">BlobSeer: Next Generation Data Management for Large Scale Infrastructures</title>
        <author>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
          <persName key="paris-2008-idm188504272032">
            <foreName>Diana</foreName>
            <surname>Moise</surname>
            <initial>D.</initial>
          </persName>
          <persName key="paris-2008-idm188504275952">
            <foreName>Alexandra</foreName>
            <surname>Carpen-Amarie</surname>
            <initial>A.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-editorial-board="yes" x-international-audience="yes">
        <title level="j">Journal of Parallel and Distributed Computing</title>
        <imprint>
          <biblScope type="volume">71</biblScope>
          <biblScope type="number">2</biblScope>
          <dateStruct>
            <month>February</month>
            <year>2011</year>
          </dateStruct>
          <biblScope type="pages">169-184</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00511414/en/" type="hal" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00511414/<allowbreak/>en/<allowbreak/></ref>
        </imprint>
      </monogr>
      <note type="bnote">Special issue on data intensive computing. To appear</note>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid27" type="inproceedings" rend="refer" n="refercite:nicolae:2011:inria-00570682:1">
      <identifiant type="hal" value="inria-00570682"/>
      <analytic>
        <title level="a">Going Back and Forth: Efficient Multi-Deployment and Multi-Snapshotting on Clouds</title>
        <author>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName>
            <foreName>John</foreName>
            <surname>Bresnahan</surname>
            <initial>J.</initial>
          </persName>
          <persName>
            <foreName>Kate</foreName>
            <surname>Keahey</surname>
            <initial>K.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">The 20th International ACM Symposium on High-Performance Parallel and Distributed Computing (HPDC 2011)</title>
        <loc>San José, CA, United States</loc>
        <imprint>
          <dateStruct>
            <month>June</month>
            <year>2011</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/inria-00570682/en" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00570682/<allowbreak/>en</ref>
        </imprint>
      </monogr>
      <note type="bnote">Selection rate: 12.9%</note>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid29" type="inproceedings" rend="refer" n="refercite:nicolae:2010:inria-00456801:1">
      <analytic>
        <title level="a">BlobSeer: Bringing High Throughput under Heavy Concurrency to Hadoop Map-Reduce Applications</title>
        <author>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2008-idm188504272032">
            <foreName>Diana</foreName>
            <surname>Moise</surname>
            <initial>D.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027550560">
            <foreName>Matthieu</foreName>
            <surname>Dorier</surname>
            <initial>M.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">24th IEEE International Parallel and Distributed Processing Symposium (IPDPS 2010)</title>
        <loc>Atlanta</loc>
        <imprint>
          <publisher>
            <orgName type="organisation">IEEE and ACM</orgName>
          </publisher>
          <dateStruct>
            <month>Apr</month>
            <year>2010</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/inria-00456801" type="hal" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00456801</ref>
        </imprint>
      </monogr>
      <note type="bnote">A preliminary version of this paper has been published as Inria Research Report RR-7140</note>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid35" type="article" rend="refer" n="refercite:tran:hal-00640900">
      <identifiant type="doi" value="10.1145/2146382.2146387"/>
      <identifiant type="hal" value="hal-00640900"/>
      <analytic>
        <title level="a">Towards Scalable Array-Oriented Active Storage: the Pyramid Approach</title>
        <author>
          <persName key="kerdata-2009-idm140027572976">
            <foreName>Viet-Trung</foreName>
            <surname>Tran</surname>
            <initial>V.-T.</initial>
          </persName>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">ACM Operating Systems Review</title>
        <imprint>
          <biblScope type="volume">46</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">19-25</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00640900" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00640900</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid33" type="inproceedings" rend="refer" n="refercite:tudoran:hal-00670725">
      <identifiant type="hal" value="hal-00670725"/>
      <analytic>
        <title level="a">TomusBlobs: Towards Communication-Efficient Storage for MapReduce Applications in Azure</title>
        <author>
          <persName key="kerdata-2011-idm367107176512">
            <foreName>Radu</foreName>
            <surname>Tudoran</surname>
            <initial>R.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Hakan</foreName>
            <surname>Soncu</surname>
            <initial>H.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="m">12th IEEE/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGrid'2012)</title>
        <loc>Ottawa, Canada</loc>
        <imprint>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00670725" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00670725</ref>
        </imprint>
      </monogr>
      <note type="bnote">A-Brain project, Inria-Microsoft Research Joint Centre</note>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid30" type="article" rend="refer" n="refercite:MORALES:2007:Inria-00446067:1">
      <identifiant type="doi" value="10.1109/TNSM.2007.070903"/>
      <identifiant type="hal" value="inria-00446067"/>
      <analytic>
        <title level="a">MOve:Design and Evaluation of A Malleable Overlay for Group-Based Applications</title>
        <author>
          <persName>
            <foreName>Ramsés</foreName>
            <surname>Morales</surname>
            <initial>R.</initial>
          </persName>
          <persName key="paris-2006-idm124331438528">
            <foreName>Sébastien</foreName>
            <surname>Monnet</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Indranil</foreName>
            <surname>Gupta</surname>
            <initial>I.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Network and Service Management, Special Issue on Self-Management</title>
        <imprint>
          <biblScope type="volume">4</biblScope>
          <dateStruct>
            <year>2007</year>
          </dateStruct>
          <biblScope type="pages">107-116</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00446067/en/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00446067/<allowbreak/>en/</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="1797" subtype="nonparu" id="kerdata-2012-bid21" type="article" rend="year" n="cite:antoniu:hal-00767029">
      <identifiant type="hal" value="hal-00767029"/>
      <analytic>
        <title level="a">Towards Scalable Data Management for Map-Reduce-based Data-Intensive Applications on Cloud and Hybrid Infrastructures</title>
        <author>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2007-idm243644570096">
            <foreName>Julien</foreName>
            <surname>Bigot</surname>
            <initial>J.</initial>
          </persName>
          <persName>
            <foreName>Christophe</foreName>
            <surname>Blanchet</surname>
            <initial>C.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
          <persName>
            <foreName>François</foreName>
            <surname>Briant</surname>
            <initial>F.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610725168">
            <foreName>Franck</foreName>
            <surname>Cappello</surname>
            <initial>F.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="graal-2006-idm329937294144">
            <foreName>Frédéric</foreName>
            <surname>Desprez</surname>
            <initial>F.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610702240">
            <foreName>Gilles</foreName>
            <surname>Fedak</surname>
            <initial>G.</initial>
          </persName>
          <persName key="graal-2010-idm486711849760">
            <foreName>Sylvain</foreName>
            <surname>Gault</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Kate</foreName>
            <surname>Keahey</surname>
            <initial>K.</initial>
          </persName>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2006-idm124332484688">
            <foreName>Christian</foreName>
            <surname>Pérez</surname>
            <initial>C.</initial>
          </persName>
          <persName key="graal-2011-idm335566973488">
            <foreName>Anthony</foreName>
            <surname>Simonet</surname>
            <initial>A.</initial>
          </persName>
          <persName key="algorille-2006-idm304998935584">
            <foreName>Frédéric</foreName>
            <surname>Suter</surname>
            <initial>F.</initial>
          </persName>
          <persName key="graal-2009-idm463509981664">
            <foreName>Bing</foreName>
            <surname>Tang</surname>
            <initial>B.</initial>
          </persName>
          <persName>
            <foreName>Raphael</foreName>
            <surname>Terreux</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-editorial-board="yes" x-international-audience="yes" id="rid0246111111110">
        <idno type="issn">2043-9989</idno>
        <title level="j">International Journal of Cloud Computing (IJCC)</title>
        <imprint>
          <dateStruct>
            <year>2013</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00767029" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00767029</ref>
        </imprint>
      </monogr>
      <note type="bnote">To appear</note>
    </biblStruct>
    <biblStruct dedoublkey="0009" id="kerdata-2012-bid20" type="article" rend="year" n="cite:antoniu:hal-00684384">
      <identifiant type="hal" value="hal-00684384"/>
      <analytic>
        <title level="a">A-Brain: Using the Cloud to Understand the Impact of Genetic Variability on the Brain</title>
        <author>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="parietal-2011-idm378416023648">
            <foreName>Benoit</foreName>
            <surname>Da Mota</surname>
            <initial>B.</initial>
          </persName>
          <persName key="parietal-2008-idm343501673536">
            <foreName>Bertrand</foreName>
            <surname>Thirion</surname>
            <initial>B.</initial>
          </persName>
          <persName key="kerdata-2011-idm367107176512">
            <foreName>Radu</foreName>
            <surname>Tudoran</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-editorial-board="yes" x-international-audience="yes" id="rid00554">
        <idno type="issn">0926-4981</idno>
        <title level="j">ERCIM News</title>
        <imprint>
          <dateStruct>
            <month>April</month>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">21-22</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00684384" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00684384</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="1788" id="kerdata-2012-bid19" type="article" rend="year" n="cite:carpenamarie:hal-00670923">
      <identifiant type="hal" value="hal-00670923"/>
      <analytic>
        <title level="a">Towards a Generic Security Framework for Cloud Data Management Environments</title>
        <author>
          <persName key="paris-2008-idm188504275952">
            <foreName>Alexandra</foreName>
            <surname>Carpen-Amarie</surname>
            <initial>A.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName>
            <foreName>Catalin</foreName>
            <surname>Leordeanu</surname>
            <initial>C.</initial>
          </persName>
          <persName>
            <foreName>Cristina</foreName>
            <surname>Basescu</surname>
            <initial>C.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-editorial-board="yes" x-international-audience="yes" id="rid02386">
        <idno type="issn">1947-3532</idno>
        <title level="j">International Journal of Distributed Systems and Technologies (IJDST), Special Issue on Security, Privacy and Trust</title>
        <imprint>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00670923" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00670923</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="1783" id="kerdata-2012-bid22" type="article" rend="year" n="cite:costan:hal-00767034">
      <identifiant type="hal" value="hal-00767034"/>
      <analytic>
        <title level="a">TomusBlobs: Scalable Data-intensive Processing on Azure Clouds</title>
        <author>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="kerdata-2011-idm367107176512">
            <foreName>Radu</foreName>
            <surname>Tudoran</surname>
            <initial>R.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Goetz</foreName>
            <surname>Brasche</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-editorial-board="yes" x-international-audience="yes" id="rid00483">
        <idno type="issn">1532-0626</idno>
        <title level="j">Concurrency and Computation Practice and Experience</title>
        <imprint>
          <dateStruct>
            <year>2013</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00767034" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00767034</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="1796" id="kerdata-2012-bid18" type="article" rend="year" n="cite:tran:hal-00640900">
      <identifiant type="doi" value="10.1145/2146382.2146387"/>
      <identifiant type="hal" value="hal-00640900"/>
      <analytic>
        <title level="a">Towards Scalable Array-Oriented Active Storage: the Pyramid Approach</title>
        <author>
          <persName key="kerdata-2009-idm140027572976">
            <foreName>Viet-Trung</foreName>
            <surname>Tran</surname>
            <initial>V.-T.</initial>
          </persName>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-editorial-board="yes" x-international-audience="yes" id="rid00008">
        <idno type="issn">0163-5980</idno>
        <title level="j">ACM Operating Systems Review</title>
        <imprint>
          <biblScope type="volume">46</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">19-25</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00640900" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00640900</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="4892" id="kerdata-2012-bid5" type="inproceedings" rend="year" n="cite:antoniu:hal-00684866">
      <identifiant type="hal" value="hal-00684866"/>
      <analytic>
        <title level="a">Towards Scalable Data Management for Map-Reduce-based Data-Intensive Applications on Cloud and Hybrid Infrastructures</title>
        <author>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2007-idm243644570096">
            <foreName>Julien</foreName>
            <surname>Bigot</surname>
            <initial>J.</initial>
          </persName>
          <persName>
            <foreName>Christophe</foreName>
            <surname>Blanchet</surname>
            <initial>C.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
          <persName>
            <foreName>François</foreName>
            <surname>Briant</surname>
            <initial>F.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610725168">
            <foreName>Franck</foreName>
            <surname>Cappello</surname>
            <initial>F.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="graal-2006-idm329937294144">
            <foreName>Frédéric</foreName>
            <surname>Desprez</surname>
            <initial>F.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610702240">
            <foreName>Gilles</foreName>
            <surname>Fedak</surname>
            <initial>G.</initial>
          </persName>
          <persName key="graal-2010-idm486711849760">
            <foreName>Sylvain</foreName>
            <surname>Gault</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Kate</foreName>
            <surname>Keahey</surname>
            <initial>K.</initial>
          </persName>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2006-idm124332484688">
            <foreName>Christian</foreName>
            <surname>Pérez</surname>
            <initial>C.</initial>
          </persName>
          <persName key="graal-2011-idm335566973488">
            <foreName>Anthony</foreName>
            <surname>Simonet</surname>
            <initial>A.</initial>
          </persName>
          <persName key="algorille-2006-idm304998935584">
            <foreName>Frédéric</foreName>
            <surname>Suter</surname>
            <initial>F.</initial>
          </persName>
          <persName key="graal-2009-idm463509981664">
            <foreName>Bing</foreName>
            <surname>Tang</surname>
            <initial>B.</initial>
          </persName>
          <persName>
            <foreName>Raphael</foreName>
            <surname>Terreux</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">1st International IBM Cloud Academy Conference - ICA CON 2012</title>
        <loc>Research Triangle Park, North Carolina, États-Unis</loc>
        <imprint>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00684866" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00684866</ref>
        </imprint>
        <meeting id="cid623478">
          <title>International IBM Cloud Academy Conference</title>
          <num>1</num>
          <abbr type="sigle">ICA CON</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="3527" id="kerdata-2012-bid12" type="inproceedings" rend="year" n="cite:chihoub:hal-00734050">
      <identifiant type="hal" value="hal-00734050"/>
      <analytic>
        <title level="a">Harmony: Towards Automated Self-Adaptive Consistency in Cloud Storage</title>
        <author>
          <persName key="kerdata-2010-idm58934898256">
            <foreName>Houssem-Eddine</foreName>
            <surname>Chihoub</surname>
            <initial>H.-E.</initial>
          </persName>
          <persName key="kerdata-2011-idm367107193856">
            <foreName>Shadi</foreName>
            <surname>Ibrahim</surname>
            <initial>S.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>María</foreName>
            <surname>Pérez</surname>
            <initial>M.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">2012 IEEE International Conference on Cluster Computing</title>
        <loc>Beijing, Chine</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>September</month>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00734050" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00734050</ref>
        </imprint>
        <meeting id="cid81665">
          <title>IEEE International Conference on Cluster Computing</title>
          <num>2012</num>
          <abbr type="sigle">Cluster</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="3072" id="kerdata-2012-bid15" type="inproceedings" rend="year" n="cite:dorier:hal-00715252">
      <identifiant type="hal" value="hal-00715252"/>
      <analytic>
        <title level="a">Damaris: How to Efficiently Leverage Multicore Parallelism to Achieve Scalable, Jitter-free I/O</title>
        <author>
          <persName key="kerdata-2009-idm140027550560">
            <foreName>Matthieu</foreName>
            <surname>Dorier</surname>
            <initial>M.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610725168">
            <foreName>Franck</foreName>
            <surname>Cappello</surname>
            <initial>F.</initial>
          </persName>
          <persName>
            <foreName>Marc</foreName>
            <surname>Snir</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Leigh</foreName>
            <surname>Orf</surname>
            <initial>L.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">CLUSTER - IEEE International Conference on Cluster Computing</title>
        <loc>Beijing, Chine</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>September</month>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00715252" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00715252</ref>
        </imprint>
        <meeting id="cid81665">
          <title>IEEE International Conference on Cluster Computing</title>
          <num>2012</num>
          <abbr type="sigle">Cluster</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="3840" id="kerdata-2012-bid6" type="inproceedings" rend="year" n="cite:ibrahim:hal-00670813">
      <identifiant type="hal" value="hal-00670813"/>
      <analytic>
        <title level="a">Maestro: Replica-Aware Map Scheduling for MapReduce</title>
        <author>
          <persName key="kerdata-2011-idm367107193856">
            <foreName>Shadi</foreName>
            <surname>Ibrahim</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Hai</foreName>
            <surname>Jin</surname>
            <initial>H.</initial>
          </persName>
          <persName key="graal-2010-idm486711892560">
            <foreName>Lu</foreName>
            <surname>Lu</surname>
            <initial>L.</initial>
          </persName>
          <persName>
            <foreName>Bingsheng</foreName>
            <surname>He</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Song</foreName>
            <surname>Wu</surname>
            <initial>S.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">The 12th IEEE/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGRID'2012)</title>
        <loc>Ottawa, Canada</loc>
        <imprint>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00670813" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00670813</ref>
        </imprint>
        <meeting id="cid88920">
          <title>IEEE International Symposium on Cluster Computing and the Grid</title>
          <num>12</num>
          <abbr type="sigle">CCGRID</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="4146" id="kerdata-2012-bid24" type="inproceedings" rend="year" n="cite:moise:hal-00706844">
      <identifiant type="hal" value="hal-00706844"/>
      <analytic>
        <title level="a">On-the-fly Task Execution for Speeding Up Pipelined MapReduce</title>
        <author>
          <persName key="paris-2008-idm188504272032">
            <foreName>Diana</foreName>
            <surname>Moise</surname>
            <initial>D.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">Euro-Par - 18th International European Conference on Parallel and Distributed Computing - 2012</title>
        <loc>Rhodes Island, Grèce</loc>
        <imprint>
          <dateStruct>
            <month>August</month>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00706844" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00706844</ref>
        </imprint>
        <meeting id="cid306382">
          <title>International Euro-Par Conference on Parallel Processing</title>
          <num>18</num>
          <abbr type="sigle">Euro-Par</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="2658" subtype="nonparu" id="kerdata-2012-bid23" type="inproceedings" rend="year" n="cite:tudoran:hal-00677842">
      <identifiant type="hal" value="hal-00677842"/>
      <analytic>
        <title level="a">A Performance Evaluation of Azure and Nimbus Clouds for Scientific Applications</title>
        <author>
          <persName key="kerdata-2011-idm367107176512">
            <foreName>Radu</foreName>
            <surname>Tudoran</surname>
            <initial>R.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">CloudCP 2012 – 2nd International Workshop on Cloud Computing Platforms, Held in conjunction with the ACM SIGOPS Eurosys 12 conference</title>
        <loc>Bern, Suisse</loc>
        <imprint>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00677842" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00677842</ref>
        </imprint>
        <meeting id="cid367773">
          <title>International Workshop on Cloud Computing Platforms</title>
          <num>2</num>
          <abbr type="sigle">CloudCP</abbr>
        </meeting>
      </monogr>
      <note type="bnote">To appear</note>
    </biblStruct>
    <biblStruct dedoublkey="4851" id="kerdata-2012-bid10" type="inproceedings" rend="year" n="cite:tudoran:hal-00670725">
      <identifiant type="hal" value="hal-00670725"/>
      <analytic>
        <title level="a">TomusBlobs: Towards Communication-Efficient Storage for MapReduce Applications in Azure</title>
        <author>
          <persName key="kerdata-2011-idm367107176512">
            <foreName>Radu</foreName>
            <surname>Tudoran</surname>
            <initial>R.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Hakan</foreName>
            <surname>Soncu</surname>
            <initial>H.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">12th IEEE/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGrid'2012)</title>
        <loc>Ottawa, Canada</loc>
        <imprint>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00670725" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00670725</ref>
        </imprint>
        <meeting id="cid88920">
          <title>IEEE International Symposium on Cluster Computing and the Grid</title>
          <num>12</num>
          <abbr type="sigle">CCGRID</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    <biblStruct dedoublkey="3853" subtype="nonparu" id="kerdata-2012-bid11" type="inproceedings" rend="year" n="cite:tudoran:hal-00684814">
      <identifiant type="hal" value="hal-00684814"/>
      <analytic>
        <title level="a">MapIterativeReduce: A Framework for Reduction-Intensive Data Processing on Azure Clouds</title>
        <author>
          <persName key="kerdata-2011-idm367107176512">
            <foreName>Radu</foreName>
            <surname>Tudoran</surname>
            <initial>R.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027562000">
            <foreName>Alexandru</foreName>
            <surname>Costan</surname>
            <initial>A.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">Third International Workshop on MapReduce and its Applications (MAPREDUCE'12), held in conjunction with ACM HPDC'12</title>
        <loc>Delft, Pays-Bas</loc>
        <imprint>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00684814" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00684814</ref>
        </imprint>
        <meeting id="cid399515">
          <title>International Workshop on MapReduce and its Applications</title>
          <num>3</num>
          <abbr type="sigle">MAPREDUCE</abbr>
        </meeting>
      </monogr>
      <note type="bnote">To appear</note>
    </biblStruct>
    <biblStruct dedoublkey="6080" id="kerdata-2012-bid25" type="techreport" rend="year" n="cite:canon:hal-00675964">
      <identifiant type="hal" value="hal-00675964"/>
      <monogr>
        <title level="m">Scheduling Associative Reductions with Homogeneous Costs when Overlapping Communications and Computations</title>
        <author>
          <persName key="algorille-2007-idm186081019872">
            <foreName>Louis-Claude</foreName>
            <surname>Canon</surname>
            <initial>L.-C.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
        <imprint>
          <biblScope type="number">RR-7898</biblScope>
          <publisher>
            <orgName type="institution">Inria</orgName>
          </publisher>
          <dateStruct>
            <month>March</month>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00675964" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00675964</ref>
        </imprint>
      </monogr>
      <note type="typdoc">Rapport de recherche</note>
    </biblStruct>
    <biblStruct dedoublkey="5756" id="kerdata-2012-bid13" type="techreport" rend="year" n="cite:chihoub:hal-00756314">
      <identifiant type="hal" value="hal-00756314"/>
      <monogr>
        <title level="m">Consistency in the Cloud:When Money Does Matter!</title>
        <author>
          <persName key="kerdata-2010-idm58934898256">
            <foreName>Houssem-Eddine</foreName>
            <surname>Chihoub</surname>
            <initial>H.-E.</initial>
          </persName>
          <persName key="kerdata-2011-idm367107193856">
            <foreName>Shadi</foreName>
            <surname>Ibrahim</surname>
            <initial>S.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>María</foreName>
            <surname>Pérez</surname>
            <initial>M.</initial>
          </persName>
        </author>
        <imprint>
          <publisher>
            <orgName type="institution">Inria</orgName>
          </publisher>
          <dateStruct>
            <month>November</month>
            <year>2012</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/hal-00756314" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00756314</ref>
        </imprint>
      </monogr>
      <note type="typdoc">Rapport de recherche</note>
    </biblStruct>
    <biblStruct dedoublkey="5790" id="kerdata-2012-bid26" type="techreport" rend="year" n="cite:dorier:inria-00614597">
      <identifiant type="hal" value="inria-00614597"/>
      <monogr>
        <title level="m">Damaris: Leveraging Multicore Parallelism to Mask I/O Jitter</title>
        <author>
          <persName key="kerdata-2009-idm140027550560">
            <foreName>Matthieu</foreName>
            <surname>Dorier</surname>
            <initial>M.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610725168">
            <foreName>Franck</foreName>
            <surname>Cappello</surname>
            <initial>F.</initial>
          </persName>
          <persName>
            <foreName>Marc</foreName>
            <surname>Snir</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Leigh</foreName>
            <surname>Orf</surname>
            <initial>L.</initial>
          </persName>
        </author>
        <imprint>
          <biblScope type="number">RR-7706</biblScope>
          <publisher>
            <orgName type="institution">Inria</orgName>
          </publisher>
          <dateStruct>
            <month>April</month>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">36</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00614597" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00614597</ref>
        </imprint>
      </monogr>
      <note type="typdoc">Rapport de recherche</note>
    </biblStruct>
    <biblStruct dedoublkey="5808" id="kerdata-2012-bid14" type="techreport" rend="year" n="cite:tran:hal-00766219">
      <identifiant type="hal" value="hal-00766219"/>
      <monogr>
        <title level="m">DStore: An in-memory document-oriented store</title>
        <author>
          <persName key="kerdata-2009-idm140027572976">
            <foreName>Viet-Trung</foreName>
            <surname>Tran</surname>
            <initial>V.-T.</initial>
          </persName>
          <persName>
            <foreName>Dushyanth</foreName>
            <surname>Narayanan</surname>
            <initial>D.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
        </author>
        <imprint>
          <biblScope type="number">RR-8188</biblScope>
          <publisher>
            <orgName type="institution">Inria</orgName>
          </publisher>
          <dateStruct>
            <month>December</month>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">24</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00766219" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00766219</ref>
        </imprint>
      </monogr>
      <note type="typdoc">Rapport de recherche</note>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid1" type="misc" rend="foot" n="footcite:AmazonMapReduce">
      <monogr>
        <title level="m">Amazon Elastic MapReduce</title>
        <imprint>
          <ref xlink:href="http://aws.amazon.com/elasticmapreduce/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>aws.<allowbreak/>amazon.<allowbreak/>com/<allowbreak/>elasticmapreduce/</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid16" type="article" rend="foot" n="footcite:bryan:2002">
      <identifiant type="doi" value="10.1175/1520-0493(2002)130&lt;2917:ABSFMN&gt;2.0.CO;2"/>
      <analytic>
        <title level="a">A Benchmark Simulation for Moist Nonhydrostatic Numerical Models</title>
        <author>
          <persName>
            <foreName>George H.</foreName>
            <surname>Bryan</surname>
            <initial>G. H.</initial>
          </persName>
          <persName>
            <foreName>J. Michael</foreName>
            <surname>Fritsch</surname>
            <initial>J. M.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Monthly Weather Review</title>
        <imprint>
          <biblScope type="volume">130</biblScope>
          <biblScope type="number">12</biblScope>
          <dateStruct>
            <year>2002</year>
          </dateStruct>
          <biblScope type="pages">2917–2928</biblScope>
          <ref xlink:href="http://journals.ametsoc.org/doi/abs/10.1175/1520-0493%282002%29130%3C2917%3AABSFMN%3E2.0.CO%3B2" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>journals.<allowbreak/>ametsoc.<allowbreak/>org/<allowbreak/>doi/<allowbreak/>abs/<allowbreak/>10.<allowbreak/>1175/<allowbreak/>1520-0493%282002%29130%3C2917%3AABSFMN%3E2.<allowbreak/>0.<allowbreak/>CO%3B2</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid0" type="article" rend="foot" n="footcite:mapreduce">
      <analytic>
        <title level="a">MapReduce: simplified data processing on large clusters</title>
        <author>
          <persName>
            <foreName>Jeffrey</foreName>
            <surname>Dean</surname>
            <initial>J.</initial>
          </persName>
          <persName>
            <foreName>Sanjay</foreName>
            <surname>Ghemawat</surname>
            <initial>S.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Communications of the ACM</title>
        <imprint>
          <biblScope type="volume">51</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <year>2008</year>
          </dateStruct>
          <biblScope type="pages">107–113</biblScope>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid9" type="inproceedings" rend="foot" n="footcite:Dinu:2012:UEI:2287076.2287108">
      <identifiant type="doi" value="10.1145/2287076.2287108"/>
      <analytic>
        <title level="a">Understanding the effects and implications of compute node related failures in hadoop</title>
        <author>
          <persName>
            <foreName>Florin</foreName>
            <surname>Dinu</surname>
            <initial>F.</initial>
          </persName>
          <persName>
            <foreName>T.S. Eugene</foreName>
            <surname>Ng</surname>
            <initial>T. E.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="m">Proceedings of the 21st international symposium on High-Performance Parallel and Distributed Computing</title>
        <loc>New York, NY, USA</loc>
        <title level="s">HPDC '12</title>
        <imprint>
          <publisher>
            <orgName>ACM</orgName>
          </publisher>
          <dateStruct>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">187–198</biblScope>
          <ref xlink:href="http://doi.acm.org/10.1145/2287076.2287108" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>doi.<allowbreak/>acm.<allowbreak/>org/<allowbreak/>10.<allowbreak/>1145/<allowbreak/>2287076.<allowbreak/>2287108</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid3" type="misc" rend="foot" n="footcite:eesi">
      <monogr>
        <title level="m">European Exascale Software Initiative</title>
        <imprint>
          <ref xlink:href="http://www.eesi-project.eu" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>www.<allowbreak/>eesi-project.<allowbreak/>eu</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid17" type="misc" rend="foot" n="footcite:Nek5000">
      <monogr>
        <title level="m">nek5000 Web page</title>
        <author>
          <persName>
            <foreName>Paul F.</foreName>
            <surname>Fischer</surname>
            <initial>P. F.</initial>
          </persName>
          <persName>
            <foreName>James W.</foreName>
            <surname>Lottes</surname>
            <initial>J. W.</initial>
          </persName>
          <persName>
            <foreName>Stefan G.</foreName>
            <surname>Kerkemeier</surname>
            <initial>S. G.</initial>
          </persName>
        </author>
        <imprint>
          <dateStruct>
            <year>2008</year>
          </dateStruct>
          <ref xlink:href="http://nek5000.mcs.anl.gov" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>nek5000.<allowbreak/>mcs.<allowbreak/>anl.<allowbreak/>gov</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid2" type="misc" rend="foot" n="footcite:iesp">
      <monogr>
        <title level="m">International Exascale Software Program</title>
        <imprint>
          <ref xlink:href="http://www.exascale.org/iesp/Main_Page" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>www.<allowbreak/>exascale.<allowbreak/>org/<allowbreak/>iesp/<allowbreak/>Main_Page</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid7" type="article" rend="foot" n="footcite:G5K06">
      <analytic>
        <title level="a">Grid'5000: a large scale and highly reconfigurable experimental Grid testbed</title>
        <author>
          <persName key="paris-2006-idm124332493040">
            <foreName>Yvon</foreName>
            <surname>Jégou</surname>
            <initial>Y.</initial>
          </persName>
          <persName key="caiman-2006-idm450347370576">
            <foreName>Stephane</foreName>
            <surname>Lantéri</surname>
            <initial>S.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610613648">
            <foreName>Julien</foreName>
            <surname>Leduc</surname>
            <initial>J.</initial>
          </persName>
          <persName>
            <foreName>Melab</foreName>
            <surname>Noredine</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Guillaume</foreName>
            <surname>Mornet</surname>
            <initial>G.</initial>
          </persName>
          <persName key="runtime-2006-idm507675120096">
            <foreName>Raymond</foreName>
            <surname>Namyst</surname>
            <initial>R.</initial>
          </persName>
          <persName>
            <foreName>Pascale</foreName>
            <surname>Primet</surname>
            <initial>P.</initial>
          </persName>
          <persName key="grand-large-2006-idm343610631216">
            <foreName>Benjamin</foreName>
            <surname>Quetier</surname>
            <initial>B.</initial>
          </persName>
          <persName key="mescal-2006-idm111690845920">
            <foreName>Olivier</foreName>
            <surname>Richard</surname>
            <initial>O.</initial>
          </persName>
          <persName key="dolphin-2006-idm72490914608">
            <foreName>El-Ghazali</foreName>
            <surname>Talbi</surname>
            <initial>E.-G.</initial>
          </persName>
          <persName>
            <foreName>Touche</foreName>
            <surname>Iréa</surname>
            <initial>T.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of High Performance Computing Applications</title>
        <imprint>
          <biblScope type="volume">20</biblScope>
          <biblScope type="number">4</biblScope>
          <dateStruct>
            <month>November</month>
            <year>2006</year>
          </dateStruct>
          <biblScope type="pages">481-494</biblScope>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid8" type="inproceedings" rend="foot" n="footcite:Memishi:2012:6266995">
      <identifiant type="doi" value="10.1109/HPCSim.2012.6266995"/>
      <analytic>
        <title level="a">Enhanced failure detection mechanism in MapReduce</title>
        <author>
          <persName key="kerdata-2011-idm367107170400">
            <foreName>B.</foreName>
            <surname>Memishi</surname>
            <initial>B.</initial>
          </persName>
          <persName>
            <foreName>M.S.</foreName>
            <surname>Perez</surname>
            <initial>M.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="m">High Performance Computing and Simulation (HPCS), 2012 International Conference on</title>
        <imprint>
          <dateStruct>
            <month>july</month>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">690 -692</biblScope>
          <ref xlink:href="http://dx.doi.org/10.1109/HPCSim.2012.6266995" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>dx.<allowbreak/>doi.<allowbreak/>org/<allowbreak/>10.<allowbreak/>1109/<allowbreak/>HPCSim.<allowbreak/>2012.<allowbreak/>6266995</ref>
        </imprint>
      </monogr>
    </biblStruct>
    <biblStruct id="kerdata-2012-bid4" type="inproceedings" rend="foot" n="footcite:nicolae:2010:inria-00456801:1">
      <analytic>
        <title level="a">BlobSeer: Bringing High Throughput under Heavy Concurrency to Hadoop Map-Reduce Applications</title>
        <author>
          <persName key="paris-2007-idm243644537488">
            <foreName>Bogdan</foreName>
            <surname>Nicolae</surname>
            <initial>B.</initial>
          </persName>
          <persName key="paris-2008-idm188504272032">
            <foreName>Diana</foreName>
            <surname>Moise</surname>
            <initial>D.</initial>
          </persName>
          <persName key="paris-2006-idm124332495696">
            <foreName>Gabriel</foreName>
            <surname>Antoniu</surname>
            <initial>G.</initial>
          </persName>
          <persName key="paris-2006-idm124332467968">
            <foreName>Luc</foreName>
            <surname>Bougé</surname>
            <initial>L.</initial>
          </persName>
          <persName key="kerdata-2009-idm140027550560">
            <foreName>Matthieu</foreName>
            <surname>Dorier</surname>
            <initial>M.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-international-audience="yes" x-proceedings="yes">
        <title level="m">24th IEEE International Parallel and Distributed Processing Symposium (IPDPS 2010)</title>
        <loc>Atlanta, GA, USA</loc>
        <imprint>
          <publisher>
            <orgName type="organisation">IEEE and ACM</orgName>
          </publisher>
          <dateStruct>
            <month>April</month>
            <year>2010</year>
          </dateStruct>
          <ref xlink:href="http://hal.inria.fr/inria-00456801/en/" type="hal" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00456801/<allowbreak/>en/<allowbreak/></ref>
        </imprint>
      </monogr>
      <note type="bnote">A preliminary version of this paper has been published as Inria Research Report RR-7140</note>
    </biblStruct>
  </biblio>
</raweb>
