<?xml version="1.0" encoding="utf-8"?>
<raweb xmlns:xlink="http://www.w3.org/1999/xlink" xml:lang="en" year="2018">
  <identification id="perception" isproject="true">
    <shortname>PERCEPTION</shortname>
    <projectName>Interpretation and Modelling of Images and Videos</projectName>
    <theme-de-recherche>Vision, perception and multimedia interpretation</theme-de-recherche>
    <domaine-de-recherche>Perception, Cognition and Interaction</domaine-de-recherche>
    <urlTeam>http://team.inria.fr/perception</urlTeam>
    <structure_exterieure type="Labs">
      <libelle>Laboratoire Jean Kuntzmann (LJK)</libelle>
    </structure_exterieure>
    <header_dates_team>Creation of the Team: 2006 September 01, updated into Project-Team: 2008 January 01</header_dates_team>
    <LeTypeProjet>Project-Team</LeTypeProjet>
    <keywordsSdN>
      <term>A3.4. - Machine learning and statistics</term>
      <term>A5.1. - Human-Computer Interaction</term>
      <term>A5.3. - Image processing and analysis</term>
      <term>A5.4. - Computer vision</term>
      <term>A5.7. - Audio modeling and processing</term>
      <term>A5.10.2. - Perception</term>
      <term>A5.10.5. - Robot interaction (with the environment, humans, other robots)</term>
      <term>A9.2. - Machine learning</term>
      <term>A9.5. - Robotics</term>
    </keywordsSdN>
    <keywordsSecteurs>
      <term>B5.6. - Robotic systems</term>
    </keywordsSecteurs>
    <UR name="Grenoble"/>
  </identification>
  <team id="uid1">
    <person key="perception-2018-idp146240">
      <firstname>Radu Patrice</firstname>
      <lastname>Horaud</lastname>
      <categoryPro>Chercheur</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Team leader, Inria, Senior Researcher</moreinfo>
      <hdr>oui</hdr>
    </person>
    <person key="perception-2018-idp149152">
      <firstname>Xavier</firstname>
      <lastname>Alameda-Pineda</lastname>
      <categoryPro>Chercheur</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, Researcher</moreinfo>
    </person>
    <person key="perception-2018-idp151616">
      <firstname>Xiaofei</firstname>
      <lastname>Li</lastname>
      <categoryPro>Chercheur</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, Starting Research Position</moreinfo>
    </person>
    <person key="perception-2018-idp154112">
      <firstname>Pablo</firstname>
      <lastname>Mesejo Santiago</lastname>
      <categoryPro>Chercheur</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, Starting Research Position, until March 2018</moreinfo>
    </person>
    <person key="perception-2018-idp156608">
      <firstname>Laurent</firstname>
      <lastname>Girin</lastname>
      <categoryPro>Enseignant</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Institut polytechnique de Grenoble, Professor</moreinfo>
      <hdr>oui</hdr>
    </person>
    <person key="perception-2018-idp159488">
      <firstname>Simon</firstname>
      <lastname>Leglaive</lastname>
      <categoryPro>PostDoc</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, since February 2018</moreinfo>
    </person>
    <person key="perception-2018-idp161968">
      <firstname>Mostafa</firstname>
      <lastname>Sadeghi</lastname>
      <categoryPro>PostDoc</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, since August 2018</moreinfo>
    </person>
    <person key="perception-2018-idp164448">
      <firstname>Yutong</firstname>
      <lastname>Ban</lastname>
      <categoryPro>PhD</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria</moreinfo>
    </person>
    <person key="perception-2018-idp166848">
      <firstname>Guillaume</firstname>
      <lastname>Delorme</lastname>
      <categoryPro>PhD</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria</moreinfo>
    </person>
    <person key="perception-2018-idp169280">
      <firstname>Sylvain</firstname>
      <lastname>Guy</lastname>
      <categoryPro>PhD</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Univ Grenoble Alpes</moreinfo>
    </person>
    <person key="perception-2018-idp171728">
      <firstname>Stephane</firstname>
      <lastname>Lathuiliere</lastname>
      <categoryPro>PhD</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, until May 2018</moreinfo>
    </person>
    <person key="perception-2018-idp174160">
      <firstname>Benoit</firstname>
      <lastname>Masse</lastname>
      <categoryPro>PhD</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Univ Grenoble Alpes, until November 2018</moreinfo>
    </person>
    <person key="perception-2018-idp176624">
      <firstname>Yihong</firstname>
      <lastname>Xu</lastname>
      <categoryPro>PhD</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, since September 2018</moreinfo>
    </person>
    <person key="perception-2018-idp179040">
      <firstname>Soraya</firstname>
      <lastname>Arias</lastname>
      <categoryPro>Technique</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria</moreinfo>
    </person>
    <person key="perception-2018-idp181504">
      <firstname>Bastien</firstname>
      <lastname>Mourgue</lastname>
      <categoryPro>Technique</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria</moreinfo>
    </person>
    <person key="perception-2018-idp183968">
      <firstname>Guillaume</firstname>
      <lastname>Sarrazin</lastname>
      <categoryPro>Technique</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria</moreinfo>
    </person>
    <person key="perception-2018-idp186432">
      <firstname>Victor</firstname>
      <lastname>Bros</lastname>
      <categoryPro>Stagiaire</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, from Jun 2018 until Jul 2018</moreinfo>
    </person>
    <person key="perception-2018-idp188912">
      <firstname>Caroline</firstname>
      <lastname>Dam Hieu</lastname>
      <categoryPro>Stagiaire</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, from May 2018 until Aug 2018</moreinfo>
    </person>
    <person key="perception-2018-idp191392">
      <firstname>Fatbardha</firstname>
      <lastname>Hoxha</lastname>
      <categoryPro>Stagiaire</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria, from Apr 2018 until Jul 2018</moreinfo>
    </person>
    <person key="thoth-2018-idp213856">
      <firstname>Nathalie</firstname>
      <lastname>Gillot</lastname>
      <categoryPro>Assistant</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Inria</moreinfo>
    </person>
    <person key="perception-2018-idp196336">
      <firstname>Christine</firstname>
      <lastname>Evers</lastname>
      <categoryPro>Visiteur</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Imperial College London, from Jul 2018 until Aug 2018</moreinfo>
    </person>
    <person key="perception-2018-idp198832">
      <firstname>Sharon</firstname>
      <lastname>Gannot</lastname>
      <categoryPro>Visiteur</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>Bar Ilan University, from Jan 2018 until Feb 2018</moreinfo>
    </person>
    <person key="perception-2018-idp201328">
      <firstname>Tomislav</firstname>
      <lastname>Pribanic</lastname>
      <categoryPro>Visiteur</categoryPro>
      <research-centre>Grenoble</research-centre>
      <moreinfo>University of Zagreb, from Apr 2018 until Aug 2018</moreinfo>
    </person>
  </team>
  <presentation id="uid2">
    <bodyTitle>Overall Objectives</bodyTitle>
    <subsection id="uid3" level="1">
      <bodyTitle>Audio-Visual Machine Perception</bodyTitle>
      <object id="uid4">
        <table>
          <tr>
            <td>
              <ressource xlink:href="IMG/vavit.png" type="figure" width="422.73235pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
            </td>
          </tr>
        </table>
        <caption>This figure illustrates the audio-visual multiple-person tracking that has been developed by the team <ref xlink:href="#perception-2018-bid0" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid1" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid2" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The tracker is based on variational inference <ref xlink:href="#perception-2018-bid3" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> and on supervised sound-source localization <ref xlink:href="#perception-2018-bid4" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid5" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. Each person is identified with a digit. Green digits denote active speakers while red digits denote silent persons. The next rows show the covariances (uncertainties) associated with the visual (second row), audio (third row) and dynamic (fourth row) contributions for tracking a varying number of persons. Notice the large uncertainty associated with audio and the small uncertainty associated with the dynamics of the tracker. In the light of this example, one may notice the complementary roles played by vision and audio: vision data are more accurate while audio data provide speaker information. These developments have been supported by the European Union via the FP7 STREP project <i>“Embodied Audition for Robots"</i> (EARS) and the ERC advanced grant <i>“Vision and Hearing in Action"</i> (VHIA).</caption>
      </object>
      <p>Auditory and visual perception play a complementary role in human interaction. Perception enables people to communicate based on verbal (speech and language) and non-verbal (facial expressions, visual gaze, head movements, hand and body gesturing) communication. These communication modalities have a large degree of overlap, in particular in social contexts. Moreover, the modalities disambiguate each other whenever one of the modalities is weak, ambiguous, or corrupted by various perturbations. Human-computer interaction (HCI) has attempted to address these issues, e.g., using smart &amp; portable devices. In HCI the user is in the loop for decision taking: images and sounds are recorded purposively in order to optimize their quality with respect to the task at hand.</p>
      <p>However, the robustness of HCI based on speech recognition degrades significantly as the microphones are located a few meters away from the user. Similarly, face detection and recognition work well under limited lighting conditions and if the cameras are properly oriented towards a person. Altogether, the HCI paradigm cannot be easily extended to less constrained interaction scenarios which involve several users and whenever is important to consider the <i>social context</i>.</p>
      <p>The PERCEPTION team investigates the fundamental role played by audio and visual perception in human-robot interaction (HRI). The main difference between HCI and HRI is that, while the former is user-controlled, the latter is robot-controlled, namely <i>it is implemented with intelligent robots that take decisions and act autonomously</i>. The mid term objective of PERCEPTION is to develop computational models, methods, and applications for enabling non-verbal and verbal interactions between people, analyze their intentions and their dialogue, extract information and synthesize appropriate behaviors, e.g., the robot waves to a person, turns its head towards the dominant speaker, nods, gesticulates, asks questions, gives advices, waits for instructions, etc. The following topics are thoroughly addressed by the team members: audio-visual sound-source separation and localization in natural environments, for example to detect and track moving speakers, inference of temporal models of verbal and non-verbal activities (diarisation), continuous recognition of particular gestures and words, context recognition, and multimodal dialogue.</p>
      <p noindent="true">Video: <ref xlink:href="https://team.inria.fr/perception/demos/lito-video/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>demos/<allowbreak/>lito-video/</ref></p>
    </subsection>
  </presentation>
  <fondements id="uid5">
    <bodyTitle>Research Program</bodyTitle>
    <subsection id="uid6" level="1">
      <bodyTitle>Audio-Visual Scene Analysis</bodyTitle>
      <p>From 2006 to 2009, R. Horaud was the scientific coordinator of the collaborative European project POP (Perception on Purpose), an interdisciplinary effort to understand visual and auditory perception at the crossroads of several disciplines (computational and biological vision, computational auditory analysis, robotics, and psychophysics). This allowed the PERCEPTION team to launch an interdisciplinary research agenda that has been very active for the last five years. There are very few teams in the world that gather scientific competences spanning computer vision, audio signal processing, machine learning and human-robot interaction.
The fusion of several sensorial modalities resides at the heart of the most recent biological theories of perception. Nevertheless, multi-sensor processing is still poorly understood from a computational point of view. In particular and so far, audio-visual fusion has been investigated in the framework of speech processing using close-distance cameras and microphones. The vast majority of these approaches attempt to model the temporal correlation between the auditory signals and the dynamics of lip and facial movements. Our original contribution has been to consider that audio-visual localization and recognition are equally important. We have proposed to take into account the fact that the audio-visual objects of interest live in a three-dimensional physical space and hence we contributed to the emergence of <i>audio-visual scene analysis</i> as a scientific topic in its own right. We proposed several novel statistical approaches based on supervised and unsupervised mixture models. The <i>conjugate mixture model</i> (CMM) is an unsupervised probabilistic model that allows to cluster observations from different modalities (e.g., vision and audio) living in different mathematical spaces <ref xlink:href="#perception-2018-bid6" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid7" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. We thoroughly investigated CMM, provided practical resolution algorithms and studied their convergence properties. We developed several methods for sound localization using two or more microphones <ref xlink:href="#perception-2018-bid8" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The <i>Gaussian locally-linear model</i> (GLLiM) is a partially supervised mixture model that allows to map high-dimensional observations (audio, visual, or concatenations of audio-visual vectors) onto low-dimensional manifolds with a partially known structure <ref xlink:href="#perception-2018-bid9" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. This model is particularly well suited for perception because it encodes both observable and unobservable phenomena. A variant of this model, namely <i>probabilistic piecewise affine mapping</i> has also been proposed and successfully applied to the problem of sound-source localization and separation <ref xlink:href="#perception-2018-bid10" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The European projects HUMAVIPS (2010-2013) coordinated by R. Horaud and EARS (2014-2017), applied audio-visual scene analysis to human-robot interaction.</p>
    </subsection>
    <subsection id="uid7" level="1">
      <bodyTitle>Stereoscopic Vision</bodyTitle>
      <p>Stereoscopy is one of the most studied topics in biological and computer vision. Nevertheless, classical approaches of addressing this problem fail to integrate eye/camera vergence. From a geometric point of view, the integration of vergence is difficult because one has to re-estimate the epipolar geometry at every new eye/camera rotation. From an algorithmic point of view, it is not clear how to combine depth maps obtained with different eyes/cameras relative orientations.
Therefore, we addressed the more general problem of binocular vision that combines the low-level eye/camera geometry, sensor rotations, and practical algorithms based on global optimization <ref xlink:href="#perception-2018-bid11" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid12" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. We studied the link between mathematical and computational approaches to stereo (global optimization and Markov random fields) and the brain plausibility of some of these approaches: indeed, we proposed an original mathematical model for the complex cells in visual-cortex areas V1 and V2 that is based on steering Gaussian filters and that admits simple solutions <ref xlink:href="#perception-2018-bid13" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. This addresses the fundamental issue of how local image structure is represented in the brain/computer and how this structure is used for estimating a dense disparity field. Therefore, the main originality of our work is to address both computational and biological issues within a unifying model of binocular vision. Another equally important problem that still remains to be solved is how to integrate binocular depth maps over time. Recently, we have addressed this problem and proposed a semi-global optimization framework that starts with sparse yet reliable matches and proceeds with propagating them over both space and time. The concept of seed-match propagation has then been extended to TOF-stereo fusion <ref xlink:href="#perception-2018-bid14" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
    </subsection>
    <subsection id="uid8" level="1">
      <bodyTitle>Audio Signal Processing</bodyTitle>
      <p>Audio-visual fusion algorithms necessitate that the two modalities are represented in the same mathematical space. Binaural audition allows to extract sound-source localization (SSL) information from the acoustic signals recorded with two microphones. We have developed several methods, that perform sound localization in the temporal and the spectral domains. If a direct path is assumed, one can exploit the <i>time difference of arrival</i> (TDOA) between two microphones to recover the position of the sound source with respect to the position of the two microphones. The solution is not unique in this case, the sound source lies onto a 2D manifold. However, if one further assumes that the sound source lies in a horizontal plane, it is then possible to extract the azimuth. We used this approach to predict possible sound locations in order to estimate the direction of a speaker <ref xlink:href="#perception-2018-bid7" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. We also developed a geometric formulation and we showed that with four non-coplanar microphones the azimuth and elevation of a single source can be estimated without ambiguity <ref xlink:href="#perception-2018-bid8" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.
We also investigated SSL in the spectral domain. This exploits the filtering effects of the head related transfer function (HRTF): there is a different HRTF for the left and right microphones. The interaural spectral features, namely the ILD (interaural level difference) and IPD (interaural phase difference) can be extracted from the short-time Fourier transforms of the two signals. The sound direction is encoded in these interaural features but it is not clear how to make SSL explicit in this case. We proposed a supervised learning formulation that estimates a mapping from interaural spectral features (ILD and IPD) to source directions using two different setups: audio-motor learning <ref xlink:href="#perception-2018-bid10" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> and audio-visual learning <ref xlink:href="#perception-2018-bid4" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.
</p>
    </subsection>
    <subsection id="uid9" level="1">
      <bodyTitle>Visual Reconstruction With Multiple Color and Depth Cameras</bodyTitle>
      <p>For the last decade, one of the most active topics in computer vision has been the visual reconstruction of objects, people, and complex scenes using a multiple-camera setup. The PERCEPTION team has pioneered this field and by 2006 several team members published seminal papers in the field. Recent work has concentrated onto the robustness of the 3D reconstructed data using probabilistic outlier rejection techniques combined with algebraic geometry principles and linear algebra solvers <ref xlink:href="#perception-2018-bid15" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. Subsequently, we proposed to combine 3D representations of shape (meshes) with photometric data <ref xlink:href="#perception-2018-bid16" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The originality of this work was to represent photometric information as a scalar function over a discrete Riemannian manifold, thus <i>generalizing image analysis to mesh and graph analysis</i>. Manifold equivalents of local-structure detectors and descriptors were developed <ref xlink:href="#perception-2018-bid17" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The outcome of this pioneering work has been twofold: the formulation of a new research topic now addressed by several teams in the world, and allowed us to start a three year collaboration with Samsung Electronics. We developed the novel concept of <i>mixed camera systems</i> combining high-resolution color cameras with low-resolution depth cameras <ref xlink:href="#perception-2018-bid18" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid19" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>,<ref xlink:href="#perception-2018-bid20" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. Together with our start-up company 4D Views Solutions and with Samsung, we developed the first practical depth-color multiple-camera multiple-PC system and the first algorithms to reconstruct high-quality 3D content <ref xlink:href="#perception-2018-bid14" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.
</p>
    </subsection>
    <subsection id="uid10" level="1">
      <bodyTitle>Registration, Tracking and Recognition of People and Actions</bodyTitle>
      <p>The analysis of articulated shapes has challenged standard computer vision algorithms for a long time. There are two difficulties associated with this problem, namely how to represent articulated shapes and how to devise robust registration and tracking methods. We addressed both these difficulties and we proposed a novel kinematic representation that integrates concepts from robotics and from the geometry of vision. In 2008 we proposed a method that parameterizes the occluding contours of a shape with its intrinsic kinematic parameters, such that there is a direct mapping between observed image features and joint parameters <ref xlink:href="#perception-2018-bid21" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. This deterministic model has been motivated by the use of 3D data gathered with multiple cameras. However, this method was not robust to various data flaws and could not achieve state-of-the-art results on standard dataset.
Subsequently, we addressed the problem using probabilistic generative models. We formulated the problem of articulated-pose estimation as a maximum-likelihood with missing data and we devised several tractable algorithms <ref xlink:href="#perception-2018-bid22" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid23" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. We proposed several expectation-maximization procedures applied to various articulated shapes: human bodies, hands, etc. In parallel, we proposed to segment and register articulated shapes represented with graphs by embedding these graphs using the spectral properties of graph Laplacians <ref xlink:href="#perception-2018-bid24" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. This turned out to be a very original approach that has been followed by many other researchers in computer vision and computer graphics.</p>
    </subsection>
  </fondements>
  <highlights id="uid11">
    <bodyTitle>Highlights of the Year</bodyTitle>
    <subsection id="uid12" level="1">
      <bodyTitle>Highlights of the Year</bodyTitle>
      <simplelist>
        <li id="uid13">
          <p noindent="true">As an ERC Advanced Grant holder, Radu Horaud was awarded a Proof of Concept grant for his project Vision and Hearing in Action Laboratory (VHIALab). The project started in February 2018 for a duration of 12 months. Software packages enabling companion robots to robustly interact with multiple users are being developed.</p>
          <p noindent="true">Website: <ref xlink:href="https://team.inria.fr/perception/projects/poc-vhialab/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>projects/<allowbreak/>poc-vhialab/</ref></p>
        </li>
        <li id="uid14">
          <p noindent="true">The 2018 winner of the prestigious ACM Special Interest Group on Multimedia (SIGMM) Rising Star Award is Perception team member Dr. Xavier Alameda-Pineda. The award is given in recognition of Xavier's contributions to multimodal social behavior understanding.</p>
          <p noindent="true">Website: <ref xlink:href="http://sigmm.org/news/sigmm_rising_star_award_2018" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>sigmm.<allowbreak/>org/<allowbreak/>news/<allowbreak/>sigmm_rising_star_award_2018</ref></p>
        </li>
        <li id="uid15">
          <p noindent="true">A book was published by Academic Press (Elsevier), entitled “Multimodal Behavior Analysis in the Wild", co-edited by Xavier Alameda Pineda, Elisa Ricci (Fondazione Bruno Kessler and University of Trento) and Nicu Sebe (University of Trento). The book gathers 20 chapters written by 75 researchers from all over the world <ref xlink:href="#perception-2018-bid25" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
        </li>
      </simplelist>
    </subsection>
  </highlights>
  <logiciels id="uid16">
    <bodyTitle>New Software and Platforms</bodyTitle>
    <subsection id="uid17" level="1">
      <bodyTitle>ECMPR</bodyTitle>
      <p>
        <i>Expectation Conditional Maximization for the Joint Registration of Multiple Point Sets</i>
      </p>
      <p noindent="true"><span class="smallcap" align="left">Functional Description:</span> Rigid registration of two or several point sets based on probabilistic matching between point pairs and a Gaussian mixture model</p>
      <simplelist>
        <li id="uid18">
          <p noindent="true">Participants: Florence Forbes, Manuel Yguel and Radu Horaud</p>
        </li>
        <li id="uid19">
          <p noindent="true">Contact: Patrice Horaud</p>
        </li>
        <li id="uid20">
          <p noindent="true">URL: <ref xlink:href="https://team.inria.fr/perception/research/jrmpc/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>jrmpc/</ref></p>
        </li>
      </simplelist>
    </subsection>
    <subsection id="uid21" level="1">
      <bodyTitle>Mixcam</bodyTitle>
      <p>
        <i>Reconstruction using a mixed camera system</i>
      </p>
      <p noindent="true"><span class="smallcap" align="left">Keywords:</span> Computer vision - 3D reconstruction</p>
      <p noindent="true"><span class="smallcap" align="left">Functional Description:</span> We developed a multiple camera platform composed of both high-definition color cameras and low-resolution depth cameras. This platform combines the advantages of the two camera types. On one side, depth (time-of-flight) cameras provide coarse low-resolution 3D scene information. On the other side, depth and color cameras can be combined such as to provide high-resolution 3D scene reconstruction and high-quality rendering of textured surfaces. The software package developed during the period 2011-2014 contains the calibration of TOF cameras, alignment between TOF and color cameras, TOF-stereo fusion, and image-based rendering. These software developments were performed in collaboration with the Samsung Advanced Institute of Technology, Seoul, Korea. The multi-camera platform and the basic software modules are products of 4D Views Solutions SAS, a start-up company issued from the PERCEPTION group.</p>
      <simplelist>
        <li id="uid22">
          <p noindent="true">Participants: Clément Ménier, Georgios Evangelidis, Michel Amat, Miles Hansard, Patrice Horaud, Pierre Arquier, Quentin Pelorson, Radu Horaud, Richard Broadbridge and Soraya Arias</p>
        </li>
        <li id="uid23">
          <p noindent="true">Contact: Patrice Horaud</p>
        </li>
        <li id="uid24">
          <p noindent="true">URL: <ref xlink:href="https://team.inria.fr/perception/mixcam-project/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>mixcam-project/</ref></p>
        </li>
      </simplelist>
    </subsection>
    <subsection id="uid25" level="1">
      <bodyTitle>NaoLab</bodyTitle>
      <p>
        <i>Distributed middleware architecture for interacting with NAO</i>
      </p>
      <p noindent="true"><span class="smallcap" align="left">Functional Description:</span> This software provides a set of librairies and tools to simply the control of NAO robot from a remote machine. The main challenge is to make easy prototuping applications for NAO ising C++ and Matlab programming environments. Thus NaoLab provides a prototyping-friendly interface to retrieve sensor date (video and sound streams, odometric data...) and to control the robot actuators (head, arms, legs...) from a remote machine.This interface is available on Naoqi SDK, developed by Aldebarab company, Naoqi SDK is needed as it provides the tools to acess the embedded NAO services (low-level motor command, sensor data access...)</p>
      <simplelist>
        <li id="uid26">
          <p noindent="true">Authors: Fabien Badeig, Quentin Pelorson and Patrice Horaud</p>
        </li>
        <li id="uid27">
          <p noindent="true">Contact: Patrice Horaud</p>
        </li>
        <li id="uid28">
          <p noindent="true">URL: <ref xlink:href="https://team.inria.fr/perception/research/naolab/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>naolab/</ref></p>
        </li>
      </simplelist>
    </subsection>
    <subsection id="uid29" level="1">
      <bodyTitle>Stereo matching and recognition library</bodyTitle>
      <p><span class="smallcap" align="left">Keyword:</span> Computer vision</p>
      <p noindent="true"><span class="smallcap" align="left">Functional Description:</span> Library providing stereo matching components to rectify stereo images, to retrieve faces from left and right images, to track faces and method to recognise simple gestures</p>
      <simplelist>
        <li id="uid30">
          <p noindent="true">Participants: Jan Cech, Jordi Sanchez-Riera, Radu Horaud and Soraya Arias</p>
        </li>
        <li id="uid31">
          <p noindent="true">Contact: Soraya Arias</p>
        </li>
        <li id="uid32">
          <p noindent="true">URL: <ref xlink:href="https://code.humavips.eu/projects/stereomatch" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>code.<allowbreak/>humavips.<allowbreak/>eu/<allowbreak/>projects/<allowbreak/>stereomatch</ref></p>
        </li>
      </simplelist>
    </subsection>
    <subsection id="uid33" level="1">
      <bodyTitle>Platforms</bodyTitle>
      <subsection id="uid34" level="2">
        <bodyTitle>Audio-Visual Head Popeye+</bodyTitle>
        <p>In 2016 our audio-visual platform was upgraded from Popeye to Popeye+. Popeye+ has two high-definition cameras with a wide field of view. We also upgraded the software libraries that perform synchronized acquisition of audio signals and color images. Popeye+ has been used for several datasets.</p>
        <p noindent="true">Websites:</p>
        <p noindent="true">
          <ref xlink:href="https://team.inria.fr/perception/projects/popeye/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>projects/<allowbreak/>popeye/</ref>
        </p>
        <p noindent="true">
          <ref xlink:href="https://team.inria.fr/perception/projects/popeye-plus/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>projects/<allowbreak/>popeye-plus/</ref>
        </p>
        <p noindent="true">
          <ref xlink:href="https://team.inria.fr/perception/avtrack1/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>avtrack1/</ref>
        </p>
        <p noindent="true">
          <ref xlink:href="https://team.inria.fr/perception/avdiar/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>avdiar/</ref>
        </p>
      </subsection>
      <subsection id="uid35" level="2">
        <bodyTitle>NAO Robots</bodyTitle>
        <p>The PERCEPTION team selected the companion robot NAO for experimenting and demonstrating various audio-visual skills as well as for developing the concept of social robotics that is able to recognize human presence, to understand human gestures and voice, and to communicate by synthesizing appropriate behavior. The main challenge of our team is to enable human-robot interaction in the real world.</p>
        <object id="uid36">
          <table>
            <tr>
              <td>
                <ressource xlink:href="IMG/DSC03568.jpg" type="inline" height="149.4526pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
              </td>
              <td>
                <ressource xlink:href="IMG/DSC03567.jpg" type="inline" height="149.4526pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
              </td>
            </tr>
          </table>
          <caption>The Popeye+ audio-visual platform (left) delivers high-quality, high-resolution and wide-angle images at 30FPS. The NAO prototype used by PERCEPTION in the EARS STREP project has a twelve-channel spherical microphone array synchronized with a stereo camera pair.</caption>
        </object>
        <p>The humanoid robot NAO is manufactured by SoftBank Robotics Europe. Standing, the robot is roughly 60 cm tall, and 35cm when it is sitting. Approximately 30 cm large, NAO includes two CPUs. The first one, placed in the torso, together with the batteries, controls the motors and hence provides kinematic motions with 26 degrees of freedom. The other CPU is placed in the head and is in charge of managing the proprioceptive sensing, the communications, and the audio-visual sensors (two cameras and four microphones, in our case). NAO's on-board computing resources can be accessed either via wired or wireless communication protocols.</p>
        <p>NAO's commercially available head is equipped with two cameras that are arranged along a vertical axis: these cameras are neither synchronized nor a significant common field of view. Hence, they cannot be used in combination with stereo vision. Within the EU project HUMAVIPS, Aldebaran Robotics developed a binocular camera system that is arranged horizontally. It is therefore possible to implement stereo vision algorithms on NAO. In particular, one can take advantage of both the robot's cameras and microphones. The cameras deliver VGA sequences of image pairs at 12 FPS, while the sound card delivers the audio signals arriving from all four microphones and sampled at 48 kHz. Subsequently, Aldebaran developed a second binocular camera system to go into the head of NAO v5.</p>
        <p>In order to manage the information flow gathered by all these sensors, we implemented several middleware packages. In 2012 we implemented Robotics Services Bus (RSB) developed by the University of Bielefeld. Subsequently (2015-2016) the PERCEPTION team developed NAOLab, a middleware for hosting robotic applications in C, C++, Python and Matlab, using the computing power available with NAO, augmented with a networked PC. In 2017 we abandoned RSB and NAOLab and converted all our robotics software packages to ROS (Robotic Operating System).</p>
        <p>Websites:</p>
        <p noindent="true">
          <ref xlink:href="https://team.inria.fr/perception/nao/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>nao/</ref>
        </p>
        <p noindent="true">
          <ref xlink:href="https://team.inria.fr/perception/research/naolab/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>naolab/</ref>
        </p>
      </subsection>
    </subsection>
  </logiciels>
  <resultats id="uid37">
    <bodyTitle>New Results</bodyTitle>
    <subsection id="uid38" level="1">
      <bodyTitle>Multichannel Speech Separation and Enhancement Using the Convolutive Transfer Function</bodyTitle>
      <p>We addressed the problem of speech separation and enhancement from multichannel convolutive and noisy mixtures, <i>assuming known mixing filters</i>. We proposed to perform the speech separation and enhancement tasks in the short-time Fourier transform domain, using the convolutive transfer function (CTF) approximation <ref xlink:href="#perception-2018-bid26" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. Compared to time-domain filters, CTF has much less taps, consequently it has less near-common zeros among channels and less computational complexity. The work proposes three speech-source recovery methods, namely: (i) the multichannel inverse filtering method, i.e. the multiple input/output inverse theorem (MINT), is exploited in the CTF domain, and for the multi-source case, (ii) a beamforming-like multichannel inverse filtering method applying single source MINT and using power minimization, which is suitable whenever the source CTFs are not all known, and (iii) a constrained Lasso method, where the sources are recovered by minimizing the <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><msub><mi>ℓ</mi><mn>1</mn></msub></math></formula>-norm to impose their spectral sparsity, with the constraint that the <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><msub><mi>ℓ</mi><mn>2</mn></msub></math></formula>-norm fitting cost, between the microphone signals and the mixing model involving the unknown source signals, is less than a tolerance. The noise can be reduced by setting a tolerance onto the noise power. Experiments under various acoustic conditions are carried out to evaluate the three proposed methods. The comparison between them as well as with the baseline methods is presented.</p>
    </subsection>
    <subsection id="uid39" level="1">
      <bodyTitle>Speech Dereverberation and Noise Reduction Using the Convolutive Transfer Function</bodyTitle>
      <p>We address the problems of blind multichannel identification and equalization for <i>joint speech dereverberation and noise reduction</i>.
The standard time-domain cross-relation methods are hardly applicable for blind room impulse response identification due to the near-common zeros of the long impulse responses. We extend the cross-relation formulation to the short-time Fourier transform (STFT) domain, in which the time-domain impulse response is approximately represented by the convolutive transfer function (CTF) with much less coefficients.
For the oversampled STFT, CTFs suffer from the common zeros caused by the non-flat-top STFT window. To overcome this, we propose to identify CTFs using the STFT framework with oversampled signals and critically sampled CTFs, which is a good trade-off between the frequency aliasing of the signals and the common zeros problem of CTFs. The phases of the identified CTFs are inaccurate due to the frequency aliasing of the CTFs, and thus only their magnitudes are used. This leads to a non-negative multichannel equalization method based on a non-negative convolution model between the STFT magnitude of the source signal and the CTF magnitude. To recover the STFT magnitude of the source signal and to reduce the additive noise, the <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><msub><mi>ℓ</mi><mn>2</mn></msub></math></formula>-norm fitting error between the STFT magnitude of the microphone signals and the non-negative convolution is constrained to be less than a noise power related tolerance. Meanwhile, the <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><msub><mi>ℓ</mi><mn>1</mn></msub></math></formula>-norm of the STFT magnitude of the source signal is minimized to impose the sparsity <ref xlink:href="#perception-2018-bid27" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
      <p noindent="true">Website: <ref xlink:href="https://team.inria.fr/perception/research/ctf-dereverberation/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>ctf-dereverberation/</ref>.
</p>
    </subsection>
    <subsection id="uid40" level="1">
      <bodyTitle>Speech Enhancement with a Variational Auto-Encoder</bodyTitle>
      <p>We addressed the problem of enhancing speech signals in noisy mixtures using a source separation approach. We explored the use of neural networks as an alternative to a popular speech variance model based on supervised non-negative matrix factorization (NMF). More precisely, we use a variational auto-encoder as a speaker-independent supervised generative speech model, highlighting the conceptual similarities that this approach shares with its NMF-based counterpart. In order to be free of generalization issues regarding the noisy recording environments, we follow the approach of having a supervised model only for the target speech signal, the noise model being based on unsupervised NMF. We developed a Monte Carlo expectation-maximization algorithm for inferring the latent variables in the variational auto-encoder and estimating the unsupervised model parameters. Experiments show that the proposed method outperforms a semi-supervised NMF baseline and a state-of-the-art fully supervised deep learning approach.</p>
      <p noindent="true">Website: <ref xlink:href="https://team.inria.fr/perception/research/ieee-mlsp-2018/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>ieee-mlsp-2018/</ref>.</p>
    </subsection>
    <subsection id="uid41" level="1">
      <bodyTitle>Audio-Visual Speaker Tracking and Diarization</bodyTitle>
      <p>We are particularly interested in modeling the interaction between an intelligent device and a group of people. For that purpose we develop audio-visual person tracking methods <ref xlink:href="#perception-2018-bid28" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. As the observed persons are supposed to carry out a conversation, we also include speaker diarization into our tracking methodology.
We cast the diarization problem into a tracking formulation whereby the active speaker is detected and tracked over time. A probabilistic tracker exploits the spatial coincidence of visual and auditory observations and infers a single latent variable which represents the identity of the active speaker. Visual and auditory observations are fused using our recently developed weighted-data mixture model <ref xlink:href="#perception-2018-bid29" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, while several options for the speaking turns dynamics are fulfilled by a multi-case transition model. The modules that translate raw audio and visual data into image observations are also described in detail. The performance of the proposed method are tested on challenging datasets that are available from recent contributions which are used as baselines for comparison <ref xlink:href="#perception-2018-bid28" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
      <p noindent="true">Websites:</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/wdgmm/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>wdgmm/</ref>,</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/speakerloc/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>speakerloc/</ref>,</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/speechturndet/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>speechturndet/</ref>, and</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/avdiarization/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>avdiarization/</ref>.</p>
      <object id="uid42">
        <table>
          <tr>
            <td>
              <ressource xlink:href="IMG/avdiar.png" type="figure" width="422.73235pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
            </td>
          </tr>
        </table>
        <caption>This figure illustrates the audiovisual tracking and diarization method that we have recently developed. First row: A number is associated with each tracked person. Second row: diarization result. Third row: the ground truth diarization. Fourth row: acoustic signal recorded by one of the two microphones.</caption>
      </object>
    </subsection>
    <subsection id="uid43" level="1">
      <bodyTitle>Tracking Eye Gaze and of Visual Focus of Attention</bodyTitle>
      <p>The visual focus of attention (VFOA) has been recognized as a prominent conversational cue. We are interested in estimating and tracking the VFOAs associated with multi-party social interactions. We note that in this type of situations the participants either look at each other or at an object of interest; therefore their eyes are not always visible. Consequently both gaze and VFOA estimation cannot be based on eye detection and tracking. We propose a method that exploits the correlation between eye gaze and head movements. Both VFOA and gaze are modeled as latent variables in a Bayesian switching state-space model (also named switching Kalman filter). The proposed formulation leads to a tractable learning method and to an efficient online inference procedure that simultaneously tracks gaze and visual focus. The method is tested and benchmarked using two publicly available datasets, Vernissage and LAEO, that contain typical multi-party human-robot and human-human interactions <ref xlink:href="#perception-2018-bid30" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
      <p noindent="true">Website: <ref xlink:href="https://team.inria.fr/perception/research/eye-gaze/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>eye-gaze/</ref>.</p>
      <object id="uid44">
        <table>
          <tr>
            <td>
              <ressource xlink:href="IMG/vfoa_laeo.png" type="figure" width="422.73235pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
            </td>
          </tr>
        </table>
        <caption>This figure shows some results obtained with the <i>LAEO</i> dataset. The top row shows results obtained with coarse head orientation and the bottom row shows results obtained with fine head orientation. Head orientations are shown with red arrows. The algorithm infers gaze directions (green arrows) and VFOAs (blue circles). People looking at each others are shown with a dashed blue line.</caption>
      </object>
    </subsection>
    <subsection id="uid45" level="1">
      <bodyTitle>Variational Bayesian Inference of Multiple-Person Tracking</bodyTitle>
      <p>We addressed the problem of tracking multiple speakers using audio information or via the fusion of visual and auditory information. We proposed to exploit the complementary nature of these two modalities in order to accurately estimate smooth trajectories of the tracked persons, to deal with the partial or total absence of one of the modalities over short periods of time, and to estimate the acoustic status – either speaking or silent – of each tracked person along time, e.g. Figure <ref xlink:href="#uid4" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. We proposed to cast the problem at hand into a generative audio-visual fusion (or association) model formulated as a latent-variable temporal graphical model. This may well be viewed as the problem of maximizing the posterior joint distribution of a set of continuous and discrete latent variables given the past and current observations, which is intractable. We propose a variational inference model which amounts to approximate the joint distribution with a factorized distribution. The solutions take the form of closed-form expectation maximization procedures using Gaussian distributions <ref xlink:href="#perception-2018-bid0" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid2" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid1" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> or the von Mises distribution for circular variables <ref xlink:href="#perception-2018-bid31" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. We described in detail the inference algorithms, we evaluate their performance and we compared them with several baseline methods. These experiments show that the proposed audio and audio-visual trackers perform well in informal meetings involving a time-varying number of people.</p>
      <p noindent="true">Websites:</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/var-av-track/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>var-av-track/</ref>,</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/audiotrack-vonm/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>audiotrack-vonm/</ref>.</p>
    </subsection>
    <subsection id="uid46" level="1">
      <bodyTitle>High-Dimensional and Deep Regression</bodyTitle>
      <p>One of the most important achievements for the last years has been the development of high-dimensional to low-dimensional regression methods. The motivation for investigating this problem raised from several problems that appeared both in audio signal processing and in computer vision. Indeed, often the task in data-driven methods is to recover low-dimensional properties and associated parameterizations from high-dimensional observations. Traditionally, this can be formulated as either an unsupervised method (dimensionality reduction of manifold learning) or a supervised method (regression). We developed a learning methodology at the crossroads of these two alternatives: the output variable can be either fully observed or partially observed. This was cast into the framework of
linear-Gaussian mixture models in conjunction with the concept of inverse regression. It gave rise to several closed-form and approximate inference algorithms <ref xlink:href="#perception-2018-bid9" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The method is referred to as <i>Gaussian locally linear mapping</i>, or GLLiM.
As already mentioned, high-dimensional regression is useful in a number of data processing tasks because the sensory data often lies in high-dimensional spaces.
Each one of these tasks required a special-purpose version of our general framework. Sound-source localization was the first to benefit from our formulation. Nevertheless, the sparse nature of speech spectrograms required the development of a GLLiM version that is able to with full-spectrum sounds and to test with sparse-spectrum ones <ref xlink:href="#perception-2018-bid4" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. This could be immediately applied to audio-visual alignment and to sound-source separation and localization <ref xlink:href="#perception-2018-bid10" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
      <p>In conjunction with our computer vision work, high-dimensional regression is a very useful methodology since visual features, obtained either by hand-crafted feature extraction methods or using convolutional neural networks, lie in high-dimensional spaces. Such properties as object pose lie in low-dimensional spaces and must be extracted from features. We took such an approach and proposed a head pose estimator <ref xlink:href="#perception-2018-bid32" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. Visual tracking can also benefit from GLLiM. Indeed, it is not practical to track objects based on high-dimensional features. We therefore combined GLLiM with switching linear dynamic systems. In 2018 we proposed a robust deep regression method <ref xlink:href="#perception-2018-bid33" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. In parallel we thoroughly benchmarked and analyzed deep regression tasks using several CNN architectures <ref xlink:href="#perception-2018-bid34" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
    </subsection>
    <subsection id="uid47" level="1">
      <bodyTitle>Human-Robot Interaction</bodyTitle>
      <p>Audio-visual fusion raises interesting problems whenever it is implemented onto a robot. Robotic platforms have their own hardware and software constraints. In addition, commercialized robots have economical constraints which leads to the use of cheap components. A robot must be reactive to changes in its environment and hence it must take fast decisions. This often implies that most of the computing resources must be onboard of the robot.</p>
      <p>Over the last decade we have tried to do our best to take these constraints into account. Starting from our scientific developments, we put a lot of efforts into robotics implementations. For example, the audio-visual fusion method described in <ref xlink:href="#perception-2018-bid7" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> used a specific robotic middleware that allowed fast communication between the robot and an external computing unit. Subsequently we developed a powerful software package that enables distributed computing. We also put a lot of emphasis on the implementation of low-level audio and visual processing algorithms.
In particular, our single- and multiple audio source methods were implemented in real time onto the humanoid robot NAO <ref xlink:href="#perception-2018-bid35" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid36" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The multiple person tracker <ref xlink:href="#perception-2018-bid3" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> was also implemented onto our robotic platforms <ref xlink:href="#perception-2018-bid37" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, e.g. Figure <ref xlink:href="#uid48" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
      <object id="uid48">
        <table>
          <tr>
            <td>
              <ressource xlink:href="IMG/nao_motvs.png" type="figure" width="341.6013pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
            </td>
          </tr>
        </table>
        <caption>The multi-person tracking method is combined with a visual servoing module. The latter estimates the optimal robot commands and the expected impact of the tracked
person locations. The multi-person tracking module refines the locations of the persons with the new observations and the information provided by the visual
servoing.</caption>
      </object>
      <p>More recently, we investigated the use of reinforcement learning (RL) as an alternative to sensor-based robot control <ref xlink:href="#perception-2018-bid38" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid39" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The robotic task consists of turning the robot head (gaze control) towards speaking people. The method is more general in spirit than visual (or audio) servoing because it can handle an arbitrary number of speaking or non speaking persons and it can improve its behavior online, as the robot experiences new situations. An overview of the proposed method is shown in Fig. <ref xlink:href="#uid49" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>. The reinforcement learning formulation enables a robot to learn where to look for people and to favor speaking people via a trial-and-error strategy.</p>
      <object id="uid49">
        <table>
          <tr>
            <td>
              <ressource xlink:href="IMG/pipeline.png" type="figure" width="422.73235pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
            </td>
          </tr>
        </table>
        <caption>Overview of the proposed deep RL method for controlling the gaze of a robot. At each time index <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><mi>t</mi></math></formula>, audio and visual data are represented as binary maps which, together with motor positions, form the set of observations <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><msub><mi>𝐎</mi><mi>t</mi></msub></math></formula>. A motor action <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><msub><mi>A</mi><mi>t</mi></msub></math></formula> (rotate the head left, right, up, down, or stay still) is selected based on past and present observations via maximization of current and future rewards. The rewards <formula type="inline"><math xmlns="http://www.w3.org/1998/Math/MathML" overflow="scroll"><mi>R</mi></math></formula> are based on the number of visible persons as well as on the presence of speech sources in the camera field of view. We use a deep Q-network (DQN) model that can be learned both off-line and on-line. Please consult <ref xlink:href="#perception-2018-bid38" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid39" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> for further details.</caption>
      </object>
      <p>Past, present and future HRI developments require datasets for training, validation, test as well as for benchmarking. HRI datasets are challenging because it is not easy to record realistic interactions between a robot and users. RL avoids systematic recourse to annotated datasets for training. In <ref xlink:href="#perception-2018-bid38" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, <ref xlink:href="#perception-2018-bid39" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/> we proposed the use of a simulated environment for pre-training the RL parameters, thus avoiding spending hours of tedious interaction.</p>
      <p noindent="true">Websites:</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/deep-rl-for-gaze-control/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>deep-rl-for-gaze-control/</ref>,</p>
      <p noindent="true"><ref xlink:href="https://team.inria.fr/perception/research/mot-servoing/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>mot-servoing/</ref>.
</p>
    </subsection>
    <subsection id="uid50" level="1">
      <bodyTitle>Generation of Diverse Behavioral Data</bodyTitle>
      <p>We target the automatic generation of visual data depicting human behavior, and in particular how to design a method able to learn the generation of <i>data diversity</i>. In particular, we focus on smiles, because each smile is unique: one person surely smiles in different ways (e.g. closing/opening the eyes or mouth). We wonder if given one input image of a neutral face, we can generate multiple smile videos with distinctive characteristics. To tackle this one-to-many video generation problem, we propose a novel deep learning architecture named Conditional MultiMode Network (CMM-Net). To better encode the dynamics of facial expressions, CMM-Net explicitly exploits facial landmarks for generating smile sequences. Specifically, a variational auto-encoder is used to learn a facial landmark embedding. This single embedding is then exploited by a conditional recurrent network which generates a landmark embedding sequence conditioned on a specific expression (e.g. spontaneous smile), implemented as a Conditional LSTM. Next, the generated landmark embeddings are fed into a multi-mode recurrent landmark generator, producing a set of landmark sequences still associated to the given smile class but clearly distinct from each other, we call that a Multi-Mode LSTM. Finally, these landmark sequences are translated into face videos. Our experimental results, see Figure <ref xlink:href="#uid51" location="intern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, demonstrate the effectiveness of our CMM-Net in generating realistic videos of multiple smile expressions <ref xlink:href="#perception-2018-bid40" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
      <object id="uid51">
        <table>
          <tr>
            <td>
              <ressource xlink:href="IMG/smile-want.png" type="figure" width="422.73235pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
            </td>
          </tr>
        </table>
        <caption>Multi-mode generation example with a sequence: landmarks (left) and associated face images (right) after the landmark-to-image decoding step based on Variational Auto-Encoders. The rows correspond to the original sequence (first), output of the Conditional LSTM (second), and output of the Multi-Mode LSTM (last three rows).</caption>
      </object>
    </subsection>
    <subsection id="uid52" level="1">
      <bodyTitle>Registration of Multiple Point Sets</bodyTitle>
      <p>We have also addressed the rigid registration problem of multiple 3D point sets. While the vast majority of state-of-the-art techniques build on pairwise registration, we proposed a generative model that explains jointly registered multiple sets: back-transformed points are considered realizations of a single Gaussian mixture model (GMM) whose means play the role of the (unknown) scene points. Under this assumption, the joint registration problem is cast into a probabilistic clustering framework. We formally derive an expectation-maximization procedure that robustly estimates both the GMM parameters and the rigid transformations that map each individual cloud onto an under-construction reference set, that is, the GMM means. GMM variances carry rich information as well, thus leading to a noise- and outlier-free scene model as a by-product. A second version of the algorithm is also proposed whereby newly captured sets can be registered online. A thorough discussion and validation on challenging data-sets against several state-of-the-art methods confirm the potential of the proposed model for jointly registering real depth data <ref xlink:href="#perception-2018-bid41" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>.</p>
      <p noindent="true">Website: <ref xlink:href="https://team.inria.fr/perception/research/jrmpc/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>research/<allowbreak/>jrmpc/</ref></p>
      <object id="uid53">
        <table>
          <tr>
            <td>
              <ressource xlink:href="IMG/jrmpc.png" type="figure" width="422.73235pt" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest" media="WEB"/>
            </td>
          </tr>
        </table>
        <caption>Integrated point clouds from the joint registration of 10 TOF images that record a static scene (EXBI data-set). <i>Top</i>: color images that roughly show the scene content of each range image (occlusions due to cameras baseline may cause texture artefacts). <i>Bottom:</i> front-view and top-view of integrated sets after joint registration. The results obtained with the proposed method (JRMPC-B) are compared with several other methods.</caption>
      </object>
    </subsection>
  </resultats>
  <partenariat id="uid54">
    <bodyTitle>Partnerships and Cooperations</bodyTitle>
    <subsection id="uid55" level="1">
      <bodyTitle>European Initiatives</bodyTitle>
      <subsection id="uid56" level="2">
        <bodyTitle>VHIA</bodyTitle>
        <sanspuceslist>
          <li id="uid57">
            <p noindent="true">Title: Vision and Hearing in Action</p>
          </li>
          <li id="uid58">
            <p noindent="true">EU framework: FP7</p>
          </li>
          <li id="uid59">
            <p noindent="true">Type: ERC Advanced Grant</p>
          </li>
          <li id="uid60">
            <p noindent="true">Duration: February 2014 - January 2019</p>
          </li>
          <li id="uid61">
            <p noindent="true">Coordinator: Inria</p>
          </li>
          <li id="uid62">
            <p noindent="true">Inria contact: Radu Horaud</p>
          </li>
          <li id="uid63">
            <p noindent="true">'The objective of VHIA is to elaborate a holistic computational paradigm of perception and of perception-action loops. We plan to develop a completely novel twofold approach: (i) learn from mappings between auditory/visual inputs and structured outputs, and from sensorimotor contingencies, and (ii) execute perception-action interaction cycles in the real world with a humanoid robot. VHIA will achieve a unique fine coupling between methodological findings and proof-of-concept implementations using the consumer humanoid NAO manufactured in Europe. The proposed multimodal approach is in strong contrast with current computational paradigms influenced by unimodal biological theories. These theories have hypothesized a modular view, postulating quasi-independent and parallel perceptual pathways in the brain. VHIA will also take a radically different view than today's audiovisual fusion models that rely on clean-speech signals and on accurate frontal-images of faces; These models assume that videos and sounds are recorded with hand-held or head-mounted sensors, and hence there is a human in the loop who intentionally supervises perception and interaction. Our approach deeply contradicts the belief that complex and expensive humanoids (often manufactured in Japan) are required to implement research ideas. VHIA's methodological program addresses extremely difficult issues: how to build a joint audiovisual space from heterogeneous, noisy, ambiguous and physically different visual and auditory stimuli, how to model seamless interaction, how to deal with high-dimensional input data, and how to achieve robust and efficient human-humanoid communication tasks through a well-thought tradeoff between offline training and online execution. VHIA bets on the high-risk idea that in the next decades, social robots will have a considerable economical impact, and there will be millions of humanoids, in our homes, schools and offices, which will be able to naturally communicate with us.</p>
            <p noindent="true">Website: <ref xlink:href="https://team.inria.fr/perception/projects/erc-vhia/" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>team.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>perception/<allowbreak/>projects/<allowbreak/>erc-vhia/</ref></p>
          </li>
        </sanspuceslist>
      </subsection>
      <subsection id="uid64" level="2">
        <bodyTitle>VHIALab</bodyTitle>
        <sanspuceslist>
          <li id="uid65">
            <p noindent="true">Title: Vision and Hearing in Action Laboratory</p>
          </li>
          <li id="uid66">
            <p noindent="true">EU framework: H2020</p>
          </li>
          <li id="uid67">
            <p noindent="true">Type: ERC Proof of Concept</p>
          </li>
          <li id="uid68">
            <p noindent="true">Duration: February 2018 - January 2019</p>
          </li>
          <li id="uid69">
            <p noindent="true">Coordinator: Inria</p>
          </li>
          <li id="uid70">
            <p noindent="true">Inria contact: Radu Horaud</p>
          </li>
          <li id="uid71">
            <p noindent="true">The objective of VHIALab is the development and commercialization of software packages enabling a robot companion to easily and naturally interact with people. The methodologies developed in ERC VHIA propose state of the art solutions to human-robot interaction (HRI) problems in a general setting and based on audio-visual information. The ambitious goal of VHIALab will be to build software packages based on VHIA, thus opening the door to commercially available multi-party multi-modal human-robot interaction. The methodology investigated in VHIA may well be viewed as a generalization of existing single-user spoken dialog systems. VHIA enables a robot (i) to detect and to locate speaking persons, (ii) to track several persons over time, (iii) to recognize their behavior, and (iv) to extract the speech signal of each person for subsequent speech recognition and face-to-face dialog. These methods will be turned into software packages compatible with a large variety of companion robots. VHIALab will add a strong valorization potential to VHIA by addressing emerging and new market sectors. Industrial collaborations set up in VHIA will be strengthened.</p>
          </li>
        </sanspuceslist>
      </subsection>
    </subsection>
    <subsection id="uid72" level="1">
      <bodyTitle>International Research Visitors</bodyTitle>
      <subsection id="uid73" level="2">
        <bodyTitle>Visits of International Scientists</bodyTitle>
        <simplelist>
          <li id="uid74">
            <p noindent="true">Professor Sharon Gannot, Bar Ilan University, Tel Aviv, Israel.</p>
          </li>
          <li id="uid75">
            <p noindent="true">Professor Tomislav Pribanic, University of Zagreb, Zagreb, Croatia.</p>
          </li>
          <li id="uid76">
            <p noindent="true">Doctor Christine Evers, Imperial College, London, United Kingdom.</p>
          </li>
        </simplelist>
      </subsection>
    </subsection>
  </partenariat>
  <diffusion id="uid77">
    <bodyTitle>Dissemination</bodyTitle>
    <subsection id="uid78" level="1">
      <bodyTitle>Promoting Scientific Activities</bodyTitle>
      <subsection id="uid79" level="2">
        <bodyTitle>Scientific Events Organisation</bodyTitle>
        <subsection id="uid80" level="3">
          <bodyTitle>Member of the Organizing Committees</bodyTitle>
          <p>Xavier Alameda-Pineda organized several workshops in conjunction with IEEE CVPR'18, ECCV'18, and ACM Multimedia'18.</p>
        </subsection>
        <subsection id="uid81" level="3">
          <bodyTitle>Reviewer</bodyTitle>
          <p>Xavier Alameda-PIneda was a reviewer for IEEE CVPR'18, NIPS'18, IEEE ICASSP'18, ACM Multimedia'18 and IEEE ICRA'18.</p>
        </subsection>
      </subsection>
      <subsection id="uid82" level="2">
        <bodyTitle>Journal</bodyTitle>
        <subsection id="uid83" level="3">
          <bodyTitle>Member of the Editorial Boards</bodyTitle>
          <p>Radu Horaud is associated editor for the International Journal of Computer Vision and for the IEEE Robotics and Automation Letters.</p>
        </subsection>
        <subsection id="uid84" level="3">
          <bodyTitle>Guest Editor</bodyTitle>
          <p>Xavier Alameda-Pineda was co-guest editor of a special issue of the ACM Transactions on Multimedia Computing Communications and Applications on “Multimodal Understanding of Social, Affective, and Subjective Attributes".</p>
        </subsection>
      </subsection>
      <subsection id="uid85" level="2">
        <bodyTitle>Invited Talks</bodyTitle>
        <simplelist>
          <li id="uid86">
            <p noindent="true">Radu Horaud gave an invited talk at the Multimodal Machine Perception Workshop, Google, San Francisco, and at SRI International, Menlo Park, USA, on "Audio-Visual Machine Perception for Human-Robot Interaction".</p>
          </li>
          <li id="uid87">
            <p noindent="true">Xavier Alameda-Pineda was invited to give a seminar at the University in May 2018 on “Audio-Visual Multiple Speaker Tracking with Robotic Platforms", and</p>
          </li>
          <li id="uid88">
            <p noindent="true">Xavier Alameda-Pineda gave an invited talk at the SOUND Workshop at Bar-Ilan University, Israel, December 2018 on “Multi-modal Automatic Detection of Social Attractors in Crowded Meetings".</p>
          </li>
        </simplelist>
      </subsection>
    </subsection>
    <subsection id="uid89" level="1">
      <bodyTitle>Teaching - Supervision - Juries</bodyTitle>
      <subsection id="uid90" level="2">
        <bodyTitle>Teaching</bodyTitle>
        <simplelist>
          <li id="uid91">
            <p noindent="true">Laurent Girin is professor at Grenoble National Polytechnic Institute (G-INP) where he teaches signal processing and machine learning on the basis of a full professor service (192 hours/year)</p>
          </li>
          <li id="uid92">
            <p noindent="true">Xavier Alameda-Pineda is involved with the M2 course of the MSIAM Masters Program: “Fundamentals of probabilistic data mining, modeling seminars and projects”, for the practical sessions. Xavier is also preparing a doctoral course on “Learning with Multi-Modal data for Scene Understanding and Human-Robot Interaction” to be taught in spring 2019.</p>
          </li>
        </simplelist>
      </subsection>
      <subsection id="uid93" level="2">
        <bodyTitle>Supervision</bodyTitle>
        <simplelist>
          <li id="uid94">
            <p noindent="true">Radu Horaud has supervised the following PhD students: Israel Dejene-Gebru <ref xlink:href="#perception-2018-bid42" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, Stéphane Lathuilière <ref xlink:href="#perception-2018-bid43" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, Benoît Massé <ref xlink:href="#perception-2018-bid44" location="biblio" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest"/>, Yutong Ban, Guillaume Delorme and Sylvain Guy.</p>
          </li>
          <li id="uid95">
            <p noindent="true">Xavier Alameda-PIneda has co-supervised Israel Dejene-Gebru, Yutong Ban, Guillaume Delorme and has supervised Yihong Xu.</p>
          </li>
        </simplelist>
      </subsection>
      <subsection id="uid96" level="2">
        <bodyTitle>Juries</bodyTitle>
        <p>Xavier Alameda-Pineda was reviewer and examiner of the PhD dissertations of Wei Wang (now post-doctoral fellow at EPFL) and of Dr. Dan Xu (now post-doctoral fellow at U. Oxford), both at University of Trento, Italy.</p>
      </subsection>
    </subsection>
  </diffusion>
  <biblio id="bibliography" html="bibliography" numero="10" titre="Bibliography">
    
    <biblStruct id="perception-2018-bid8" type="article" rend="refer" n="refercite:alamedapineda:hal-00975293">
      <identifiant type="doi" value="10.1109/TASLP.2014.2317989"/>
      <identifiant type="hal" value="hal-00975293"/>
      <analytic>
        <title level="a">A Geometric Approach to Sound Source Localization from Time-Delay Estimates</title>
        <author>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Audio, Speech and Language Processing</title>
        <imprint>
          <biblScope type="volume">22</biblScope>
          <biblScope type="number">6</biblScope>
          <dateStruct>
            <month>June</month>
            <year>2014</year>
          </dateStruct>
          <biblScope type="pages">1082-1095</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-00975293" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00975293</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid7" type="article" rend="refer" n="refercite:alamedapineda:hal-00990766">
      <identifiant type="doi" value="10.1177/0278364914548050"/>
      <identifiant type="hal" value="hal-00990766"/>
      <analytic>
        <title level="a">Vision-Guided Robot Hearing</title>
        <author>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of Robotics Research</title>
        <imprint>
          <biblScope type="volume">34</biblScope>
          <biblScope type="number">4-5</biblScope>
          <dateStruct>
            <month>April</month>
            <year>2015</year>
          </dateStruct>
          <biblScope type="pages">437-456</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-00990766" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00990766</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid58" type="article" rend="refer" n="refercite:andreff:hal-00520167">
      <identifiant type="hal" value="hal-00520167"/>
      <analytic>
        <title level="a">Visual Servoing from Lines</title>
        <author>
          <persName>
            <foreName>Nicolas</foreName>
            <surname>Andreff</surname>
            <initial>N.</initial>
          </persName>
          <persName>
            <foreName>Bernard</foreName>
            <surname>Espiau</surname>
            <initial>B.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of Robotics Research</title>
        <imprint>
          <biblScope type="volume">21</biblScope>
          <biblScope type="number">8</biblScope>
          <dateStruct>
            <year>2002</year>
          </dateStruct>
          <biblScope type="pages">679–700</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00520167" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00520167</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid3" type="article" rend="refer" n="refercite:ba:hal-01349763">
      <identifiant type="doi" value="10.1016/j.cviu.2016.07.006"/>
      <identifiant type="hal" value="hal-01349763"/>
      <analytic>
        <title level="a">An On-line Variational Bayesian Model for Multi-Person Tracking from Cluttered Scenes</title>
        <author>
          <persName>
            <foreName>Sileye</foreName>
            <surname>Ba</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName>
            <foreName>Alessio</foreName>
            <surname>Xompero</surname>
            <initial>A.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Computer Vision and Image Understanding</title>
        <imprint>
          <biblScope type="volume">153</biblScope>
          <dateStruct>
            <month>December</month>
            <year>2016</year>
          </dateStruct>
          <biblScope type="pages">64–76</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01349763" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01349763</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid37" type="inproceedings" rend="refer" n="refercite:ban:hal-01542987">
      <identifiant type="hal" value="hal-01542987"/>
      <analytic>
        <title level="a">Tracking a Varying Number of People with a Visually-Controlled Robotic Head</title>
        <author>
          <persName key="perception-2018-idp164448">
            <foreName>Yutong</foreName>
            <surname>Ban</surname>
            <initial>Y.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName>
            <foreName>Fabien</foreName>
            <surname>Badeig</surname>
            <initial>F.</initial>
          </persName>
          <persName>
            <foreName>Sileye</foreName>
            <surname>Ba</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">IEEE/RSJ International Conference on Intelligent Robots and Systems</title>
        <loc>Vancouver, Canada</loc>
        <imprint>
          <dateStruct>
            <month>September</month>
            <year>2017</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01542987" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01542987</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid24" type="article" rend="refer" n="refercite:cuzzolin:hal-01053737">
      <identifiant type="doi" value="10.1007/s11263-014-0754-0"/>
      <identifiant type="hal" value="hal-01053737"/>
      <analytic>
        <title level="a">Robust Temporally Coherent Laplacian Protrusion Segmentation of 3D Articulated Bodies</title>
        <author>
          <persName>
            <foreName>Fabio</foreName>
            <surname>Cuzzolin</surname>
            <initial>F.</initial>
          </persName>
          <persName>
            <foreName>Diana</foreName>
            <surname>Mateus</surname>
            <initial>D.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of Computer Vision</title>
        <imprint>
          <biblScope type="volume">112</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <month>March</month>
            <year>2015</year>
          </dateStruct>
          <biblScope type="pages">43-70</biblScope>
          <ref xlink:href="https://hal.archives-ouvertes.fr/hal-01053737" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>archives-ouvertes.<allowbreak/>fr/<allowbreak/>hal-01053737</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid10" type="article" rend="refer" n="refercite:deleforge:hal-00960796">
      <identifiant type="doi" value="10.1142/S0129065714400036"/>
      <identifiant type="hal" value="hal-00960796"/>
      <analytic>
        <title level="a">Acoustic Space Learning for Sound-Source Separation and Localization on Binaural Manifolds</title>
        <author>
          <persName key="multispeech-2018-idp127968">
            <foreName>Antoine</foreName>
            <surname>Deleforge</surname>
            <initial>A.</initial>
          </persName>
          <persName key="mistis-2018-idp124336">
            <foreName>Florence</foreName>
            <surname>Forbes</surname>
            <initial>F.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of Neural Systems</title>
        <imprint>
          <biblScope type="volume">25</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <month>February</month>
            <year>2015</year>
          </dateStruct>
          <biblScope type="pages">21</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-00960796" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00960796</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid9" type="article" rend="refer" n="refercite:deleforge:hal-00863468">
      <identifiant type="doi" value="10.1007/s11222-014-9461-5"/>
      <identifiant type="hal" value="hal-00863468"/>
      <analytic>
        <title level="a">High-Dimensional Regression with Gaussian Mixtures and Partially-Latent Response Variables</title>
        <author>
          <persName key="multispeech-2018-idp127968">
            <foreName>Antoine</foreName>
            <surname>Deleforge</surname>
            <initial>A.</initial>
          </persName>
          <persName key="mistis-2018-idp124336">
            <foreName>Florence</foreName>
            <surname>Forbes</surname>
            <initial>F.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Statistics and Computing</title>
        <imprint>
          <biblScope type="volume">25</biblScope>
          <biblScope type="number">5</biblScope>
          <dateStruct>
            <month>September</month>
            <year>2015</year>
          </dateStruct>
          <biblScope type="pages">893-911</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-00863468" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00863468</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid4" type="article" rend="refer" n="refercite:deleforge:hal-01112834">
      <identifiant type="doi" value="10.1109/TASLP.2015.2405475"/>
      <identifiant type="hal" value="hal-01112834"/>
      <analytic>
        <title level="a">Co-Localization of Audio Sources in Images Using Binaural Features and Locally-Linear Regression</title>
        <author>
          <persName key="multispeech-2018-idp127968">
            <foreName>Antoine</foreName>
            <surname>Deleforge</surname>
            <initial>A.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName>
            <foreName>Yoav Y.</foreName>
            <surname>Schechner</surname>
            <initial>Y. Y.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Audio, Speech and Language Processing</title>
        <imprint>
          <biblScope type="volume">23</biblScope>
          <biblScope type="number">4</biblScope>
          <dateStruct>
            <month>April</month>
            <year>2015</year>
          </dateStruct>
          <biblScope type="pages">718-731</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01112834" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01112834</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid32" type="article" rend="refer" n="refercite:drouard:hal-01413406">
      <identifiant type="doi" value="10.1109/TIP.2017.2654165"/>
      <identifiant type="hal" value="hal-01413406"/>
      <analytic>
        <title level="a">Robust Head-Pose Estimation Based on Partially-Latent Mixture of Linear Regressions</title>
        <author>
          <persName>
            <foreName>Vincent</foreName>
            <surname>Drouard</surname>
            <initial>V.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName key="multispeech-2018-idp127968">
            <foreName>Antoine</foreName>
            <surname>Deleforge</surname>
            <initial>A.</initial>
          </persName>
          <persName>
            <foreName>Silèye</foreName>
            <surname>Ba</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Georgios</foreName>
            <surname>Evangelidis</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes">
        <title level="j">IEEE Transactions on Image Processing</title>
        <imprint>
          <biblScope type="volume">26</biblScope>
          <biblScope type="number">3</biblScope>
          <dateStruct>
            <month>March</month>
            <year>2017</year>
          </dateStruct>
          <biblScope type="pages">1428 - 1440</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01413406" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01413406</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid14" type="article" rend="refer" n="refercite:evangelidis:hal-01110031">
      <identifiant type="doi" value="10.1109/TPAMI.2015.2400465"/>
      <identifiant type="hal" value="hal-01110031"/>
      <analytic>
        <title level="a">Fusion of Range and Stereo Data for High-Resolution Scene-Modeling</title>
        <author>
          <persName>
            <foreName>Georgios</foreName>
            <surname>Evangelidis</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">37</biblScope>
          <biblScope type="number">11</biblScope>
          <dateStruct>
            <month>November</month>
            <year>2015</year>
          </dateStruct>
          <biblScope type="pages">2178 - 2192</biblScope>
          <ref xlink:href="https://hal.archives-ouvertes.fr/hal-01110031" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>archives-ouvertes.<allowbreak/>fr/<allowbreak/>hal-01110031</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid29" type="article" rend="refer" n="refercite:gebru:hal-01261374">
      <identifiant type="doi" value="10.1109/TPAMI.2016.2522425"/>
      <identifiant type="hal" value="hal-01261374"/>
      <analytic>
        <title level="a">EM Algorithms for Weighted-Data Clustering with Application to Audio-Visual Scene Analysis</title>
        <author>
          <persName>
            <foreName>Israel Dejene</foreName>
            <surname>Gebru</surname>
            <initial>I. D.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="mistis-2018-idp124336">
            <foreName>Florence</foreName>
            <surname>Forbes</surname>
            <initial>F.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">38</biblScope>
          <biblScope type="number">12</biblScope>
          <dateStruct>
            <month>December</month>
            <year>2016</year>
          </dateStruct>
          <biblScope type="pages">2402 - 2415</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01261374" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01261374</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid20" type="article" rend="refer" n="refercite:hansard:hal-01059891">
      <identifiant type="doi" value="10.1016/j.cviu.2014.09.001"/>
      <identifiant type="hal" value="hal-01059891"/>
      <analytic>
        <title level="a">Cross-Calibration of Time-of-flight and Colour Cameras</title>
        <author>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Georgios</foreName>
            <surname>Evangelidis</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Quentin</foreName>
            <surname>Pelorson</surname>
            <initial>Q.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Computer Vision and Image Understanding</title>
        <imprint>
          <biblScope type="volume">134</biblScope>
          <dateStruct>
            <month>April</month>
            <year>2015</year>
          </dateStruct>
          <biblScope type="pages">105-115</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01059891" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01059891</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid19" type="article" rend="refer" n="refercite:hansard:hal-00936333">
      <identifiant type="doi" value="10.1016/j.cviu.2014.01.007"/>
      <identifiant type="hal" value="hal-00936333"/>
      <analytic>
        <title level="a">Automatic Detection of Calibration Grids in Time-of-Flight Images</title>
        <author>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName>
            <foreName>Michel</foreName>
            <surname>Amat</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Georgios</foreName>
            <surname>Evangelidis</surname>
            <initial>G.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Computer Vision and Image Understanding</title>
        <imprint>
          <biblScope type="volume">121</biblScope>
          <dateStruct>
            <month>April</month>
            <year>2014</year>
          </dateStruct>
          <biblScope type="pages">108-118</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-00936333" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00936333</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid57" type="article" rend="refer" n="refercite:hansard:inria-00435548">
      <identifiant type="doi" value="10.1364/JOSAA.25.002357"/>
      <identifiant type="hal" value="inria-00435548"/>
      <analytic>
        <title level="a">Cyclopean geometry of binocular vision</title>
        <author>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Journal of the Optical Society of America A</title>
        <imprint>
          <biblScope type="volume">25</biblScope>
          <biblScope type="number">9</biblScope>
          <dateStruct>
            <month>September</month>
            <year>2008</year>
          </dateStruct>
          <biblScope type="pages">2357-2369</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00435548" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00435548</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid11" type="article" rend="refer" n="refercite:hansard:inria-00435549">
      <identifiant type="doi" value="10.1109/TSMCB.2009.2024211"/>
      <identifiant type="hal" value="inria-00435549"/>
      <analytic>
        <title level="a">Cyclorotation Models for Eyes and Cameras</title>
        <author>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Systems, Man, and Cybernetics, Part B: Cybernetics</title>
        <imprint>
          <biblScope type="volume">40</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <month>March</month>
            <year>2010</year>
          </dateStruct>
          <biblScope type="pages">151-161</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00435549" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00435549</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid13" type="article" rend="refer" n="refercite:hansard:inria-00590266">
      <identifiant type="doi" value="10.1162/NECO_a_00163"/>
      <identifiant type="hal" value="inria-00590266"/>
      <analytic>
        <title level="a">A Differential Model of the Complex Cell</title>
        <author>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Neural Computation</title>
        <imprint>
          <biblScope type="volume">23</biblScope>
          <biblScope type="number">9</biblScope>
          <dateStruct>
            <month>September</month>
            <year>2011</year>
          </dateStruct>
          <biblScope type="pages">2324-2357</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00590266" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00590266</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid18" type="book" rend="refer" n="refercite:hansard:hal-00725654">
      <identifiant type="hal" value="hal-00725654"/>
      <monogr x-international-audience="yes">
        <title level="m">Time of Flight Cameras: Principles, Methods, and Applications</title>
        <title level="s">Springer Briefs in Computer Science</title>
        <author>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Seungkyu</foreName>
            <surname>Lee</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Ouk</foreName>
            <surname>Choi</surname>
            <initial>O.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
        <imprint>
          <publisher>
            <orgName>Springer</orgName>
          </publisher>
          <dateStruct>
            <month>October</month>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">95</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00725654" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00725654</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid59" type="article" rend="refer" n="refercite:horaud:inria-00590127">
      <identifiant type="doi" value="10.1109/34.895977"/>
      <identifiant type="hal" value="inria-00590127"/>
      <analytic>
        <title level="a">Stereo Calibration from Rigid Motions</title>
        <author>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName>
            <foreName>Gabriela</foreName>
            <surname>Csurka</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>David</foreName>
            <surname>Demirdjian</surname>
            <initial>D.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">22</biblScope>
          <biblScope type="number">12</biblScope>
          <dateStruct>
            <month>December</month>
            <year>2000</year>
          </dateStruct>
          <biblScope type="pages">1446–1452</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00590127" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00590127</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid23" type="article" rend="refer" n="refercite:horaud:inria-00590265">
      <identifiant type="doi" value="10.1109/TPAMI.2010.94"/>
      <identifiant type="hal" value="inria-00590265"/>
      <analytic>
        <title level="a">Rigid and Articulated Point Registration with Expectation Conditional Maximization</title>
        <author>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName key="mistis-2018-idp124336">
            <foreName>Florence</foreName>
            <surname>Forbes</surname>
            <initial>F.</initial>
          </persName>
          <persName>
            <foreName>Manuel</foreName>
            <surname>Yguel</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Guillaume</foreName>
            <surname>Dewaele</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Jian</foreName>
            <surname>Zhang</surname>
            <initial>J.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">33</biblScope>
          <biblScope type="number">3</biblScope>
          <dateStruct>
            <month>March</month>
            <year>2011</year>
          </dateStruct>
          <biblScope type="pages">587-602</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00590265" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00590265</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid22" type="article" rend="refer" n="refercite:horaud:inria-00446898">
      <identifiant type="doi" value="10.1109/TPAMI.2008.108"/>
      <identifiant type="hal" value="inria-00446898"/>
      <analytic>
        <title level="a">Human Motion Tracking by Registering an Articulated Surface to 3-D Points and Normals</title>
        <author>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName>
            <foreName>Matti</foreName>
            <surname>Niskanen</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Guillaume</foreName>
            <surname>Dewaele</surname>
            <initial>G.</initial>
          </persName>
          <persName key="morpheo-2018-idp143488">
            <foreName>Edmond</foreName>
            <surname>Boyer</surname>
            <initial>E.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">31</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <month>January</month>
            <year>2009</year>
          </dateStruct>
          <biblScope type="pages">158-163</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00446898" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00446898</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid6" type="article" rend="refer" n="refercite:khalidov:inria-00590267">
      <identifiant type="doi" value="10.1162/NECO_a_00074"/>
      <identifiant type="hal" value="inria-00590267"/>
      <analytic>
        <title level="a">Conjugate Mixture Models for Clustering Multimodal Data</title>
        <author>
          <persName>
            <foreName>Vasil</foreName>
            <surname>Khalidov</surname>
            <initial>V.</initial>
          </persName>
          <persName key="mistis-2018-idp124336">
            <foreName>Florence</foreName>
            <surname>Forbes</surname>
            <initial>F.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Neural Computation</title>
        <imprint>
          <biblScope type="volume">23</biblScope>
          <biblScope type="number">2</biblScope>
          <dateStruct>
            <month>February</month>
            <year>2011</year>
          </dateStruct>
          <biblScope type="pages">517-557</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00590267" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00590267</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid21" type="article" rend="refer" n="refercite:knossow:inria-00590247">
      <identifiant type="doi" value="10.1007/s11263-007-0116-2"/>
      <identifiant type="hal" value="inria-00590247"/>
      <analytic>
        <title level="a">Human Motion Tracking with a Kinematic Parameterization of Extremal Contours</title>
        <author>
          <persName>
            <foreName>David</foreName>
            <surname>Knossow</surname>
            <initial>D.</initial>
          </persName>
          <persName key="imagine-2018-idp117904">
            <foreName>Remi</foreName>
            <surname>Ronfard</surname>
            <initial>R.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of Computer Vision</title>
        <imprint>
          <biblScope type="volume">79</biblScope>
          <biblScope type="number">3</biblScope>
          <dateStruct>
            <month>September</month>
            <year>2008</year>
          </dateStruct>
          <biblScope type="pages">247-269</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00590247" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00590247</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid56" type="article" rend="refer" n="refercite:kounadesbastian:hal-01301762">
      <identifiant type="doi" value="10.1109/TASLP.2016.2554286"/>
      <identifiant type="hal" value="hal-01301762"/>
      <analytic>
        <title level="a">A Variational EM Algorithm for the Separation of Time-Varying Convolutive Audio Mixtures</title>
        <author>
          <persName>
            <foreName>Dionyssos</foreName>
            <surname>Kounades-Bastian</surname>
            <initial>D.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE/ACM Transactions on Audio, Speech and Language Processing</title>
        <imprint>
          <biblScope type="volume">24</biblScope>
          <biblScope type="number">8</biblScope>
          <dateStruct>
            <month>August</month>
            <year>2016</year>
          </dateStruct>
          <biblScope type="pages">1408-1423</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01301762" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01301762</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid35" type="inproceedings" rend="refer" n="refercite:li:hal-01349771">
      <identifiant type="doi" value="10.1109/IROS.2016.7759437"/>
      <identifiant type="hal" value="hal-01349771"/>
      <analytic>
        <title level="a">Reverberant Sound Localization with a Robot Head Based on Direct-Path Relative Transfer Function</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName>
            <foreName>Fabien</foreName>
            <surname>Badeig</surname>
            <initial>F.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="m">IEEE/RSJ International Conference on Intelligent Robots and Systems</title>
        <loc>Daejeon, South Korea</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <publisher>
            <orgName type="organisation">IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>October</month>
            <year>2016</year>
          </dateStruct>
          <biblScope type="pages">2819-2826</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01349771" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01349771</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid5" type="article" rend="refer" n="refercite:li:hal-01349691">
      <identifiant type="doi" value="10.1109/TASLP.2016.2598319"/>
      <identifiant type="hal" value="hal-01349691"/>
      <analytic>
        <title level="a">Estimation of the Direct-Path Relative Transfer Function for Supervised Sound-Source Localization</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE/ACM Transactions on Audio, Speech and Language Processing</title>
        <imprint>
          <biblScope type="volume">24</biblScope>
          <biblScope type="number">11</biblScope>
          <dateStruct>
            <month>November</month>
            <year>2016</year>
          </dateStruct>
          <biblScope type="pages">2171 - 2186</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01349691" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01349691</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid55" type="article" rend="refer" n="refercite:li:hal-01413417">
      <identifiant type="doi" value="10.1109/TASLP.2017.2740001"/>
      <identifiant type="hal" value="hal-01413417"/>
      <analytic>
        <title level="a">Multiple-Speaker Localization Based on Direct-Path Features and Likelihood Maximization with Spatial Sparsity Regularization</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes">
        <title level="j">IEEE/ACM Transactions on Audio, Speech and Language Processing</title>
        <imprint>
          <biblScope type="volume">25</biblScope>
          <biblScope type="number">10</biblScope>
          <dateStruct>
            <month>October</month>
            <year>2017</year>
          </dateStruct>
          <biblScope type="pages">1997 - 2012</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01413417" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01413417</ref>
        </imprint>
      </monogr>
      <note type="bnote">16 pages, 4 figures, 4 tables</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid12" type="article" rend="refer" n="refercite:sapienza:hal-00768615">
      <identifiant type="doi" value="10.1007/s10514-012-9311-2"/>
      <identifiant type="hal" value="hal-00768615"/>
      <analytic>
        <title level="a">Real-time Visuomotor Update of an Active Binocular Head</title>
        <author>
          <persName>
            <foreName>Michael</foreName>
            <surname>Sapienza</surname>
            <initial>M.</initial>
          </persName>
          <persName>
            <foreName>Miles</foreName>
            <surname>Hansard</surname>
            <initial>M.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">Autonomous Robots</title>
        <imprint>
          <biblScope type="volume">34</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <month>January</month>
            <year>2013</year>
          </dateStruct>
          <biblScope type="pages">33-45</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00768615" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00768615</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid16" type="article" rend="refer" n="refercite:zaharescu:inria-00590271">
      <identifiant type="doi" value="10.1109/TPAMI.2010.116"/>
      <identifiant type="hal" value="inria-00590271"/>
      <analytic>
        <title level="a">Topology-Adaptive Mesh Deformation for Surface Evolution, Morphing, and Multi-View Reconstruction</title>
        <author>
          <persName>
            <foreName>Andrei</foreName>
            <surname>Zaharescu</surname>
            <initial>A.</initial>
          </persName>
          <persName key="morpheo-2018-idp143488">
            <foreName>Edmond</foreName>
            <surname>Boyer</surname>
            <initial>E.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">33</biblScope>
          <biblScope type="number">4</biblScope>
          <dateStruct>
            <month>April</month>
            <year>2011</year>
          </dateStruct>
          <biblScope type="pages">823-837</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00590271" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00590271</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid17" type="article" rend="refer" n="refercite:zaharescu:hal-00699620">
      <identifiant type="doi" value="10.1007/s11263-012-0528-5"/>
      <identifiant type="hal" value="hal-00699620"/>
      <analytic>
        <title level="a">Keypoints and Local Descriptors of Scalar Functions on 2D Manifolds</title>
        <author>
          <persName>
            <foreName>Andrei</foreName>
            <surname>Zaharescu</surname>
            <initial>A.</initial>
          </persName>
          <persName key="morpheo-2018-idp143488">
            <foreName>Edmond</foreName>
            <surname>Boyer</surname>
            <initial>E.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of Computer Vision</title>
        <imprint>
          <biblScope type="volume">100</biblScope>
          <biblScope type="number">1</biblScope>
          <dateStruct>
            <month>October</month>
            <year>2012</year>
          </dateStruct>
          <biblScope type="pages">78-98</biblScope>
          <ref xlink:href="http://hal.inria.fr/hal-00699620" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-00699620</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid15" type="article" rend="refer" n="refercite:zaharescu:inria-00446987">
      <identifiant type="doi" value="10.1007/s11263-008-0169-x"/>
      <identifiant type="hal" value="inria-00446987"/>
      <analytic>
        <title level="a">Robust Factorization Methods Using A Gaussian/Uniform Mixture Model</title>
        <author>
          <persName>
            <foreName>Andrei</foreName>
            <surname>Zaharescu</surname>
            <initial>A.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr>
        <title level="j">International Journal of Computer Vision</title>
        <imprint>
          <biblScope type="volume">81</biblScope>
          <biblScope type="number">3</biblScope>
          <dateStruct>
            <month>March</month>
            <year>2009</year>
          </dateStruct>
          <biblScope type="pages">240-258</biblScope>
          <ref xlink:href="http://hal.inria.fr/inria-00446987" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">http://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>inria-00446987</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid25" type="book" rend="year" n="cite:alamedapineda:hal-01858395">
      <identifiant type="hal" value="hal-01858395"/>
      <monogr x-scientific-popularization="no" x-international-audience="yes">
        <title level="m">Multimodal behavior analysis in the wild: Advances and challenges</title>
        <author>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName>
            <foreName>Elisa</foreName>
            <surname>Ricci</surname>
            <initial>E.</initial>
          </persName>
          <persName>
            <foreName>Nicu</foreName>
            <surname>Sebe</surname>
            <initial>N.</initial>
          </persName>
        </author>
        <imprint>
          <publisher>
            <orgName>Academic Press (Elsevier)</orgName>
          </publisher>
          <dateStruct>
            <month>December</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01858395" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01858395</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid42" type="phdthesis" rend="year" n="cite:gebru:tel-01774233">
      <identifiant type="hal" value="tel-01774233"/>
      <monogr>
        <title level="m">Audio-Visual Analysis In the Framework of Humans Interacting with Robots</title>
        <author>
          <persName>
            <foreName>Israel D.</foreName>
            <surname>Gebru</surname>
            <initial>I. D.</initial>
          </persName>
        </author>
        <imprint>
          <publisher>
            <orgName type="school">Université Grenoble Alpes</orgName>
          </publisher>
          <dateStruct>
            <month>April</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/tel-01774233" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>tel-01774233</ref>
        </imprint>
      </monogr>
      <note type="typdoc">Theses</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid43" type="phdthesis" rend="year" n="cite:lathuiliere:tel-01801807">
      <identifiant type="hal" value="tel-01801807"/>
      <monogr>
        <title level="m">Deep Regression Models and Computer Vision Applications for Multiperson Human-Robot Interaction</title>
        <author>
          <persName key="perception-2018-idp171728">
            <foreName>Stéphane</foreName>
            <surname>Lathuilière</surname>
            <initial>S.</initial>
          </persName>
        </author>
        <imprint>
          <publisher>
            <orgName type="school">Université Grenoble Alpes</orgName>
          </publisher>
          <dateStruct>
            <month>May</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://tel.archives-ouvertes.fr/tel-01801807" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>tel.<allowbreak/>archives-ouvertes.<allowbreak/>fr/<allowbreak/>tel-01801807</ref>
        </imprint>
      </monogr>
      <note type="typdoc">Theses</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid44" type="phdthesis" rend="year" n="cite:masse:tel-01936821">
      <identifiant type="hal" value="tel-01936821"/>
      <monogr>
        <title level="m">Gaze Direction in the context of Social Human-Robot Interaction</title>
        <author>
          <persName key="perception-2018-idp174160">
            <foreName>Benoit</foreName>
            <surname>Massé</surname>
            <initial>B.</initial>
          </persName>
        </author>
        <imprint>
          <publisher>
            <orgName type="school">Université Grenoble - Alpes</orgName>
          </publisher>
          <dateStruct>
            <month>October</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/tel-01936821" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>tel-01936821</ref>
        </imprint>
      </monogr>
      <note type="typdoc">Theses</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid41" type="article" rend="year" n="cite:evangelidis:hal-01413414">
      <identifiant type="doi" value="10.1109/TPAMI.2017.2717829"/>
      <identifiant type="hal" value="hal-01413414"/>
      <analytic>
        <title level="a">Joint Alignment of Multiple Point Sets with Batch and Incremental Expectation-Maximization</title>
        <author>
          <persName>
            <foreName>Georgios</foreName>
            <surname>Evangelidis</surname>
            <initial>G.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid00747">
        <idno type="issn">0162-8828</idno>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">40</biblScope>
          <biblScope type="number">6</biblScope>
          <dateStruct>
            <month>June</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">1397 - 1410</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01413414" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01413414</ref>
        </imprint>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1609.01466" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1609.<allowbreak/>01466</ref>
      </note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid28" type="article" rend="year" n="cite:gebru:hal-01413403">
      <identifiant type="doi" value="10.1109/TPAMI.2017.2648793"/>
      <identifiant type="hal" value="hal-01413403"/>
      <analytic>
        <title level="a">Audio-Visual Speaker Diarization Based on Spatiotemporal Bayesian Fusion</title>
        <author>
          <persName>
            <foreName>Israel</foreName>
            <surname>Gebru</surname>
            <initial>I.</initial>
          </persName>
          <persName>
            <foreName>Sileye</foreName>
            <surname>Ba</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid00747">
        <idno type="issn">0162-8828</idno>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">40</biblScope>
          <biblScope type="number">5</biblScope>
          <dateStruct>
            <month>July</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">1086 - 1099</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01413403" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01413403</ref>
        </imprint>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1603.09725" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1603.<allowbreak/>09725</ref>
      </note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid49" type="incollection" rend="year" n="cite:girin:hal-01943375">
      <identifiant type="doi" value="10.1016/B978-0-12-814601-9.00022-5"/>
      <identifiant type="hal" value="hal-01943375"/>
      <analytic>
        <title level="a">Audio source separation into the wild</title>
        <author>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no">
        <title level="m">Multimodal Behavior Analysis in the Wild</title>
        <title level="s">Computer Vision and Pattern Recognition</title>
        <imprint>
          <publisher>
            <orgName>Academic Press (Elsevier)</orgName>
          </publisher>
          <dateStruct>
            <month>November</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">53-78</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01943375" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01943375</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid39" type="article" rend="year" n="cite:lathuiliere:hal-01643775">
      <identifiant type="doi" value="10.1016/j.patrec.2018.05.023"/>
      <identifiant type="hal" value="hal-01643775"/>
      <analytic>
        <title level="a">Neural Network Based Reinforcement Learning for Audio-Visual Gaze Control in Human-Robot Interaction</title>
        <author>
          <persName key="perception-2018-idp171728">
            <foreName>Stéphane</foreName>
            <surname>Lathuilière</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp174160">
            <foreName>Benoît</foreName>
            <surname>Massé</surname>
            <initial>B.</initial>
          </persName>
          <persName>
            <foreName>Pablo</foreName>
            <surname>Mesejo</surname>
            <initial>P.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid01563">
        <idno type="issn">0167-8655</idno>
        <title level="j">Pattern Recognition Letters</title>
        <imprint>
          <dateStruct>
            <month>May</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01643775" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01643775</ref>
        </imprint>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1711.06834" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1711.<allowbreak/>06834</ref>
      </note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid27" type="article" rend="year" n="cite:li:hal-01645749">
      <identifiant type="doi" value="10.1109/TASLP.2018.2839362"/>
      <identifiant type="hal" value="hal-01645749"/>
      <analytic>
        <title level="a">Multichannel Identification and Nonnegative Equalization for Dereverberation and Noise Reduction based on Convolutive Transfer Function</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid00706">
        <idno type="issn">1558-7916</idno>
        <title level="j">IEEE/ACM Transactions on Audio, Speech and Language Processing</title>
        <imprint>
          <biblScope type="volume">26</biblScope>
          <biblScope type="number">10</biblScope>
          <dateStruct>
            <month>May</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">1755-1768</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01645749" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01645749</ref>
        </imprint>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1711.07911" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1711.<allowbreak/>07911</ref>
      </note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid26" type="article" rend="year" n="cite:li:hal-01799809">
      <identifiant type="doi" value="10.1109/TASLP.2019.2892412"/>
      <identifiant type="hal" value="hal-01799809"/>
      <analytic>
        <title level="a">Multichannel Speech Separation and Enhancement Using the Convolutive Transfer Function</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid00706">
        <idno type="issn">1558-7916</idno>
        <title level="j">IEEE/ACM Transactions on Audio, Speech and Language Processing</title>
        <imprint>
          <dateStruct>
            <month>January</month>
            <year>2019</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01799809" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01799809</ref>
        </imprint>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1711.07911" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1711.<allowbreak/>07911</ref>
      </note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid53" type="article" rend="year" n="cite:li:hal-01982250">
      <identifiant type="doi" value="10.1049/trit.2018.1061"/>
      <identifiant type="hal" value="hal-01982250"/>
      <analytic>
        <title level="a">Expectation-Maximization for Speech Source Separation using Convolutive Transfer Function</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid03398">
        <idno type="issn">2468-2322</idno>
        <title level="j">CAAI Transactions on Intelligent Technologies</title>
        <imprint>
          <dateStruct>
            <month>January</month>
            <year>2019</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01982250" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01982250</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid46" type="article" rend="year" n="cite:marriott:hal-01663984">
      <identifiant type="doi" value="10.1016/j.patrec.2018.03.024"/>
      <identifiant type="hal" value="hal-01663984"/>
      <analytic>
        <title level="a">Plane-extraction from depth-data using a Gaussian mixture regression model</title>
        <author>
          <persName>
            <foreName>Richard T</foreName>
            <surname>Marriott</surname>
            <initial>R. T.</initial>
          </persName>
          <persName key="thoth-2018-idp167168">
            <foreName>Alexander</foreName>
            <surname>Pashevich</surname>
            <initial>A.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid01563">
        <idno type="issn">0167-8655</idno>
        <title level="j">Pattern Recognition Letters</title>
        <imprint>
          <biblScope type="volume">110</biblScope>
          <dateStruct>
            <month>July</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">44-50</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01663984" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01663984</ref>
        </imprint>
      </monogr>
      <note type="bnote"><ref xlink:href="https://arxiv.org/abs/1710.01925" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1710.<allowbreak/>01925</ref> - 2 figures, 1 table</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid30" type="article" rend="year" n="cite:masse:hal-01511414">
      <identifiant type="doi" value="10.1109/TPAMI.2017.2782819"/>
      <identifiant type="hal" value="hal-01511414"/>
      <analytic>
        <title level="a">Tracking Gaze and Visual Focus of Attention of People Involved in Social Interaction</title>
        <author>
          <persName key="perception-2018-idp174160">
            <foreName>Benoît</foreName>
            <surname>Massé</surname>
            <initial>B.</initial>
          </persName>
          <persName>
            <foreName>Silèye</foreName>
            <surname>Ba</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid00747">
        <idno type="issn">0162-8828</idno>
        <title level="j">IEEE Transactions on Pattern Analysis and Machine Intelligence</title>
        <imprint>
          <biblScope type="volume">40</biblScope>
          <biblScope type="number">11</biblScope>
          <dateStruct>
            <month>November</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">2711 - 2724</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01511414" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01511414</ref>
        </imprint>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1703.04727" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1703.<allowbreak/>04727</ref>
      </note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid48" type="article" rend="year" n="cite:xu:hal-01803694">
      <identifiant type="doi" value="10.1109/TIP.2018.2837381"/>
      <identifiant type="hal" value="hal-01803694"/>
      <analytic>
        <title level="a">Cross-Paced Representation Learning with Partial Curricula for Sketch-based Image Retrieval</title>
        <author>
          <persName>
            <foreName>Dan</foreName>
            <surname>Xu</surname>
            <initial>D.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName>
            <foreName>Jingkuan</foreName>
            <surname>Song</surname>
            <initial>J.</initial>
          </persName>
          <persName>
            <foreName>Elisa</foreName>
            <surname>Ricci</surname>
            <initial>E.</initial>
          </persName>
          <persName>
            <foreName>Nicu</foreName>
            <surname>Sebe</surname>
            <initial>N.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-editorial-board="yes" x-international-audience="yes" id="rid00728">
        <idno type="issn">1057-7149</idno>
        <title level="j">IEEE Transactions on Image Processing</title>
        <imprint>
          <biblScope type="volume">27</biblScope>
          <biblScope type="number">9</biblScope>
          <dateStruct>
            <month>September</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">4410-4421</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01803694" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01803694</ref>
        </imprint>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid0" type="inproceedings" rend="year" n="cite:ban:hal-01718114">
      <identifiant type="doi" value="10.1109/ICASSP.2018.8462100"/>
      <identifiant type="hal" value="hal-01718114"/>
      <analytic>
        <title level="a">Accounting for Room Acoustics in Audio-Visual Multi-Speaker Tracking</title>
        <author>
          <persName key="perception-2018-idp164448">
            <foreName>Yutong</foreName>
            <surname>Ban</surname>
            <initial>Y.</initial>
          </persName>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">ICASSP 2018 - IEEE International Conference on Acoustics, Speech and Signal Processing</title>
        <loc>Calgary, Alberta, Canada</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>April</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">6553-6557</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01718114" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01718114</ref>
        </imprint>
        <meeting id="cid80145">
          <title>IEEE International Conference on Acoustics, Speech and Signal Processing</title>
          <num>43</num>
          <abbr type="sigle">ICASSP</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid38" type="inproceedings" rend="year" n="cite:lathuiliere:hal-01851738">
      <identifiant type="doi" value="10.1109/IROS.2018.8594327"/>
      <identifiant type="hal" value="hal-01851738"/>
      <analytic>
        <title level="a">Deep Reinforcement Learning for Audio-Visual Gaze Control</title>
        <author>
          <persName key="perception-2018-idp171728">
            <foreName>Stéphane</foreName>
            <surname>Lathuilière</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp174160">
            <foreName>Benoit</foreName>
            <surname>Massé</surname>
            <initial>B.</initial>
          </persName>
          <persName>
            <foreName>Pablo</foreName>
            <surname>Mesejo</surname>
            <initial>P.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">IROS 2018 - IEEE/RSJ International Conference on Intelligent Robots and Systems</title>
        <loc>Madrid, Spain</loc>
        <imprint>
          <dateStruct>
            <month>October</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">1-8</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01851738" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01851738</ref>
        </imprint>
        <meeting id="cid93437">
          <title>IEEE RSJ International Conference on Intelligent Robots and Systems</title>
          <num>2018</num>
          <abbr type="sigle">IROS</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid33" type="inproceedings" rend="year" n="cite:lathuiliere:hal-01851511">
      <identifiant type="hal" value="hal-01851511"/>
      <analytic>
        <title level="a">DeepGUM: Learning Deep Robust Regression with a Gaussian-Uniform Mixture Model</title>
        <author>
          <persName key="perception-2018-idp171728">
            <foreName>Stéphane</foreName>
            <surname>Lathuilière</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Pablo</foreName>
            <surname>Mesejo</surname>
            <initial>P.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">ECCV 2018 - European Conference on Computer Vision</title>
        <loc>Munich, Germany</loc>
        <imprint>
          <dateStruct>
            <month>September</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">1-16</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01851511" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01851511</ref>
        </imprint>
        <meeting id="cid66293">
          <title>European Conference on Computer Vision</title>
          <num>2018</num>
          <abbr type="sigle">ECCV</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid52" type="inproceedings" rend="year" n="cite:leglaive:hal-01832826">
      <identifiant type="doi" value="10.1109/MLSP.2018.8516711"/>
      <identifiant type="hal" value="hal-01832826"/>
      <analytic>
        <title level="a">A variance modeling framework based on variational autoencoders for speech enhancement</title>
        <author>
          <persName key="perception-2018-idp159488">
            <foreName>Simon</foreName>
            <surname>Leglaive</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">MSLP 2018 - IEEE International Workshop on Machine Learning for Signal Processing</title>
        <loc>Aalborg, Denmark</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>September</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">1-6</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01832826" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01832826</ref>
        </imprint>
        <meeting id="cid91840">
          <title>IEEE International Workshop on Machine Learning for Signal Processing</title>
          <num>2018</num>
          <abbr type="sigle">MLSP</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid54" type="inproceedings" rend="year" n="cite:li:hal-01957137">
      <identifiant type="hal" value="hal-01957137"/>
      <analytic>
        <title level="a">A Cascaded Multiple-Speaker Localization and Tracking System</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp164448">
            <foreName>Yutong</foreName>
            <surname>Ban</surname>
            <initial>Y.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">Proceedings of the LOCATA Challenge Workshop - a satellite event of IWAENC 2018</title>
        <loc>Tokyo, Japan</loc>
        <imprint>
          <dateStruct>
            <month>September</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">1-5</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01957137" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01957137</ref>
        </imprint>
        <meeting id="cid626000">
          <title>IEEE-AASP Challenge on Acoustic Source Localization and Tracking</title>
          <num>2018</num>
          <abbr type="sigle">LOCATA</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid47" type="inproceedings" rend="year" n="cite:li:hal-01718106">
      <identifiant type="doi" value="10.1109/ICASSP.2018.8462607"/>
      <identifiant type="hal" value="hal-01718106"/>
      <analytic>
        <title level="a">Multisource MINT Using the Convolutive Transfer Function</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">ICASSP 2018 - IEEE International Conference on Acoustics, Speech and Signal Processing</title>
        <loc>Calgary, Alberta, Canada</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>April</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">756-760</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01718106" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01718106</ref>
        </imprint>
        <meeting id="cid80145">
          <title>IEEE International Conference on Acoustics, Speech and Signal Processing</title>
          <num>43</num>
          <abbr type="sigle">ICASSP</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid36" type="inproceedings" rend="year" n="cite:li:hal-01795462">
      <identifiant type="doi" value="10.1109/SAM.2018.8448423"/>
      <identifiant type="hal" value="hal-01795462"/>
      <analytic>
        <title level="a">Online Localization of Multiple Moving Speakers in Reverberant Environments</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp181504">
            <foreName>Bastien</foreName>
            <surname>Mourgue</surname>
            <initial>B.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">10th IEEE Workshop on Sensor Array and Multichannel Signal Processing (SAM 2018)</title>
        <loc>Sheffield, United Kingdom</loc>
        <imprint>
          <publisher>
            <orgName>IEEE</orgName>
          </publisher>
          <dateStruct>
            <month>July</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">405-409</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01795462" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01795462</ref>
        </imprint>
        <meeting id="cid624661">
          <title>IEEE Sensor Array and Multichannel Signal Processing Workshop</title>
          <num>10</num>
          <abbr type="sigle">SAM</abbr>
        </meeting>
      </monogr>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid50" type="inproceedings" rend="year" n="cite:siarohin:hal-01761539">
      <identifiant type="hal" value="hal-01761539"/>
      <analytic>
        <title level="a">Deformable GANs for Pose-based Human Image Generation</title>
        <author>
          <persName>
            <foreName>Aliaksandr</foreName>
            <surname>Siarohin</surname>
            <initial>A.</initial>
          </persName>
          <persName>
            <foreName>Enver</foreName>
            <surname>Sangineto</surname>
            <initial>E.</initial>
          </persName>
          <persName key="perception-2018-idp171728">
            <foreName>Stéphane</foreName>
            <surname>Lathuilière</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Nicu</foreName>
            <surname>Sebe</surname>
            <initial>N.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">IEEE Conference on Computer Vision and Pattern Recognition</title>
        <loc>Salt Lake City, United States</loc>
        <imprint>
          <dateStruct>
            <month>June</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">3408-3416</biblScope>
          <ref xlink:href="https://hal.archives-ouvertes.fr/hal-01761539" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>archives-ouvertes.<allowbreak/>fr/<allowbreak/>hal-01761539</ref>
        </imprint>
        <meeting id="cid82398">
          <title>IEEE International Conference on Computer Vision and Pattern Recognition</title>
          <num>2009</num>
          <abbr type="sigle">CVPR</abbr>
        </meeting>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1801.00055" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1801.<allowbreak/>00055</ref>
      </note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid40" type="inproceedings" rend="year" n="cite:wang:hal-01759335">
      <identifiant type="hal" value="hal-01759335"/>
      <analytic>
        <title level="a">Every Smile is Unique: Landmark-Guided Diverse Smile Generation</title>
        <author>
          <persName>
            <foreName>Wei</foreName>
            <surname>Wang</surname>
            <initial>W.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName>
            <foreName>Dan</foreName>
            <surname>Xu</surname>
            <initial>D.</initial>
          </persName>
          <persName>
            <foreName>Pascal</foreName>
            <surname>Fua</surname>
            <initial>P.</initial>
          </persName>
          <persName>
            <foreName>Elisa</foreName>
            <surname>Ricci</surname>
            <initial>E.</initial>
          </persName>
          <persName>
            <foreName>Nicu</foreName>
            <surname>Sebe</surname>
            <initial>N.</initial>
          </persName>
        </author>
      </analytic>
      <monogr x-scientific-popularization="no" x-international-audience="yes" x-proceedings="yes" x-invited-conference="no" x-editorial-board="yes">
        <title level="m">IEEE Conference on Computer Vision and Pattern Recognition</title>
        <loc>Salk Lake City, United States</loc>
        <imprint>
          <dateStruct>
            <month>June</month>
            <year>2018</year>
          </dateStruct>
          <biblScope type="pages">7083-7092</biblScope>
          <ref xlink:href="https://hal.inria.fr/hal-01759335" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01759335</ref>
        </imprint>
        <meeting id="cid82398">
          <title>IEEE International Conference on Computer Vision and Pattern Recognition</title>
          <num>2009</num>
          <abbr type="sigle">CVPR</abbr>
        </meeting>
      </monogr>
      <note type="bnote">
        <ref xlink:href="https://arxiv.org/abs/1802.01873" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1802.<allowbreak/>01873</ref>
      </note>
    </biblStruct>
    
    <biblStruct subtype="nonparu" id="perception-2018-bid31" type="unpublished" rend="year" n="cite:ban:hal-01969050">
      <identifiant type="hal" value="hal-01969050"/>
      <monogr>
        <title level="m">Tracking Multiple Audio Sources with the von Mises Distribution and Variational EM</title>
        <author>
          <persName key="perception-2018-idp164448">
            <foreName>Yutong</foreName>
            <surname>Ban</surname>
            <initial>Y.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp196336">
            <foreName>Christine</foreName>
            <surname>Evers</surname>
            <initial>C.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
        <imprint>
          <dateStruct>
            <month>December</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01969050" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01969050</ref>
        </imprint>
      </monogr>
      <note type="bnote"><ref xlink:href="https://arxiv.org/abs/1812.08246" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1812.<allowbreak/>08246</ref> - Paper submitted to IEEE Signal Processing Letters</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid1" type="unpublished" rend="year" n="cite:ban:hal-01950866">
      <identifiant type="hal" value="hal-01950866"/>
      <monogr>
        <title level="m">Variational Bayesian Inference for Audio-Visual Tracking of Multiple Speakers</title>
        <author>
          <persName key="perception-2018-idp164448">
            <foreName>Yutong</foreName>
            <surname>Ban</surname>
            <initial>Y.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
        <imprint>
          <dateStruct>
            <month>December</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01950866" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01950866</ref>
        </imprint>
      </monogr>
      <note type="bnote"><ref xlink:href="https://arxiv.org/abs/1809.10961" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1809.<allowbreak/>10961</ref> - Submitted to IEEE Transactions on Pattern Analysis and Machine Intelligence</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid34" type="unpublished" rend="year" n="cite:lathuiliere:hal-01754839">
      <identifiant type="hal" value="hal-01754839"/>
      <monogr>
        <title level="m">A Comprehensive Analysis of Deep Regression</title>
        <author>
          <persName key="perception-2018-idp171728">
            <foreName>Stéphane</foreName>
            <surname>Lathuilière</surname>
            <initial>S.</initial>
          </persName>
          <persName>
            <foreName>Pablo</foreName>
            <surname>Mesejo</surname>
            <initial>P.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
        <imprint>
          <dateStruct>
            <month>March</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01754839" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01754839</ref>
        </imprint>
      </monogr>
      <note type="bnote"><ref xlink:href="https://arxiv.org/abs/1803.08450" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1803.<allowbreak/>08450</ref> - Submitted to IEEE Transactions on Pattern Analysis and Machine Intelligence</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid2" subtype="nonparu-n" type="unpublished" rend="year" n="cite:li:hal-01851985">
      <identifiant type="hal" value="hal-01851985"/>
      <monogr>
        <title level="m">Online Localization and Tracking of Multiple Moving Speakers in Reverberant Environment</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp164448">
            <foreName>Yutong</foreName>
            <surname>Ban</surname>
            <initial>Y.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
        <imprint>
          <dateStruct>
            <month>July</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01851985" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01851985</ref>
        </imprint>
      </monogr>
      <note type="bnote">Submitted to Journal on Selected Topics in Signal Processing</note>
    </biblStruct>
    
    <biblStruct subtype="nonparu" id="perception-2018-bid51" type="unpublished" rend="year" n="cite:li:hal-01969041">
      <identifiant type="hal" value="hal-01969041"/>
      <monogr>
        <title level="m">Multichannel Online Dereverberation based on Spectral Magnitude Inverse Filtering</title>
        <author>
          <persName key="perception-2018-idp151616">
            <foreName>Xiaofei</foreName>
            <surname>Li</surname>
            <initial>X.</initial>
          </persName>
          <persName key="perception-2018-idp156608">
            <foreName>Laurent</foreName>
            <surname>Girin</surname>
            <initial>L.</initial>
          </persName>
          <persName key="perception-2018-idp198832">
            <foreName>Sharon</foreName>
            <surname>Gannot</surname>
            <initial>S.</initial>
          </persName>
          <persName key="perception-2018-idp146240">
            <foreName>Radu</foreName>
            <surname>Horaud</surname>
            <initial>R.</initial>
          </persName>
        </author>
        <imprint>
          <dateStruct>
            <month>December</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01969041" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01969041</ref>
        </imprint>
      </monogr>
      <note type="bnote"><ref xlink:href="https://arxiv.org/abs/1812.08471" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>arxiv.<allowbreak/>org/<allowbreak/>abs/<allowbreak/>1812.<allowbreak/>08471</ref> - Paper submitted to IEEE/ACM Transactions on Audio, Speech and Language Processing</note>
    </biblStruct>
    
    <biblStruct id="perception-2018-bid45" type="unpublished" rend="year" n="cite:siarohin:hal-01858389">
      <identifiant type="hal" value="hal-01858389"/>
      <monogr>
        <title level="m">Increasing Image Memorability with Neural Style Transfer</title>
        <author>
          <persName>
            <foreName>Aliaksandr</foreName>
            <surname>Siarohin</surname>
            <initial>A.</initial>
          </persName>
          <persName>
            <foreName>Gloria</foreName>
            <surname>Zen</surname>
            <initial>G.</initial>
          </persName>
          <persName>
            <foreName>Cveta</foreName>
            <surname>Majtanovic</surname>
            <initial>C.</initial>
          </persName>
          <persName key="perception-2018-idp149152">
            <foreName>Xavier</foreName>
            <surname>Alameda-Pineda</surname>
            <initial>X.</initial>
          </persName>
          <persName>
            <foreName>Elisa</foreName>
            <surname>Ricci</surname>
            <initial>E.</initial>
          </persName>
          <persName>
            <foreName>Nicu</foreName>
            <surname>Sebe</surname>
            <initial>N.</initial>
          </persName>
        </author>
        <imprint>
          <dateStruct>
            <month>August</month>
            <year>2018</year>
          </dateStruct>
          <ref xlink:href="https://hal.inria.fr/hal-01858389" location="extern" xlink:type="simple" xlink:show="replace" xlink:actuate="onRequest">https://<allowbreak/>hal.<allowbreak/>inria.<allowbreak/>fr/<allowbreak/>hal-01858389</ref>
        </imprint>
      </monogr>
      <note type="bnote">working paper or preprint</note>
    </biblStruct>
  </biblio>
</raweb>
