<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">72196</article-id><article-id pub-id-type="doi">10.7554/eLife.72196</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Physics of Living Systems</subject></subj-group></article-categories><title-group><article-title>Learning to predict target location with turbulent odor plumes</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" id="author-246493"><name><surname>Rigolli</surname><given-names>Nicola</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0734-2105</contrib-id><email>nicola.rigolli@edu.unige.it</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-246494"><name><surname>Magnoli</surname><given-names>Nicodemo</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-246495"><name><surname>Rosasco</surname><given-names>Lorenzo</given-names></name><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf2"/></contrib><contrib contrib-type="author" corresp="yes" id="author-72412"><name><surname>Seminara</surname><given-names>Agnese</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-5633-8180</contrib-id><email>agnese.seminara@unige.it</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf3"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0107c5v14</institution-id><institution>Department of Physics, University of Genova</institution></institution-wrap><addr-line><named-content content-type="city">Genova</named-content></addr-line><country>Italy</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/042cesy50</institution-id><institution>Institut de Physique de Nice, Université Côte d’Azur, Centre National de la Recherche Scientifique</institution></institution-wrap><addr-line><named-content content-type="city">Nice</named-content></addr-line><country>France</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02v89pq06</institution-id><institution>National Institute of Nuclear Physics</institution></institution-wrap><addr-line><named-content content-type="city">Genova</named-content></addr-line><country>Italy</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0107c5v14</institution-id><institution>MalGa, Department of Civil, Chemical and Environmental Engineering, University of Genoa</institution></institution-wrap><addr-line><named-content content-type="city">Genoa</named-content></addr-line><country>Italy</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0107c5v14</institution-id><institution>MaLGa, Department of computer science, bioengineering, robotics and systems engineering, University of Genova</institution></institution-wrap><addr-line><named-content content-type="city">Genova</named-content></addr-line><country>Italy</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Goldstein</surname><given-names>Raymond E</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/013meh722</institution-id><institution>University of Cambridge</institution></institution-wrap><country>United Kingdom</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Walczak</surname><given-names>Aleksandra M</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03a26mh11</institution-id><institution>CNRS LPENS</institution></institution-wrap><country>France</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>12</day><month>08</month><year>2022</year></pub-date><pub-date pub-type="collection"><year>2022</year></pub-date><volume>11</volume><elocation-id>e72196</elocation-id><history><date date-type="received" iso-8601-date="2021-07-14"><day>14</day><month>07</month><year>2021</year></date><date date-type="accepted" iso-8601-date="2022-07-18"><day>18</day><month>07</month><year>2022</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at bioRxiv.</event-desc><date date-type="preprint" iso-8601-date="2021-06-16"><day>16</day><month>06</month><year>2021</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.48550/arXiv.2106.08988"/></event></pub-history><permissions><copyright-statement>© 2022, Rigolli et al</copyright-statement><copyright-year>2022</copyright-year><copyright-holder>Rigolli et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-72196-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-72196-figures-v1.pdf"/><abstract><p>Animal behavior and neural recordings show that the brain is able to measure both the intensity and the timing of odor encounters. However, whether intensity or timing of odor detections is more informative for olfactory-driven behavior is not understood. To tackle this question, we consider the problem of locating a target using the odor it releases. We ask whether the position of a target is best predicted by measures of timing <italic>vs</italic> intensity of its odor, sampled for a short period of time. To answer this question, we feed data from accurate numerical simulations of odor transport to machine learning algorithms that learn how to connect odor to target location. We find that both intensity and timing can separately predict target location even from a distance of several meters; however, their efficacy varies with the dilution of the odor in space. Thus, organisms that use olfaction from different ranges may have to switch among different modalities. This has implications on how the brain should represent odors as the target is approached. We demonstrate simple strategies to improve accuracy and robustness of the prediction by modifying odor sampling and appropriately combining distinct measures together. To test the predictions, animal behavior and odor representation should be monitored as the animal moves relative to the target, or in virtual conditions that mimic concentrated <italic>vs</italic> dilute environments.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>olfaction</kwd><kwd>fluid dynamics</kwd><kwd>machine learning</kwd><kwd>prediction</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>None</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100000781</institution-id><institution>European Research Council</institution></institution-wrap></funding-source><award-id>101002724</award-id><principal-award-recipient><name><surname>Seminara</surname><given-names>Agnese</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000181</institution-id><institution>Air Force Office of Scientific Research</institution></institution-wrap></funding-source><award-id>FA8655-20-1-7028</award-id><principal-award-recipient><name><surname>Rosasco</surname><given-names>Lorenzo</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01DC018789</award-id><principal-award-recipient><name><surname>Seminara</surname><given-names>Agnese</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100001665</institution-id><institution>Agence Nationale de la Recherche</institution></institution-wrap></funding-source><award-id>ANR-15-IDEX-01</award-id><principal-award-recipient><name><surname>Seminara</surname><given-names>Agnese</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Intensity of an odor and timing of its detection are complementary attributes of turbulent plumes and enable robust prediction of the location of a distant target.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Most macroscopic organisms detect odors in intermittent bursts, that may be separated by extended regions with no odor. Organisms leverage this complex dynamics efficiently for diverse tasks, including locating and identifying an odor source <xref ref-type="bibr" rid="bib32">Murlis et al., 1992</xref>; <xref ref-type="bibr" rid="bib28">Mafra-Neto and Cardé, 1994</xref>; <xref ref-type="bibr" rid="bib53">Vickers, 2000</xref>; <xref ref-type="bibr" rid="bib41">Riffell et al., 2014</xref>; <xref ref-type="bibr" rid="bib1">Ache et al., 2016</xref>; <xref ref-type="bibr" rid="bib2">Ackels et al., 2021</xref>. However, what are the most informative features of intermittent odor cues remains largely unclear. There are two broad classes of measures that quantify the dynamics of olfactory cues: those that depend on odor intensity including e.g. odor gradients in space or time, and those that do not depend on odor intensity but only on its timing, i.e. on whether the odor is on or off regardless of its concentration. To compute quantities that depend on odor intensity, an accurate representation of the odor is needed. In contrast, measuring the timing of odor detection simply requires to mark at all times whether the odor is on or off, thus a binary switch is sufficient.</p><p>Behavioral evidence suggests that animals use both intensity and timing of odor encounters for olfactory navigation <xref ref-type="bibr" rid="bib4">Baker et al., 2018</xref>. At close range, mammals appear to compare odor intensity either across nostrils or across sniffs <xref ref-type="bibr" rid="bib8">Catania, 2013</xref>; <xref ref-type="bibr" rid="bib18">Gire et al., 2016</xref>; <xref ref-type="bibr" rid="bib14">Findley et al., 2021</xref>. On the other hand, mounting evidence suggests timing of odor detection also plays a key role for olfactory navigation <xref ref-type="bibr" rid="bib1">Ache et al., 2016</xref>: moths respond to odor pulsed at specific frequencies <xref ref-type="bibr" rid="bib41">Riffell et al., 2014</xref>; <xref ref-type="bibr" rid="bib54">Vickers et al., 2001</xref>; fruit flies respond to timing since last odor detection <xref ref-type="bibr" rid="bib51">van Breugel and Dickinson, 2014</xref>; <xref ref-type="bibr" rid="bib10">Demir et al., 2020</xref>; lobsters and sharks compare odor arrival time across their paired olfactory organs and orient toward the side that detected the odor first <xref ref-type="bibr" rid="bib5">Basil and Atema, 1994</xref>; <xref ref-type="bibr" rid="bib17">Gardiner and Atema, 2010</xref>; many organisms will move upwind upon detection of an odor <xref ref-type="bibr" rid="bib25">Kennedy and Marsh, 1974</xref>; <xref ref-type="bibr" rid="bib32">Murlis et al., 1992</xref>; <xref ref-type="bibr" rid="bib49">Steck et al., 2012</xref>.</p><p>Neural recordings upon stimulation with intermittent odor cues confirm that the brain of many animals is able to record information both about intensity (and its derivatives) as well as timing of odor encounters (most information comes from work on arthropods <xref ref-type="bibr" rid="bib33">Nagel and Wilson, 2011</xref>; <xref ref-type="bibr" rid="bib54">Vickers et al., 2001</xref>; <xref ref-type="bibr" rid="bib7">Brown et al., 2005</xref>; <xref ref-type="bibr" rid="bib19">Gorur-Shandilya et al., 2017</xref>; <xref ref-type="bibr" rid="bib23">Jacob et al., 2017</xref>; <xref ref-type="bibr" rid="bib41">Riffell et al., 2014</xref>, but see also <xref ref-type="bibr" rid="bib35">Parabucki et al., 2019</xref>; <xref ref-type="bibr" rid="bib27">Lewis et al., 2021</xref>). For example, when insects are presented with intermittent odor cues, information about intensity and timing is recorded in their antennal lobe (see e.g. <xref ref-type="bibr" rid="bib54">Vickers et al., 2001</xref>; <xref ref-type="bibr" rid="bib7">Brown et al., 2005</xref>). Odors that mimic natural intermittency elicit a response that preserves an accurate measure of timing in fruit flies and moths <xref ref-type="bibr" rid="bib19">Gorur-Shandilya et al., 2017</xref>; <xref ref-type="bibr" rid="bib23">Jacob et al., 2017</xref>. In lobsters, bursting olfactory neurons encode specifically for the time between successive odor encounters, see <xref ref-type="bibr" rid="bib36">Park et al., 2014</xref>; <xref ref-type="bibr" rid="bib37">Park et al., 2016</xref> and references therein. Interestingly, the neural activity varies considerably with the dynamics of the odor cues <xref ref-type="bibr" rid="bib33">Nagel and Wilson, 2011</xref>; <xref ref-type="bibr" rid="bib54">Vickers et al., 2001</xref>; <xref ref-type="bibr" rid="bib27">Lewis et al., 2021</xref>, but how intermittency of an odor affects its neural representation is not well understood.</p><p>This evidence suggests animals are able to identify when they detect an odor as well as how intense it is; but whether they record and rely on both kinds of information is not understood. From a physical perspective, these two measures clearly provide information about source location. Indeed, we know from theoretical <xref ref-type="bibr" rid="bib47">Shraiman and Siggia, 2000</xref>; <xref ref-type="bibr" rid="bib13">Falkovich et al., 2001</xref>; <xref ref-type="bibr" rid="bib9">Celani et al., 2014</xref> and experimental <xref ref-type="bibr" rid="bib24">Justus et al., 2002</xref>; <xref ref-type="bibr" rid="bib31">Moore and Crimaldi, 2004</xref> work that turbulence causes the odor to be distributed in highly intermittent patches separated by blanks with no odor. Both intensity and timing of these intermittent bursts vary depending on the location of the source <xref ref-type="bibr" rid="bib9">Celani et al., 2014</xref>, as early recognized by <xref ref-type="bibr" rid="bib3">Atema, 1996</xref>, thus can be used to infer source location or navigate to it <xref ref-type="bibr" rid="bib52">Vergassola et al., 2007</xref>; <xref ref-type="bibr" rid="bib45">Schmuker et al., 2016</xref>; <xref ref-type="bibr" rid="bib6">Boie et al., 2018</xref>; <xref ref-type="bibr" rid="bib26">Leathers et al., 2020</xref>; <xref ref-type="bibr" rid="bib30">Michaelis et al., 2020</xref>.</p><p>Here, we ask what salient features of turbulent odor signals best predict the location of the odor source and specifically compare quantities related to intensity <italic>vs</italic> timing of odor encounters. We first compose a dataset of realistic odor fields at scales of several meters using accurate state-of-the-art fluid dynamics simulations. We then develop machine learning algorithms that predict source location based on these synthetic odor fields.</p><p>We find that measures of odor temporal dynamics based on a short memory span (down to about 1 s) hold information about source location. Close to the source or close to the substrate, measures of intensity predict distance better than measures of timing; but this ranking is reversed at further distance from the source or from the substrate. Pairing the two kinds of measure improves dramatically the quality of the prediction robustly across all datasets, whereas pairing two measures of intensity or two measures of timing is either useless or detrimental.</p><p>Our results demonstrate that timing and intensity are complementary attributes of odor dynamics and are most effective in more dilute and concentrated conditions respectively. These different conditions exist in different portions of space because odor gets transported, mixed and diluted by the fluid. As a result, the spatial range of operation of a living organism constrains the solutions it may evolve to make predictions with turbulent odors.</p></sec><sec id="s2" sec-type="results"><title>Results</title><p>Odor cues at several meters from the source are often turbulent. <xref ref-type="fig" rid="fig1">Figure 1a–c</xref> and show snapshots of the velocity field and odor cues in space, resulting from direct numerical simulations of the turbulent flow in a channel of length L, width W and height H (also see <xref ref-type="video" rid="fig1video1">Figure 1—video 1</xref>). Air flows from left to right at a mean speed <inline-formula><mml:math id="inf1"><mml:msub><mml:mi>U</mml:mi><mml:mi>b</mml:mi></mml:msub></mml:math></inline-formula> and hits a cylindrical obstacle that generates turbulence. The height of the obstacle is <inline-formula><mml:math id="inf2"><mml:mrow><mml:mi>H</mml:mi><mml:mo>/</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:math></inline-formula> and tunes the intensity of turbulent fluctuations relative to the mean velocity. To characterize the flow we show in <xref ref-type="fig" rid="fig1">Figure 1d</xref> that the mean velocity profile, for <inline-formula><mml:math id="inf3"><mml:mrow><mml:msup><mml:mi>z</mml:mi><mml:mo>+</mml:mo></mml:msup><mml:mo>≳</mml:mo><mml:mi/></mml:mrow></mml:math></inline-formula> 30, follows the law of the wall <inline-formula><mml:math id="inf4"><mml:mrow><mml:mfrac><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:mi>U</mml:mi><mml:mo stretchy="false">⟩</mml:mo></mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mi>τ</mml:mi></mml:msub></mml:mfrac><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>κ</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:msup><mml:mi>z</mml:mi><mml:mo>+</mml:mo></mml:msup></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf5"><mml:mi>κ</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf6"><mml:mi>B</mml:mi></mml:math></inline-formula> are constants and <inline-formula><mml:math id="inf7"><mml:mrow><mml:msup><mml:mi>z</mml:mi><mml:mo>+</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mfrac><mml:mi>z</mml:mi><mml:msub><mml:mi>δ</mml:mi><mml:mi>ν</mml:mi></mml:msub></mml:mfrac></mml:mrow></mml:math></inline-formula>, recovering classical statistics for channel turbulence <xref ref-type="bibr" rid="bib39">Pope, 1984</xref>. The odor field is emitted from a concentrated source downstream from the obstacle; it develops as a meandering filament that fluctuates as it travels downstream and soon breaks into discrete pockets of odor (whiffs) separated by odor-less stretches (blanks) (<xref ref-type="fig" rid="fig1">Figure 1b–c and f</xref>). The Kolomogorov scaling for the spectra of odor fluctuations <inline-formula><mml:math id="inf8"><mml:msup><mml:mi>k</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mn>5</mml:mn><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula> holds for <inline-formula><mml:math id="inf9"><mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>⁢</mml:mo><mml:mi>η</mml:mi></mml:mrow><mml:mo>≲</mml:mo><mml:mn>0.1</mml:mn></mml:mrow></mml:math></inline-formula> (see <xref ref-type="fig" rid="fig1">Figure 1e</xref>), consistent with previous experimental results in channel flow (see e.g. <xref ref-type="bibr" rid="bib44">Saddoughi and Veeravalli, 1994</xref>). Typical time courses of the odor are shown in <xref ref-type="fig" rid="fig1">Figure 1f</xref>. Note that depending on the sampling location, odor may be more or less sparse (compare for example <xref ref-type="fig" rid="fig1">Figure 1f</xref> left and right). All parameters and methods are summarized respectively in <xref ref-type="table" rid="table1">Table 1</xref> and in Materials and Methods.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Turbulent odor cues are patchy and intermittent.</title><p>Snapshot of streamwise velocity (<bold>a</bold>) in a vertical plain at mid channel; odor snapshot side view at mid channel (<bold>b</bold>) and top view at source height (<bold>c</bold>). White regions mark the cylindrical obstacle. Snapshots are obtained from direct numerical simulations of the Navier-Stokes equations and the equation for odor transport (see Materials and Methods and parameters summarized in <xref ref-type="table" rid="table1">Table 1</xref>). (<bold>d</bold>) The mean velocity profile follows the well known log law when <inline-formula><mml:math id="inf10"><mml:msubsup><mml:mi>z</mml:mi><mml:mn>3</mml:mn><mml:mo>+</mml:mo></mml:msubsup></mml:math></inline-formula> = <inline-formula><mml:math id="inf11"><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mi>δ</mml:mi><mml:mi>ν</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> &gt; 30, where <inline-formula><mml:math id="inf12"><mml:mrow><mml:msub><mml:mi>δ</mml:mi><mml:mi>ν</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi>ν</mml:mi><mml:mo>/</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mi>τ</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> where <inline-formula><mml:math id="inf13"><mml:msub><mml:mi>u</mml:mi><mml:mi>τ</mml:mi></mml:msub></mml:math></inline-formula> is the friction velocity. (<bold>e</bold>) Two dimensional spectra of odor fluctuations <inline-formula><mml:math id="inf14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>E</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mi>d</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:mfrac><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mo>∫</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">k</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>c</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="bold">k</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:msup><mml:mi>d</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> normalized with the scalar variance <inline-formula><mml:math id="inf15"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msubsup><mml:mi>σ</mml:mi><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:math></inline-formula> ; <inline-formula><mml:math id="inf16"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mover><mml:mi>c</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">k</mml:mi><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the 2D Fourier transform of the scalar concentration at source height; the integral of the spectra is the scalar variance. Wavenumbers are nondimensionalized with the inverse Kolmogorov scale <inline-formula><mml:math id="inf17"><mml:msup><mml:mi>η</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>. Error bars show standard deviation calculated over N = 420 points. (<bold>f</bold>) Typical time courses of the odor cues at locations labeled with 1 and 2 in c, visualizing noise and sparsity, particularly at location 1.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig1-v1.tif"/></fig><media mimetype="video" mime-subtype="mp4" xlink:href="elife-72196-fig1-video1.mp4" id="fig1video1"><label>Figure 1—video 1.</label><caption><title>Direct numerical simulations of the turbulent flow in a channel of length L, width W and height H evolving in time.</title></caption></media></fig-group><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Parameters of the simulation.</title><p>Length <inline-formula><mml:math id="inf18"><mml:mi>L</mml:mi></mml:math></inline-formula>, width <inline-formula><mml:math id="inf19"><mml:mi>W</mml:mi></mml:math></inline-formula>, height <inline-formula><mml:math id="inf20"><mml:mi>H</mml:mi></mml:math></inline-formula> of the computational domain; horizontal speed along the centerline <inline-formula><mml:math id="inf21"><mml:mi>U</mml:mi></mml:math></inline-formula>; mean horizontal speed <inline-formula><mml:math id="inf22"><mml:mrow><mml:msub><mml:mi>U</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:mi>u</mml:mi><mml:mo stretchy="false">⟩</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>; kinematic viscosity <inline-formula><mml:math id="inf23"><mml:mi>ν</mml:mi></mml:math></inline-formula>; diffusivity <inline-formula><mml:math id="inf24"><mml:msub><mml:mi>κ</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula>; Kolmogorov length scale <inline-formula><mml:math id="inf25"><mml:mrow><mml:mi>η</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mi>ν</mml:mi><mml:mn>3</mml:mn></mml:msup><mml:mo>/</mml:mo><mml:mi>ϵ</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> where <inline-formula><mml:math id="inf26"><mml:mi>ϵ</mml:mi></mml:math></inline-formula> is the energy dissipation rate; mean size of gridcell <inline-formula><mml:math id="inf27"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:math></inline-formula>; Kolmogorov timescale <inline-formula><mml:math id="inf28"><mml:mrow><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:msup><mml:mi>η</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>/</mml:mo><mml:mi>ν</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>; energy dissipation rate <inline-formula><mml:math id="inf29"><mml:mrow><mml:mi>ϵ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>ν</mml:mi><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mrow><mml:mo>∂</mml:mo><mml:mo>⁡</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo>∂</mml:mo><mml:mo>⁡</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mrow><mml:mo>∂</mml:mo><mml:mo>⁡</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo>∂</mml:mo><mml:mo>⁡</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">⟩</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>; Taylor microscale <inline-formula><mml:math id="inf30"><mml:mrow><mml:mi>λ</mml:mi><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">⟩</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>∂</mml:mo><mml:mo>⁡</mml:mo><mml:mi>u</mml:mi></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo>∂</mml:mo><mml:mo>⁡</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">⟩</mml:mo></mml:mrow></mml:mrow></mml:msqrt></mml:mrow></mml:math></inline-formula>; wall lengthscale <inline-formula><mml:math id="inf31"><mml:mrow><mml:msup><mml:mi>y</mml:mi><mml:mo>+</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mi>ν</mml:mi><mml:mo>/</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mi>τ</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> where the friction velocity is <inline-formula><mml:math id="inf32"><mml:mrow><mml:msub><mml:mi>u</mml:mi><mml:mi>τ</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mi>τ</mml:mi><mml:mo>/</mml:mo><mml:mi>ρ</mml:mi></mml:mrow></mml:msqrt></mml:mrow></mml:math></inline-formula> and the wall stress is <inline-formula><mml:math id="inf33"><mml:mrow><mml:mi>τ</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:mi>ρ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>ν</mml:mi><mml:mo>⁢</mml:mo><mml:mi>d</mml:mi><mml:mo>⁢</mml:mo><mml:mi>u</mml:mi></mml:mrow><mml:mo>/</mml:mo><mml:mi>d</mml:mi></mml:mrow><mml:mo>⁢</mml:mo><mml:mi>z</mml:mi></mml:mrow><mml:mo fence="true" stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>z</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula>; Reynolds number <inline-formula><mml:math id="inf34"><mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>⁢</mml:mo><mml:mi>e</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>H</mml:mi><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>/</mml:mo><mml:mi>ν</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> based on the centerline speed <inline-formula><mml:math id="inf35"><mml:mi>U</mml:mi></mml:math></inline-formula> and half height; Reynolds number <inline-formula><mml:math id="inf36"><mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>⁢</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>λ</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>U</mml:mi><mml:mo>⁢</mml:mo><mml:mi>λ</mml:mi></mml:mrow><mml:mo>/</mml:mo><mml:mi>ν</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> based on the centerline speed and the Taylor microscale <inline-formula><mml:math id="inf37"><mml:mi>λ</mml:mi></mml:math></inline-formula>; Schmidt number (Sc = <inline-formula><mml:math id="inf38"><mml:mrow><mml:mi>ν</mml:mi><mml:mo>/</mml:mo><mml:msub><mml:mi>κ</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> = Pe/Re); magnitude of velocity fluctuations <inline-formula><mml:math id="inf39"><mml:msup><mml:mi>u</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:math></inline-formula> relative to the centerline speed; large eddy turnover time <inline-formula><mml:math id="inf40"><mml:mrow><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mo>⁢</mml:mo><mml:msup><mml:mi>u</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:mrow></mml:math></inline-formula>. First row reports results in non dimensional units; second and third rows correspond to dimensional parameters in air and water assuming the velocity of the centerline is 50 cm/s in air and 12 cm/s in water.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom"><inline-formula><mml:math id="inf41"><mml:mi>L</mml:mi></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf42"><mml:mi>W</mml:mi></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf43"><mml:mi>H</mml:mi></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf44"><mml:mi>U</mml:mi></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf45"><mml:msub><mml:mi>U</mml:mi><mml:mi>b</mml:mi></mml:msub></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf46"><mml:mi>ν</mml:mi></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf47"><mml:msub><mml:mi>κ</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf48"><mml:mi>η</mml:mi></mml:math></inline-formula></th><th align="left" valign="bottom"><inline-formula><mml:math id="inf49"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:math></inline-formula></th></tr></thead><tbody><tr><td align="left" valign="bottom"/><td align="char" char="." valign="bottom">40</td><td align="char" char="." valign="bottom">8</td><td align="char" char="." valign="bottom">4</td><td align="char" char="." valign="bottom">32</td><td align="char" char="." valign="bottom">23</td><td align="char" char="." valign="bottom">1/250</td><td align="char" char="." valign="bottom">1/250</td><td align="char" char="." valign="bottom">0.006</td><td align="char" char="." valign="bottom">0.025</td></tr><tr><td align="left" valign="bottom">air</td><td align="char" char="." valign="bottom">9.50 m</td><td align="char" char="." valign="bottom">1.90 m</td><td align="char" char="." valign="bottom">0.96 m</td><td align="char" char="." valign="bottom">50 cm/s</td><td align="char" char="." valign="bottom">36 cm/s</td><td align="char" char="." valign="bottom">1.510<sup>-5</sup> m<sup>2</sup>/s</td><td align="char" char="." valign="bottom">1.510<sup>-5</sup> m<sup>2</sup>/s</td><td align="char" char="." valign="bottom">0.15 cm</td><td align="char" char="." valign="bottom">0.6 cm</td></tr><tr><td align="left" valign="bottom">water</td><td align="char" char="." valign="bottom">2.66 m</td><td align="char" char="." valign="bottom">0.53 m</td><td align="char" char="." valign="bottom">0.27 m</td><td align="char" char="." valign="bottom">12 cm/s</td><td align="char" char="." valign="bottom">8.6 cm/s</td><td align="left" valign="bottom">10<sup>-6</sup> m<sup>2</sup>/s</td><td align="left" valign="bottom">10<sup>-6</sup> m<sup>2</sup>/s</td><td align="char" char="." valign="bottom">0.04 cm</td><td align="char" char="." valign="bottom">0.2 cm</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"><inline-formula><mml:math id="inf50"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula></td><td align="left" valign="bottom">ε</td><td align="left" valign="bottom">λ</td><td align="left" valign="bottom">y<sup>+</sup></td><td align="left" valign="bottom">Re</td><td align="left" valign="bottom">Re<sub>λ</sub></td><td align="left" valign="bottom">Sc</td><td align="left" valign="bottom"><inline-formula><mml:math id="inf51"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msup><mml:mi>u</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>U</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula></td><td align="left" valign="bottom"><inline-formula><mml:math id="inf52"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">0.01</td><td align="left" valign="bottom">39</td><td align="left" valign="bottom">0.17</td><td align="left" valign="bottom">0.0035</td><td align="left" valign="bottom">16000</td><td align="left" valign="bottom">1360</td><td align="left" valign="bottom">1</td><td align="left" valign="bottom">11%</td><td align="left" valign="bottom"><inline-formula><mml:math id="inf53"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>64</mml:mn><mml:mspace width="thinmathspace"/><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula></td></tr><tr><td align="left" valign="bottom">air</td><td align="left" valign="bottom">0.15 s</td><td align="left" valign="bottom">6.3e-4 m<sup>2</sup>/s<sup>3</sup></td><td align="left" valign="bottom">4 cm</td><td align="left" valign="bottom">0.09 cm</td><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">water</td><td align="left" valign="bottom">0.18 s</td><td align="left" valign="bottom">3e-5 m<sup>2</sup>/s<sup>3</sup></td><td align="left" valign="bottom">1 cm</td><td align="left" valign="bottom">0.02 cm</td><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr></tbody></table></table-wrap><p>Do odor cues bear information about source location meters away from the source? To answer this question, we develop supervised machine learning algorithms that learn the relationship between the input (odor) and the distance from the source (output) from a large dataset of examples. In order to dissect what are the best predictors of source location and how ranking depends on the statistics of the odor, we need to detail more specifically the input and output of the algorithm.</p><p>To design the input we start with the odor concentration field <inline-formula><mml:math id="inf54"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>c</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">z</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> which varies stochastically in space and time as a result of turbulent transport. Here, <inline-formula><mml:math id="inf55"><mml:mrow><mml:mi mathvariant="bold">z</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is a location in the three dimensional space and <inline-formula><mml:math id="inf56"><mml:mi>t</mml:mi></mml:math></inline-formula> is time. We focus on a plane at a fixed height, and consider the conical region where odor can be detected, the ‘cone of detection’ (<xref ref-type="fig" rid="fig2">Figure 2a</xref>). We first compose time series of the odor field; each time series is indicated with <inline-formula><mml:math id="inf57"><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> and consists of the odor sampled at <inline-formula><mml:math id="inf58"><mml:mi>M</mml:mi></mml:math></inline-formula> equally spaced times with frequency <inline-formula><mml:math id="inf59"><mml:mi>ω</mml:mi></mml:math></inline-formula> at a discrete location <inline-formula><mml:math id="inf60"><mml:msub><mml:mi mathvariant="bold">z</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> within the cone of detection. Thus, each time series is a vector <inline-formula><mml:math id="inf61"><mml:mrow><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">z</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">z</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf62"><mml:mrow><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mo>/</mml:mo><mml:mi>ω</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> is the temporal span of the time series, or memory. From each time series, <inline-formula><mml:math id="inf63"><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> we calculate five features <inline-formula><mml:math id="inf64"><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:msubsup><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>5</mml:mn></mml:msubsup></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf65"><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:msubsup></mml:math></inline-formula> is the temporal average of the concentration during whiffs in the time series <inline-formula><mml:math id="inf66"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="bold">c</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>; $x_i^2$ is its average slope (time derivative of odor upon detection, averaged across whiffs within <inline-formula><mml:math id="inf67"><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>); <inline-formula><mml:math id="inf68"><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>3</mml:mn></mml:msubsup></mml:math></inline-formula> is the average duration of blanks (stretches of time when odor is below detection within <inline-formula><mml:math id="inf69"><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>); <inline-formula><mml:math id="inf70"><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>4</mml:mn></mml:msubsup></mml:math></inline-formula> is the average duration of whiffs (stretches of time when odor is above threshold within <inline-formula><mml:math id="inf71"><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>); and <inline-formula><mml:math id="inf72"><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>5</mml:mn></mml:msubsup></mml:math></inline-formula> is the intermittency factor (the fraction of time the time series <inline-formula><mml:math id="inf73"><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is above threshold). The detection threshold is defined adaptively as discussed in Materials and methods and <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>. Features <inline-formula><mml:math id="inf74"><mml:msup><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="inf75"><mml:msup><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> depend explicitly on odor concentration, whereas features <inline-formula><mml:math id="inf76"><mml:msup><mml:mi>x</mml:mi><mml:mn>3</mml:mn></mml:msup></mml:math></inline-formula>, <inline-formula><mml:math id="inf77"><mml:msup><mml:mi>x</mml:mi><mml:mn>4</mml:mn></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="inf78"><mml:msup><mml:mi>x</mml:mi><mml:mn>5</mml:mn></mml:msup></mml:math></inline-formula> only depend on when the odor is on or off, but not on its intensity. To remark this difference, we refer to <inline-formula><mml:math id="inf79"><mml:msup><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="inf80"><mml:msup><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula> as intensity features, and <inline-formula><mml:math id="inf81"><mml:msup><mml:mi>x</mml:mi><mml:mn>3</mml:mn></mml:msup></mml:math></inline-formula>, <inline-formula><mml:math id="inf82"><mml:msup><mml:mi>x</mml:mi><mml:mn>4</mml:mn></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="inf83"><mml:msup><mml:mi>x</mml:mi><mml:mn>5</mml:mn></mml:msup></mml:math></inline-formula> as timing features. Our input <inline-formula><mml:math id="inf84"><mml:mrow><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:msubsup><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is composed of d-dimensional vectors of features and we will focus on <inline-formula><mml:math id="inf85"><mml:mrow><mml:mi>d</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>. We seek to infer distance from the source, thus our output <inline-formula><mml:math id="inf86"><mml:mi>y</mml:mi></mml:math></inline-formula> is the coordinate of the sampling point <inline-formula><mml:math id="inf87"><mml:mi mathvariant="bold">z</mml:mi></mml:math></inline-formula> in the downwind direction, i.e. <inline-formula><mml:math id="inf88"><mml:mrow><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula>, with the source placed at the origin (see sketch in <xref ref-type="fig" rid="fig2">Figure 2a</xref>). We refer to the figure supplements for results in the crosswind direction, <inline-formula><mml:math id="inf89"><mml:mrow><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula>. We train the algorithm by providing <inline-formula><mml:math id="inf90"><mml:mi>N</mml:mi></mml:math></inline-formula> examples of input-output pairs <inline-formula><mml:math id="inf91"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> selected randomly from the full simulation, and obtain the function that connects input and output: <inline-formula><mml:math id="inf92"><mml:mrow><mml:mi>y</mml:mi><mml:mo>≈</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Individual features enable inference in two dimensions.</title><p>(<bold>a</bold>) Sketch of the geometry. (<bold>b</bold>) Test error <inline-formula><mml:math id="inf93"><mml:mi>χ</mml:mi></mml:math></inline-formula> for inference using individual features as input. (<bold>c</bold>) Predicted <italic>vs</italic> actual distance for inference. Prediction for representative test points (grey circles); 30–70th percentile (patch, same color code as in (<bold>b</bold>)); trivial prediction <inline-formula><mml:math id="inf94"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=&lt;</mml:mo><mml:mi>y</mml:mi><mml:msub><mml:mo>&gt;</mml:mo><mml:mrow><mml:mtext>test</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> (solid horizontal line, corresponds to <inline-formula><mml:math id="inf95"><mml:mrow><mml:mi>χ</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>); exact prediction (bisector, corresponds to <inline-formula><mml:math id="inf96"><mml:mrow><mml:mi>χ</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>); dispersion away from the bisector visualizes the prediction error. Results are obtained with a supervised learning algorithm based on regularized empirical risk minimization (Materials and methods). Each input datum <italic>x</italic><sub><italic>i</italic></sub> is one individual scalar feature computed from the time course of odor concentration measured at location <inline-formula><mml:math id="inf97"><mml:msub><mml:mi mathvariant="bold">z</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> at 100 evenly spaced time points with sampling frequency <inline-formula><mml:math id="inf98"><mml:mrow><mml:mi>ω</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf99"><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:math></inline-formula> is Kolmogorov time. The training/test set are composed of <inline-formula><mml:math id="inf100"><mml:mrow><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>5000</mml:mn></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf101"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>13500</mml:mn></mml:mrow></mml:math></inline-formula> data points, respectively.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Fixed vs adaptive threshold.</title><p>Test error using an adaptive threshold (left column) <italic>vs</italic> a fixed threshold (right column) for different simulations described in the main text. Adaptive thresholds are defined as a fraction of the local average concentration, <inline-formula><mml:math id="inf102"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>c</mml:mi><mml:msub><mml:mo>&gt;</mml:mo><mml:mrow><mml:mtext>local</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>, computed over the memory <inline-formula><mml:math id="inf103"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>ω</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>. Fixed thresholds are defined as a fraction of the global maximum concentration <italic>c</italic><sub>0</sub>. Results are robust with respect to the choice of adaptive threshold, whereas they vary considerably with the choice of fixed thresholds. Large fixed thresholds (marked with yellow squares) prevent odor detection in dilute regions that is far from the source or from the substrate.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig2-figsupp1-v1.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Linear least square algorithm.</title><p>Prediction error for a least squares algorithm assuming the target function is a linear function of the input (no regularization). Performance is poor regardless of the input features and the dataset (dataset A to E are defined as for <xref ref-type="fig" rid="fig5">Figure 5</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig2-figsupp2-v1.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Visualisation of a typical cross validation procedure.</title><p>Visualisation of a typical cross validation procedure, exemplifying the need for regularization for this problem. Left and right color plots show the error on the training set (left) and test set (right) as a function of the two hyperparameters <inline-formula><mml:math id="inf104"><mml:mi>λ</mml:mi></mml:math></inline-formula> (Tikhonov regularization parameter) and <inline-formula><mml:math id="inf105"><mml:mi>σ</mml:mi></mml:math></inline-formula> (width of the Gaussian kernel), see Materials and methods for more details. Bottom: test and training errors for <inline-formula><mml:math id="inf106"><mml:mrow><mml:mi>λ</mml:mi><mml:mo>→</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula> as a function of <inline-formula><mml:math id="inf107"><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula>. For large values of <inline-formula><mml:math id="inf108"><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula>, the solution overfits the data and for small values of <inline-formula><mml:math id="inf109"><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula>, the solution does not overfit but is unstable. Top: test and training errors for <inline-formula><mml:math id="inf110"><mml:mrow><mml:mi>σ</mml:mi><mml:mo>→</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula> as a function of <inline-formula><mml:math id="inf111"><mml:mi>λ</mml:mi></mml:math></inline-formula>. For large values of <inline-formula><mml:math id="inf112"><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula>, the solution overfits the data and for small values of <inline-formula><mml:math id="inf113"><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula>, the solution does not overfit but is unstable.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig2-figsupp3-v1.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Test error at source height, for prediction in the crosswind direction.</title><p>Symbols as in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig2-figsupp4-v1.tif"/></fig><fig id="fig2s5" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 5.</label><caption><title>Results for three source locations.</title><p>Results vary little when the source is moved downstream of its original position. Left: sketch of the three source positions and conical domain used for training. Center: test error at source height for two individual features (average and intermittency) as well as their combination, as indicated on x-axis. Right: test error for two individual features (average, grey line and intermittency, black line) at five different heights marked with <italic>a</italic> to <italic>e</italic> as in <xref ref-type="fig" rid="fig5">Figure 5</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig2-figsupp5-v1.tif"/></fig><fig id="fig2s6" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 6.</label><caption><title>Results of training with odor emanating from one source and testing over odor fields emanating from the other two.</title><p>The algorithm is robust to source location. Left: performance of the pair average concentration and intermittency. Center: performance of the individual feature average concentration. Right: performance of individual feature intermittency. Boxes extend from the <inline-formula><mml:math id="inf114"><mml:msup><mml:mn>25</mml:mn><mml:mrow><mml:mi>t</mml:mi><mml:mo>⁢</mml:mo><mml:mi>h</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> to the <inline-formula><mml:math id="inf115"><mml:msup><mml:mn>75</mml:mn><mml:mrow><mml:mi>t</mml:mi><mml:mo>⁢</mml:mo><mml:mi>h</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> percentile; dashed line: median. Green: training and test are performed over the same dataset; blue: test is performed over the simulation with odor source closest to the obstacle; dark red: test is over the dataset obtained with the middle source; yellow: test is over the dataset with the source furthest downstream. Training dataset are indicated on the <inline-formula><mml:math id="inf116"><mml:mi>x</mml:mi></mml:math></inline-formula>-axis.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig2-figsupp6-v1.tif"/></fig></fig-group><p>We propose a machine learning approach where the different odor features are ranked based on their predictive power, rather than their fitting properties. Different data-sets of odor/distance pairs are defined. The data-sets differ in the way odor measurements are represented in terms of feature vectors. For each data-set we learn a function to predict the distance to target given the corresponding odor features. The predictive power of each function, and corresponding set of features, is then assessed. More precisely, each data-set is split in a training and a test set, as custom in machine learning. Training sets are used to learn functions connecting odor to target location, whereas test sets are used to assess their prediction properties. The training/test split is crucial since the goal is to make good predictions on new, unseen points, that are not within the training sets. From a modeling perspective, a flexible nonlinear/nonparametric approach based on kernel methods is contrasted and shown to be superior to a simpler linear model (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). A careful protocol based on hold-out cross-validation is used to select the hyper-parameters of the considered learning models (we refer to Materials and methods and <xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref> for more details).</p><p>To illustrate the results we pick the two-dimensional plane at height <inline-formula><mml:math id="inf117"><mml:mrow><mml:mi>H</mml:mi><mml:mo>/</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:math></inline-formula> that contains the source. The first result is that individual features (<inline-formula><mml:math id="inf118"><mml:mrow><mml:mi>d</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>) bear useful information for two-dimensional source localization even at several meters from the source. Performance is quantified by the normalized squared error averaged over the <inline-formula><mml:math id="inf119"><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:math></inline-formula> points in the test set <inline-formula><mml:math id="inf120"><mml:mrow><mml:mi>χ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>/</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:msubsup><mml:msup><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. For this dataset, intensity features rank higher than timing features (<xref ref-type="fig" rid="fig2">Figure 2b–c</xref>), consistent with previous work <xref ref-type="bibr" rid="bib3">Atema, 1996</xref> and predictions are more accurate in the crosswind than in the downwind direction (compare with <xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>). For reference, a random guess with flat probability within the correct lower and upper bounds yields <inline-formula><mml:math id="inf121"><mml:mrow><mml:msub><mml:mi>χ</mml:mi><mml:mtext>random</mml:mtext></mml:msub><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:math></inline-formula>, whereas a target function <inline-formula><mml:math id="inf122"><mml:mrow><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mtext>trivial</mml:mtext></mml:msub><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">⟩</mml:mo></mml:mrow><mml:mtext>test</mml:mtext></mml:msub></mml:mrow></mml:math></inline-formula> that learns the average of the output over the test set yields <inline-formula><mml:math id="inf123"><mml:mrow><mml:msub><mml:mi>χ</mml:mi><mml:mtext>trivial</mml:mtext></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>.</p><p>In order to prove that the algorithm captures the dynamics of the scalar and not of the underlying velocity field, we realize three computational fluid dynamics simulations with the odor source at different locations downstream of the obstacle. Each source is placed at a different distance from the obstacle and thus feels different velocity fields, because the flow is inhomogeneous in the downstream direction. To perform a fair comparison we train the algorithm on points sampled from a conical domain based on the most downstream source; for the other two sources, we use an identical domain shifted upstream so that the vertex of the cone is located at the respective source location (see <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref> left). Performance at source height varies little over the three locations, demonstrating that the algorithm is learning the dynamics of the scalar and not of the carrying flow (see <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref> center). Additionally we demonstrate that, irrespective of source position, timing features acquire predictive power with height, whereas the opposite is observed for intensity features (see <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref> right). We discuss this property in further detail in the following. . Finally, we train the algorithm with the odor fields emanating from one source location, and test its performance over odor fields generated by the two other sources. <xref ref-type="fig" rid="fig2s6">Figure 2—figure supplement 6</xref> shows that performance of individual features varies little when test and training are performed over different dataset. Learning from pairs of features appears somewhat more sensitive to the details of the dataset. In the aggregate, the analysis corroborates that the algorithm captures odor dynamics and is rather insensitive to details of the underlying flow. In the rest of the manuscript we focus on the leftmost source, so as to exploit the full spatial range available from the simulations.</p><p>Next we analyze whether and how the sampling strategy affects performance and ranking of the features. Most results are shown for a memory of <inline-formula><mml:math id="inf124"><mml:mrow><mml:mrow><mml:mn>100</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow><mml:mo>≈</mml:mo><mml:mrow><mml:mpadded width="+1.7pt"><mml:mn>15</mml:mn></mml:mpadded><mml:mo>⁢</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>. Performance improves with longer memory (<xref ref-type="fig" rid="fig3">Figure 3a</xref>), because this allows to better average out noise and obtain more stable estimates of the features. But improvement follows a slow power law so that waiting for example 20 times longer yields predictions only about twice as precise. On the other hand, waiting as little as <inline-formula><mml:math id="inf125"><mml:mrow><mml:mrow><mml:mn>10</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow><mml:mo>≈</mml:mo><mml:mn>1.5</mml:mn></mml:mrow></mml:math></inline-formula> seconds still allows to make predictions, albeit less precise. We then verify whether performance may improve with a larger training set. Because we infer distance from an individual (scalar) feature, the problem is one dimensional and we find that a small number of training points, which we indicate with <inline-formula><mml:math id="inf126"><mml:mi>N</mml:mi></mml:math></inline-formula>, is sufficient to reach a plateau in prediction performance (<xref ref-type="fig" rid="fig3">Figure 3b</xref>). We choose <inline-formula><mml:math id="inf127"><mml:mrow><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>5000</mml:mn></mml:mrow></mml:math></inline-formula> training points, which is also robust to the case with more than one feature (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>). Finally, sampling more frequently than once per Kolmogorov time does not essentially affect the results nor ranking (<xref ref-type="fig" rid="fig3">Figure 3c</xref>). Similar results hold for the crosswind direction (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>The sampling strategy affects performance but not ranking.</title><p>(<bold>a</bold>) Error <inline-formula><mml:math id="inf128"><mml:mi>χ</mml:mi></mml:math></inline-formula> as a function of memory in units of Kolmogorov times <inline-formula><mml:math id="inf129"><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:math></inline-formula>; memory is defined as the duration of the time series of odor concentration <inline-formula><mml:math id="inf130"><mml:mrow><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">z</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">z</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> used to compute the five features <inline-formula><mml:math id="inf131"><mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:msubsup><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>5</mml:mn></mml:msubsup></mml:mrow></mml:math></inline-formula>, i.e. memory <inline-formula><mml:math id="inf132"><mml:mrow><mml:mi/><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mo>/</mml:mo><mml:mi>ω</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>. Red and pink: Performance using <inline-formula><mml:math id="inf133"><mml:mrow><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:msubsup></mml:mrow></mml:math></inline-formula> (average concentration) and <inline-formula><mml:math id="inf134"><mml:mrow><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mn>5</mml:mn></mml:msubsup></mml:mrow></mml:math></inline-formula> (intermittency factor). The number of training points and the frequency of sampling are fixed, <inline-formula><mml:math id="inf135"><mml:mrow><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>5000</mml:mn></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf136"><mml:mrow><mml:mi>ω</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>. Dotted, dashed and solid grey lines are power laws with exponents <inline-formula><mml:math id="inf137"><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="inf138"><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>10</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf139"><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> respectively to guide the eye. (<bold>b</bold>) Error as a function of number of points in the training set <inline-formula><mml:math id="inf140"><mml:mi>N</mml:mi></mml:math></inline-formula>, with <inline-formula><mml:math id="inf141"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>13500</mml:mn></mml:mrow></mml:math></inline-formula> points in the test set, memory <inline-formula><mml:math id="inf142"><mml:mrow><mml:mi/><mml:mo>=</mml:mo><mml:mrow><mml:mn>100</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf143"><mml:mrow><mml:mi>ω</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>. Color code as in (<bold>a</bold>). (<bold>c</bold>) Performance using the five individual features as input with <inline-formula><mml:math id="inf144"><mml:mrow><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>5000</mml:mn></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="inf145"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>13500</mml:mn></mml:mrow></mml:math></inline-formula>, memory <inline-formula><mml:math id="inf146"><mml:mrow><mml:mi/><mml:mo>=</mml:mo><mml:mrow><mml:mpadded width="+1.7pt"><mml:mn>100</mml:mn></mml:mpadded><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> sampling odor at frequency <inline-formula><mml:math id="inf147"><mml:mrow><mml:mi>ω</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> (empty bars) and <inline-formula><mml:math id="inf148"><mml:mrow><mml:mi>ω</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>10</mml:mn><mml:mo>/</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> (filled bars). Key shows color coding.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Prediction error as a function of the number of points in the training set for individual features and pairs of features.</title><p>Based on this analysis we chose N=5000.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig3-figsupp1-v1.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Effect of memory, sampling frequency and number of points in the training set, for prediction in the crosswind direction.</title><p>Symbols as in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig3-figsupp2-v1.tif"/></fig></fig-group><p>Pairing two observables improves performance in some cases, but not always. In fact, pairing two features of the same category results in little to no improvement (<xref ref-type="fig" rid="fig4">Figure 4</xref> and similarly for the crosswind direction, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>). In contrast, combining one intensity and one timing feature improves performance considerably, up to 65%. This result can be understood by mapping the error done by individual features in space (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>), showing that intensity and timing features are complementary, that is intensity features perform well in locations where timing features perform poorly.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Pairing one timing feature and one intensity feature considerably improves performance.</title><p>(<bold>a</bold>) Error <inline-formula><mml:math id="inf149"><mml:mi>χ</mml:mi></mml:math></inline-formula> obtained with individual features (full bars) and pairs of features (empty bars). Grey and black indicate pairings of two intensity features and two timing features respectively; green indicates mixed pairs of one timing and one intensity feature. (<bold>b</bold>) Performance (left) and relative improvement over the best of the two paired features (right). Results for the median (bottom) and the 95th percentile (top). Within each table plot, rows from bottom to top and columns from left to right are labeled by the 5 individual features: A (average, <inline-formula><mml:math id="inf150"><mml:msup><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msup></mml:math></inline-formula>), S (slope <inline-formula><mml:math id="inf151"><mml:msup><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula>), B (blanks <inline-formula><mml:math id="inf152"><mml:msup><mml:mi>x</mml:mi><mml:mn>3</mml:mn></mml:msup></mml:math></inline-formula>), W (whiffs <inline-formula><mml:math id="inf153"><mml:msup><mml:mi>x</mml:mi><mml:mn>4</mml:mn></mml:msup></mml:math></inline-formula>), I (intermittency <inline-formula><mml:math id="inf154"><mml:msup><mml:mi>x</mml:mi><mml:mn>5</mml:mn></mml:msup></mml:math></inline-formula>). Results with individual features are shown on the diagonal; results pairing feature <inline-formula><mml:math id="inf155"><mml:mi>i</mml:mi></mml:math></inline-formula> and feature <inline-formula><mml:math id="inf156"><mml:mi>j</mml:mi></mml:math></inline-formula> are shown at position <inline-formula><mml:math id="inf157"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. Mixed pairs provide both the best performance and the largest improvement over individual features.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Effect of pairing two individual features, for prediction in the crosswind direction.</title><p>Symbols as in <xref ref-type="fig" rid="fig4">Figure 4</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig4-figsupp1-v1.tif"/></fig></fig-group><p>We next seek to clarify whether the results depend on space. To this end we compose five different dataset, <italic>a</italic> to <italic>e</italic>, obtained by extracting odor snapshots from horizontal planes at source height (<italic>b</italic>), above the source (<italic>c</italic> to <italic>e</italic>), and below the source (<italic>a</italic>) (<xref ref-type="fig" rid="fig5">Figure 5a</xref>). From <italic>a</italic> to <italic>e</italic>, sparsity increases and intensity decreases (<xref ref-type="fig" rid="fig5">Figure 5b</xref>) simply because closer to the boundary, the air slows down and the odor accumulates. By analyzing performance across these dataset, we find that ranking of individual features shifts considerably. The two intensity features outperform all timing features when the dataset is not very sparse (dataset <italic>a-b</italic>, <xref ref-type="fig" rid="fig5">Figure 5c and d</xref> left). In contrast, two timing features (intermittency factor and blank duration) outperform all others for the more sparse and less intense dataset <italic>d-e</italic> (<xref ref-type="fig" rid="fig5">Figure 5c and d</xref> right). Whiff duration performs poorly in <italic>d-e</italic> because intermittency is too severe and whiffs are short in duration thus bear little information (the average whiff duration is 1–7 time steps in over 90% of the time series). Although the ranking of individual features shifts with height, pairing one intensity and one timing feature remains the most successful strategy across all heights (<xref ref-type="fig" rid="fig5">Figure 5c and d</xref>). In contrast, combining all five features contributes little improvement (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Ranking shifts with height from the ground.</title><p>(<bold>a</bold>) Datasets <italic>a</italic> to <italic>e</italic> correspond to data obtained at heights <inline-formula><mml:math id="inf158"><mml:mrow><mml:mrow><mml:mi>z</mml:mi><mml:mo>/</mml:mo><mml:mi>H</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mn>25</mml:mn><mml:mo>%</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, 37.5%, 50%, 55% and 65% respectively. (<bold>b</bold>) Distribution of intensities (top) and intermittency factors (bottom) over the training set from <italic>a</italic> to <italic>e</italic> (left to right). Moving away from the boundary, the odor becomes less intense and more sparse. (<bold>c</bold>) Median performance as a function of average intermittency factor of the training set for individual intensity (grey) and timing (black) features, mixed pairs of one intensity and one timing feature (green) and all five features together (dark green). (<bold>d</bold>) Predicted <italic>vs</italic> actual distance, to visualize a representative subset of the results in (<bold>c</bold>), scale bar <inline-formula><mml:math id="inf159"><mml:mrow><mml:msup><mml:mn>10</mml:mn><mml:mn>3</mml:mn></mml:msup><mml:mo>⁢</mml:mo><mml:mi>η</mml:mi></mml:mrow></mml:math></inline-formula>. Ranking depends sensibly on height: intensity features outperform timing features near the substrate, where there is more odor and it is more continuous; timing features outperform intensity features further from the substrate where there is less odor and it is more sparse; mixed pairs perform best across all conditions; combining five features provides little to no improvement over mixed pairs.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig5-v1.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Test error mapped in space.</title><p>Test error mapped in space for a kernel ridge regression algorithm that takes individual observables in input and predicts distance in the downwind (left) and crosswind (right) direction for simulation b. The test error is color coded from 0 (blue) to 2 (yellow) corresponding to the limits of perfect prediction and random prediction.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig5-figsupp1-v1.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>Predicted distance vs actual distance for simulation a (top) and (<bold>d</bold>) (bottom); symbols as in <xref ref-type="fig" rid="fig5">Figure 5d</xref>.</title><p>The bisector corresponds to a perfect prediction; departure from the bisector visualizes the error.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig5-figsupp2-v1.tif"/></fig><fig id="fig5s3" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 3.</label><caption><title>Estimated predicted power of individual features.</title><p>Estimated test error of individual features at different heights using the theoretical framework <xref ref-type="disp-formula" rid="equ6 equ7 equ8 equ9 equ10 equ11">Equations 5–10</xref> outlined in the text and Materials and methods, using an estimate of the likelihood obtained empirically from our data (not shown). Symbols as in <xref ref-type="fig" rid="fig5">Figure 5c</xref> (black / grey represent timing /intensity features; dashed lines: whiffs).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig5-figsupp3-v1.tif"/></fig></fig-group><p>Let us now focus on the plane at source height and separate locations based on their distance from the source. We assemble a distal dataset and a proximal dataset, composed of points that are further and closer than 2330η from the source respectively (<xref ref-type="fig" rid="fig6">Figure 6a</xref>). The odor is more intense and more sparse closer to the source and it becomes more dilute and less sparse with distance from the source (<xref ref-type="fig" rid="fig6">Figure 6b</xref>). Performance of individual features degrades with distance (<xref ref-type="fig" rid="fig6">Figure 6d</xref>). Intensity features clearly outperform timing features at close range, as seen both from various percentiles of the test error (<xref ref-type="fig" rid="fig6">Figure 6d</xref>, left) as well as the full distribution (<xref ref-type="fig" rid="fig6">Figure 6c</xref>, left). The disparity between timing and intensity features disappears in the distal problem: the error distribution for all individual features is essentially superimposed except for the tails (<xref ref-type="fig" rid="fig6">Figure 6c</xref>, right and inset), which cause small differences in the median and other percentiles of the error (<xref ref-type="fig" rid="fig6">Figure 6d</xref>, right). Remarkably, mixed pairs outperform all individual features in both the distal and proximal problems (<xref ref-type="fig" rid="fig6">Figure 6c–d</xref>). In the aggregate, results demonstrate that, even within a single turbulent flow, ranking shifts considerably. Namely, measuring timing of odor encounters is most useful in regions where the odor is dilute, that is far from the source and from the substrate, whereas measuring intensity is most useful in concentrated conditions, that is close to the source or the substrate.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Ranking depends on distance from the source.</title><p>(<bold>a</bold>) At source height, the dataset is split in proximal (distance &lt;2330η) and distal (distance &gt;2330η). (<bold>b</bold>) Distributions of average odor intensity (left) and intermittency factor (right) over the training set; closer to the source, the odor is more intense and more sparse. (<bold>c</bold>) Distribution of test error for the proximal (left) and distal (right) problem showing intensity features (grey) outperform timing features (black) at close range, but not in the distal problem where differences in the error distribution are limited to the tails (see insets). Mixed pairs of features (green) outperform individual features either marginally (left) or considerably (right). (<bold>d</bold>) Percentiles of the error distribution in (<bold>c</bold>) for the proximal (left) and distal (right) problems confirming the picture emerged from (<bold>c</bold>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-fig6-v1.tif"/></fig><p>Thus, the predictive power of different features varies greatly in space and time. Next, we show how this spatial variation is dictated by the statistical properties of the odor plume. To this end, we provide an analytical characterization of the test error of individual features, <inline-formula><mml:math id="inf160"><mml:mi>χ</mml:mi></mml:math></inline-formula>, that connects directly to the physics of the problem. Such a characterization is consistent with the observed variation in performance. Note that for large enough samples, the test error <inline-formula><mml:math id="inf161"><mml:mi>χ</mml:mi></mml:math></inline-formula> approximates the following expected error<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi mathvariant="normal">∞</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi>R</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>⁢</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>-</mml:mo><mml:mrow><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>⁢</mml:mo><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi>R</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>⁢</mml:mo><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:mrow></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>The latter is called expected error and is the ideal version of the test error. It can be interpreted as the error summed over all possible input-output pairs, weighted by their corresponding joint probability to be sampled. Here, <inline-formula><mml:math id="inf162"><mml:mrow><mml:mrow><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">|</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">⟩</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi>R</mml:mi></mml:msubsup><mml:mrow><mml:mi>y</mml:mi><mml:mo>⁢</mml:mo><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula> is the so called regression function, which minimizes the expected error among all possible functions. The regression function can be approximated using Kernel ridge regression and sufficiently rich kernels. Indeed, kernel ridge regression is known to be a so called universal estimator <xref ref-type="bibr" rid="bib21">Hastie et al., 2001</xref>; <xref ref-type="bibr" rid="bib50">Steinwart, 2002</xref>. In the above expression, <inline-formula><mml:math id="inf163"><mml:mi>R</mml:mi></mml:math></inline-formula> is the length of the cone, <inline-formula><mml:math id="inf164"><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is the conditional probability distribution of the output given the input, <inline-formula><mml:math id="inf165"><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is the joint probability distribution of the input output pair <inline-formula><mml:math id="inf166"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> is the prior distribution on the output <inline-formula><mml:math id="inf167"><mml:mi>y</mml:mi></mml:math></inline-formula>. The idea is to relate <inline-formula><mml:math id="inf168"><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:math></inline-formula>, hence the expected error, to the distribution <inline-formula><mml:math id="inf169"><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> of observing feature <inline-formula><mml:math id="inf170"><mml:mi>x</mml:mi></mml:math></inline-formula> depending on distance <inline-formula><mml:math id="inf171"><mml:mi>y</mml:mi></mml:math></inline-formula>. Indeed, the latter is dictated by the fluid dynamics of odor plumes from a concentrated source, and hence provides a more direct connection between the expected error and the physics of the problem. Since the considered features are sample averages, in the limit of large samples, their distribution is well approximated by a Gaussian, hence fully characterized by the mean and standard deviation. Then simple estimates can be computed empirically from our data. <xref ref-type="disp-formula" rid="equ1">Equation (1)</xref>, provided with these estimates, reproduces the predictive power of individual features showed in <xref ref-type="fig" rid="fig5">Figure 5c</xref> (see <xref ref-type="fig" rid="fig5s3">Figure 5—figure supplement 3</xref>). Note that whiffs deviate considerably from a normal distribution, hence the argument needs to be revised for this feature. To move beyond empirical estimates and extend the above reasoning to other flows and generic combinations of (possibly non Gaussian) features, a generalization of the asymptotic arguments proposed in <xref ref-type="bibr" rid="bib9">Celani et al., 2014</xref> is needed.</p></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Our results demonstrate that within the cone of detection, the time course of an odor bears useful information for source localization even at meters from the source. We find that the concentration and the slope of a turbulent odor signal, averaged over a memory lag, are particularly useful to predict source location at close range or near the boundary. These features quantify the intensity of the odor and its variation. The primacy of the intensity features wanes in more challenging conditions, for example moving away from the source or away from the boundary. In these portions of space, where the odor is scarcer, features that quantify timing of odor detection become as effective as intensity features, or more effective. One of the best studied example of olfactory search in dilute conditions is arguably the case of insects.</p><p>Interestingly, olfactory receptor neurons in insects appear to encode efficiently information about timing across a wide range of intensities <xref ref-type="bibr" rid="bib19">Gorur-Shandilya et al., 2017</xref>; <xref ref-type="bibr" rid="bib29">Martelli et al., 2013</xref>.</p><p>Note that while the statistics of an odor plume clearly depends on all details of the flow and the source, see e.g. <xref ref-type="bibr" rid="bib24">Justus et al., 2002</xref>; <xref ref-type="bibr" rid="bib9">Celani et al., 2014</xref>; <xref ref-type="bibr" rid="bib12">Fackrell and Robins, 1982</xref>, here we keep all of these parameters constant and demonstrate that even within a single flow, odor dynamics and the best predictors vary considerably in space. This begs the next question: do organisms switch between different modalities depending on attributes of odor dynamics, which will vary in space? This could be the case for mice, where the neural activity in the first relay of olfactory processing does in fact depend on how sparse is the odor <xref ref-type="bibr" rid="bib27">Lewis et al., 2021</xref>. Specifically, sparse odor cues elicit individual responses that follow closely the ups and downs of the odor in time. In contrast, continuous signals elicit intense responses which are however uncorrelated to the temporal dynamics of the odor itself <xref ref-type="bibr" rid="bib27">Lewis et al., 2021</xref>.</p><p>We find that features within the same class are redundant whereas features from different classes are complementary. Indeed, features of the same class have similar patterns of performance in space, but each class has a distinct pattern. As a consequence, measuring both timing and intensity is beneficial, but using more than one feature to quantify either timing or intensity provides no advantage. Combining all features does not improve over the performance of mixed pairs, consistent with redundancy within each class. Note that there is no fundamental reason to expect features from the same class to be redundant, and further work with a larger library of features is needed to prove or disprove this notion.</p><p>Importantly, mixed time/intensity pairs of features outrank individual features robustly, that is in all portions of space, regardless of distance from the source and from the ground. This is in contrast with individual features and suggests relying on simultaneous timing and intensity features is advantageous when odors are sensed at various distances from the source and from the substrate. Interestingly, the coexistence of bursting olfactory neurons and canonical olfactory neurons in lobsters suggests these animals are in fact able to measure simultaneously timing and intensity <xref ref-type="bibr" rid="bib36">Park et al., 2014</xref>; <xref ref-type="bibr" rid="bib1">Ache et al., 2016</xref>, which is consistent with the increased predictive power of the mixed pairs of features. Similarly, in mammals, optogenetic activation of the olfactory bulb <xref ref-type="bibr" rid="bib48">Smear et al., 2013</xref> demonstrates that both kinds of measures guide behavior (lick <italic>vs</italic> no lick).</p><p>In this work, we have investigated the problem of predicting the location of a target from measures of the time course of a turbulent odor. Previous work explored a related question, that is how to best represent instantaneous snapshots of the odor to encode maximum information about source location <xref ref-type="bibr" rid="bib55">Victor et al., 2019</xref>. The two approaches are not immediately comparable: first, <xref ref-type="bibr" rid="bib55">Victor et al., 2019</xref> consider few snapshots of the odor, rather than measures of its time course. Second, maximizing information does not guarantee good predictions (to make predictions information needs to be extracted and processed, and importantly the focus is on new data that were not previously seen). We provide two comments that are relevant if information is the limiting factor for prediction accuracy: (<italic>i</italic>) binary representations were suboptimal in all conditions considered in <xref ref-type="bibr" rid="bib55">Victor et al., 2019</xref>; <xref ref-type="bibr" rid="bib6">Boie et al., 2018</xref>, that is at few tens of cm from the source. This is consistent with our results in concentrated conditions, where timing features -accessible through binary representations- are suboptimal. Our evidences suggest, however, that the result may not hold in more dilute conditions, where the gap between binary and more accurate representations should become increasingly small. (<italic>ii</italic>) Individual snapshots of odor from <xref ref-type="bibr" rid="bib55">Victor et al., 2019</xref>; <xref ref-type="bibr" rid="bib6">Boie et al., 2018</xref> contained 1–2 bits of information about source location, but allocating more resources to represent how the odor varies in time was found informative <xref ref-type="bibr" rid="bib55">Victor et al., 2019</xref>; <xref ref-type="bibr" rid="bib6">Boie et al., 2018</xref>. Our mixed pairs of features at close range achieve precisions of 5–6%, corresponding to coding for position with words of 4–4.3 bits. Our results thus confirm that memory is indeed useful, but the gain does not increase indefinitely with further memory.</p><p>The literature on olfactory navigation is vast. Although a complete review of available algorithms is beyond the scope of the present work, we remark that recent results investigated gradient descent algorithms using either concentration alone <xref ref-type="bibr" rid="bib18">Gire et al., 2016</xref>, or various measures of timing and intensity <xref ref-type="bibr" rid="bib37">Park et al., 2016</xref>; <xref ref-type="bibr" rid="bib30">Michaelis et al., 2020</xref>; <xref ref-type="bibr" rid="bib26">Leathers et al., 2020</xref>. Overall, both intensity and timing appear to have a potential to lead to an odor source, consistent with our results on individual features. A combination of the two kinds of features was found beneficial in <xref ref-type="bibr" rid="bib26">Leathers et al., 2020</xref>, consistent with our results on mixed pairs. Note that predicting source location and navigating to reach it are distinct tasks. Although they are often assumed to be intimately connected, whether good predictors may be good variables for navigation in more general contexts remains to be understood.</p><p>Here, we have analyzed the features that enable the most accurate prediction of source location. We add a few observations about the significance of the results for animal behavior. First: whether animals rely on features from either class will depend on what features best support behavior. It is often implicitly assumed that features that bear reliable information on source location are also the most useful for navigation. However, this connection between prediction and navigation is far from straightforward and more work is needed to establish whether accurate predictions imply efficient navigation. Second: animals are unlikely to have prior information on the details of the odor source, for example its intensity. Timing features are more robust than intensity features with respect to the intensity of the source and may thus be favored regardless of their performance, which was argued in <xref ref-type="bibr" rid="bib45">Schmuker et al., 2016</xref>. In our work, timing features are precisely invariant with source intensity because we define the detection threshold adaptively (see Materials and methods). More realistic conditions will need to be evaluated, where dependence on source intensity emerges as a result of non-linearities that we did not model in this work. These effects emerge for example, close to a boundary which partially absorbs the odor <xref ref-type="bibr" rid="bib20">Gorur-Shandilya et al., 2019</xref>, or in the case of fixed thresholds, although this dependence is weak in the far field where timing features are most useful <xref ref-type="bibr" rid="bib9">Celani et al., 2014</xref>. Third: we have focused on predicting source location from within the cone of detection, where an agent will detect the odor quite often. However, a crucial difficulty of turbulent navigation is to find the cone itself. We cannot address the problem of predicting source location from outside the cone because detections are so rare that we lack statistics. The distinction between inside and outside the cone of detection is key for navigation with sparse cues (see <xref ref-type="bibr" rid="bib40">Reddy et al., 2021</xref>) and deserves further attention.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Direct numerical simulations of turbulent odor plumes</title><p>To reproduce a realistic odor landscape and generate the dataset showed in <xref ref-type="fig" rid="fig1">Figure 1</xref>, we solve the Navier-Stokes (2) and the advection-diffusion equation for passive odor transport (3) at all relevant scales of motion from the smallest turbulent eddies (Kolmogorov scale <inline-formula><mml:math id="inf172"><mml:mi>η</mml:mi></mml:math></inline-formula>) to the integral scale (<inline-formula><mml:math id="inf173"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>L</mml:mi><mml:mo>&gt;</mml:mo><mml:mn>600</mml:mn><mml:mi>η</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>), using Direct numerical simulations (DNS):<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:msub><mml:mo>∂</mml:mo><mml:mi>t</mml:mi></mml:msub><mml:mo>⁡</mml:mo><mml:mi>u</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi>u</mml:mi><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>∇</mml:mo><mml:mo>⁡</mml:mo><mml:mi>u</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>ρ</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>∇</mml:mo><mml:mo>⁡</mml:mo><mml:mi>P</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi>ν</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:msup><mml:mo>∇</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>⁡</mml:mo><mml:mi>u</mml:mi></mml:mrow></mml:mrow></mml:mrow></mml:mrow><mml:mo mathvariant="italic" separator="true">  </mml:mo><mml:mrow><mml:mrow><mml:mo>∇</mml:mo><mml:mo>⋅</mml:mo><mml:mi>u</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mrow></mml:math></disp-formula><disp-formula id="equ3"> <label>(3)</label><mml:math id="m3"><mml:mrow><mml:mrow><mml:mrow><mml:msub><mml:mo>∂</mml:mo><mml:mi>t</mml:mi></mml:msub><mml:mo>⁡</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi>u</mml:mi><mml:mo>⋅</mml:mo><mml:mrow><mml:mo>∇</mml:mo><mml:mo>⁡</mml:mo><mml:mi>c</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>κ</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>⁢</mml:mo><mml:mrow><mml:msup><mml:mo>∇</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>⁡</mml:mo><mml:mi>c</mml:mi></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mi>q</mml:mi></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf174"><mml:mi mathvariant="bold">u</mml:mi></mml:math></inline-formula> is the velocity field, <inline-formula><mml:math id="inf175"><mml:mi>ρ</mml:mi></mml:math></inline-formula> is the fluid density, <inline-formula><mml:math id="inf176"><mml:mi>P</mml:mi></mml:math></inline-formula> is pressure, <inline-formula><mml:math id="inf177"><mml:mi>ν</mml:mi></mml:math></inline-formula> is the fluid kinematic viscosity, <inline-formula><mml:math id="inf178"><mml:mi>c</mml:mi></mml:math></inline-formula> is the odor concentration, <inline-formula><mml:math id="inf179"><mml:msub><mml:mi>κ</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:math></inline-formula> is its diffusivity and <inline-formula><mml:math id="inf180"><mml:mi>q</mml:mi></mml:math></inline-formula> an odor source. All parameters are listed in <xref ref-type="table" rid="table1">Table 1</xref>. Note that we use <inline-formula><mml:math id="inf181"><mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mo>⁢</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula> which is appropriate for typical odors in air but not in water. However, we expect a weak dependence on the Schmidt number as the Batchelor and Kolmogorov scales are below the size of the source and we are interested in the large scale statistics <xref ref-type="bibr" rid="bib13">Falkovich et al., 2001</xref>; <xref ref-type="bibr" rid="bib9">Celani et al., 2014</xref>; <xref ref-type="bibr" rid="bib11">Duplat et al., 2010</xref>. We simulate a turbulent channel flow with a concentrated odor source and an obstacle that generates turbulence by customizing the open-source software Nek5000 <xref ref-type="bibr" rid="bib16">Fischer et al., 2008</xref> developed at Argonne National Laboratory, Illinois. Nek5000 employs a spectral element method (SEM) <xref ref-type="bibr" rid="bib38">Patera, 1984</xref> <xref ref-type="bibr" rid="bib34">Orszag, 1980</xref> based on Legendre polynomials for discretization <xref ref-type="bibr" rid="bib22">Ho, 1989</xref>, and a 4th order Runge-Kutta scheme for time marching. The code is written in fortran77 and C and it uses MPI for parallelization.</p><p>The three-dimensional channel is divided in <inline-formula><mml:math id="inf182"><mml:mrow><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mn>160 000</mml:mn></mml:mrow></mml:math></inline-formula> discrete elements: <inline-formula><mml:math id="inf183"><mml:mrow><mml:mn>200</mml:mn><mml:mo>×</mml:mo><mml:mn>40</mml:mn><mml:mo>×</mml:mo><mml:mn>20</mml:mn></mml:mrow></mml:math></inline-formula> (number of elements in length × width × height); within each element the solution is expanded in 8th grade tensor-product polynomials so that the domain is effectively discretized in 81 920,000 elements. The average spatial resolution is equal in each direction <inline-formula><mml:math id="inf184"><mml:mrow><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo>≈</mml:mo><mml:mrow><mml:mn>4</mml:mn><mml:mo>⁢</mml:mo><mml:mi>η</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>. A cylindrical cap of height <inline-formula><mml:math id="inf185"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>=</mml:mo><mml:mn>160</mml:mn><mml:mspace width="thinmathspace"/><mml:mi>η</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> is added on the ground; the cylinder spans the entire width of the channel. The mesh is adapted to fit the cylinder. Fluid flows from left to right and the obstacle generates turbulence in the channel, in particular the height of the cylinder tunes the velocity fluctuations. The velocity fluctuations are defined as <inline-formula><mml:math id="inf186"><mml:mrow><mml:mrow><mml:mi>δ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>u</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">z</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">z</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:mrow><mml:mi>u</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">z</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">⟩</mml:mo></mml:mrow><mml:mi>y</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula>; their intensity is <inline-formula><mml:math id="inf187"><mml:mrow><mml:msup><mml:mi>u</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>δ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>u</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy="false">⟩</mml:mo></mml:mrow></mml:msqrt></mml:mrow></mml:math></inline-formula>, where averages are intended in space and time. <xref ref-type="table" rid="table1">Table 1</xref> summarizes the parameters that characterize turbulence.</p><p>Each simulation runs for 300,000 time steps where <inline-formula><mml:math id="inf188"><mml:mrow><mml:mrow><mml:mi>δ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> and follows from a severe Courant criterium with <inline-formula><mml:math id="inf189"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>U</mml:mi><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>x</mml:mi><mml:mo>&lt;</mml:mo><mml:mn>0.4</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> to ensure convergence of both the velocity and scalar fields. Snapshots of velocity and odor fields are saved at constant frequency <inline-formula><mml:math id="inf190"><mml:mrow><mml:mi>ω</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> (except for results in <xref ref-type="fig" rid="fig3">Figure 3c</xref> where snapshots are saved 10 times more frequently). Each DNS requires 2 weeks of computational time using 320 cpus.</p></sec><sec id="s4-2"><title>Boundary conditions and odor source</title><p>We impose a Poiseuille velocity profile at the inlet: <inline-formula><mml:math id="inf191"><mml:mrow><mml:mi mathvariant="bold">u</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>u</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf192"><mml:mrow><mml:mi>u</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>6</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>U</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>ζ</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mi>ζ</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf193"><mml:mrow><mml:mi>ζ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo>/</mml:mo><mml:mi>H</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula> is the vertical coordinate normalized to the height of the channel and <inline-formula><mml:math id="inf194"><mml:msub><mml:mi>U</mml:mi><mml:mi>b</mml:mi></mml:msub></mml:math></inline-formula> is the mean speed. We set a no-slip condition <inline-formula><mml:math id="inf195"><mml:mrow><mml:mi mathvariant="bold">u</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula> at the ground and on the obstacle; on the remaining boundaries we impose the turbulent outflow condition defined in <xref ref-type="bibr" rid="bib15">Fischer et al., 2007</xref> that imposes a positive exit velocity to avoid potential negative flux and the consequent instability it generates.</p><p>More precisely, the divergence ramps up from zero to a positive value along the element closest to the boundary: <inline-formula><mml:math id="inf196"><mml:mrow><mml:mrow><mml:mo>∇</mml:mo><mml:mo>⋅</mml:mo><mml:mi mathvariant="bold">u</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi>C</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mo>⟂</mml:mo></mml:msub><mml:mo>/</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi></mml:mrow><mml:mo>⁢</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf197"><mml:msub><mml:mi>z</mml:mi><mml:mo>⟂</mml:mo></mml:msub></mml:math></inline-formula> is the distance from the boundary and <inline-formula><mml:math id="inf198"><mml:mrow><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:math></inline-formula> is the minimal value that ensures convergence. For the odor, we impose a Dirichlet condition (<inline-formula><mml:math id="inf199"><mml:mrow><mml:mi>c</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>) at the ground, on the obstacle and at the inlet; while an outflow condition is set at the top, on the sides and at the outlet: <inline-formula><mml:math id="inf200"><mml:mrow><mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>∇</mml:mo><mml:mo>⁡</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>⋅</mml:mo><mml:mi mathvariant="bold">n</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>. We introduce a source located right above and downstream of the obstacle, at coordinates <italic>x</italic><sub><italic>s</italic></sub> =810η, <italic>y</italic><sub><italic>s</italic></sub> =650η, <italic>z</italic><sub><italic>s</italic></sub> =238η; odor intensity at the source is defined by a gaussian distribution <inline-formula><mml:math id="inf201"><mml:mrow><mml:mi>q</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>z</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mo>⁢</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf202"><mml:mrow><mml:mi>σ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>5</mml:mn><mml:mo>⁢</mml:mo><mml:mi>η</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>.</p></sec><sec id="s4-3"><title>Machine learning</title><p>To learn the correct position of a target source given an odor, we propose to use supervised machine learning. We next review some key ideas, and refer to standard textbooks for further details for example <xref ref-type="bibr" rid="bib21">Hastie et al., 2001</xref>.</p><p>The goal in supervised learning is to infer a function <inline-formula><mml:math id="inf203"><mml:mi>f</mml:mi></mml:math></inline-formula> given a training set <inline-formula><mml:math id="inf204"><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mi mathvariant="normal">…</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula> of input/output pairs. A good function estimate should allow to <italic>predict</italic> the outputs associated to <italic>new</italic> input points. In our setting each input <inline-formula><mml:math id="inf205"><mml:mi mathvariant="bold">x</mml:mi></mml:math></inline-formula> is a one-, two-, or five-dimensional vector whose entries are scalar features of odor time series, where the odor is sampled at a specific spatial location. From every sampling location, we compute the distance to the source and this distance is the output <inline-formula><mml:math id="inf206"><mml:mi>y</mml:mi></mml:math></inline-formula>.</p><p>To measure how close the prediction <inline-formula><mml:math id="inf207"><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is to the correct output <inline-formula><mml:math id="inf208"><mml:mi>y</mml:mi></mml:math></inline-formula>, we consider the square loss <inline-formula><mml:math id="inf209"><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula>. Following a statistical learning framework, the data are assumed to be sampled according to a fixed but unknown data distribution <inline-formula><mml:math id="inf210"><mml:mi>P</mml:mi></mml:math></inline-formula>. In this view, the ideal solution <inline-formula><mml:math id="inf211"><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:math></inline-formula> should minimize the expected loss <inline-formula><mml:math id="inf212"><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">⟩</mml:mo></mml:mrow></mml:math></inline-formula> over all data distributed according to <inline-formula><mml:math id="inf213"><mml:mi>P</mml:mi></mml:math></inline-formula>. In practice, only an empirical loss based on training data can be measured, and the search for a solution needs be restricted to a suitable class of hypothesis. Note that, the choice of the latter is critical since the nature of the function to be learnt is not known a priori. A basic choice is considering linear functions <inline-formula><mml:math id="inf214"><mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="bold">w</mml:mi><mml:mo>⋅</mml:mo><mml:mi mathvariant="bold">x</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>. In this case, minimizing the training loss reduces to linear least squares <inline-formula><mml:math id="inf215"><mml:mrow><mml:mtext>min</mml:mtext><mml:mo>⁢</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:msup><mml:mrow><mml:mo fence="true">||</mml:mo><mml:mrow><mml:mi>Y</mml:mi><mml:mo>-</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mo>⋅</mml:mo><mml:mi mathvariant="bold">w</mml:mi></mml:mrow></mml:mrow><mml:mo fence="true">||</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="inf216"><mml:mi>X</mml:mi></mml:math></inline-formula> is the matrix composed of the <inline-formula><mml:math id="inf217"><mml:mi>N</mml:mi></mml:math></inline-formula> training data input <inline-formula><mml:math id="inf218"><mml:mrow><mml:mi>X</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf219"><mml:mi>Y</mml:mi></mml:math></inline-formula> is the vector composed of the <inline-formula><mml:math id="inf220"><mml:mi>N</mml:mi></mml:math></inline-formula> labels of the training set <inline-formula><mml:math id="inf221"><mml:mrow><mml:mi>Y</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>. The corresponding solution is easily shown to be <inline-formula><mml:math id="inf222"><mml:mrow><mml:mi mathvariant="bold">w</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mi>X</mml:mi><mml:mi>T</mml:mi></mml:msup><mml:mo>⁢</mml:mo><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>⁢</mml:mo><mml:msup><mml:mi>X</mml:mi><mml:mi>T</mml:mi></mml:msup><mml:mo>⁢</mml:mo><mml:mi>Y</mml:mi></mml:mrow></mml:mrow></mml:math></inline-formula>. In <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>, we show that the choice of linear models has limited predictive power and does not allow to rank features. To tackle this issue, we consider kernel methods <xref ref-type="bibr" rid="bib46">Schölkopf and Smola, 2002</xref>, a more powerful class of nonlinear models corresponding to functions of the form <inline-formula><mml:math id="inf223"><mml:mrow><mml:mrow><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mi>k</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></inline-formula> Here, <inline-formula><mml:math id="inf224"><mml:mrow><mml:mi>k</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mi mathvariant="bold">x</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is a so called kernel, that here we will choose to be the Gaussian kernel <inline-formula><mml:math id="inf225"><mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="bold">x</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mi mathvariant="bold">x</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mo>∥</mml:mo><mml:mrow><mml:mi mathvariant="bold">x</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mi mathvariant="bold">x</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:mrow><mml:mo>∥</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>/</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mo>⁢</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>. The coefficients <inline-formula><mml:math id="inf226"><mml:mrow><mml:mi mathvariant="bold">c</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">c</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> are given by the expression<disp-formula id="equ4"> <label>(4)</label><mml:math id="m4"><mml:mrow><mml:mi>c</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>K</mml:mi><mml:mo>+</mml:mo><mml:mrow><mml:mi>λ</mml:mi><mml:mo>⁢</mml:mo><mml:mi>N</mml:mi><mml:mo>⁢</mml:mo><mml:mi>I</mml:mi></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>⁢</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>which minimizes<disp-formula id="equ5"><mml:math id="m5"><mml:mrow><mml:mrow><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>∥</mml:mo><mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mo>⁢</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>∥</mml:mo></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi>λ</mml:mi><mml:mo>⁢</mml:mo><mml:msup><mml:mi>c</mml:mi><mml:mo>⊤</mml:mo></mml:msup><mml:mo>⁢</mml:mo><mml:mi>K</mml:mi><mml:mo>⁢</mml:mo><mml:mi>c</mml:mi></mml:mrow></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>In the above expression, <inline-formula><mml:math id="inf227"><mml:mi>K</mml:mi></mml:math></inline-formula> is the <inline-formula><mml:math id="inf228"><mml:mi>N</mml:mi></mml:math></inline-formula> by <inline-formula><mml:math id="inf229"><mml:mi>N</mml:mi></mml:math></inline-formula> matrix with entries <inline-formula><mml:math id="inf230"><mml:mrow><mml:msub><mml:mi>K</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>⁢</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. The first term can be shown to be a data fit term whereas the second term can be shown to control the regularity of the obtained solution <xref ref-type="bibr" rid="bib46">Schölkopf and Smola, 2002</xref>. The <italic>regularization parameter</italic><inline-formula><mml:math id="inf231"><mml:mi>λ</mml:mi></mml:math></inline-formula> balances out the two terms and needs be tuned, together with the kernel parameters (the Gaussian width <inline-formula><mml:math id="inf232"><mml:mi>σ</mml:mi></mml:math></inline-formula> in our case).</p><p>Kernel methods offer a number of advantages. They are nonlinear, and hence can learn a wide range of complex input/output behavior. They are an example of nonparametric models, where the complexity of the model can adapt to the problem at hand and indeed learn any kind of continuous function provided enough data. This can be contrasted to linear models that clearly cannot learn any nonlinear function. Moreover, by tuning the hyper-parameters <inline-formula><mml:math id="inf233"><mml:mrow><mml:mi>λ</mml:mi><mml:mo>,</mml:mo><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula> more or less complex shape can be selected. When <inline-formula><mml:math id="inf234"><mml:mi>λ</mml:mi></mml:math></inline-formula> is small we are simply fitting the data, possibly at the price of stability, whereas for large <inline-formula><mml:math id="inf235"><mml:mi>λ</mml:mi></mml:math></inline-formula> we are favoring simpler models. With small <inline-formula><mml:math id="inf236"><mml:mi>σ</mml:mi></mml:math></inline-formula> we allow highly varying functions, whereas with large enough <inline-formula><mml:math id="inf237"><mml:mi>σ</mml:mi></mml:math></inline-formula> we essentially recover linear models.</p><p>Indeed, the choice of these parameters is crucial and tested and visualized in <xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>. Here it is shown that for <inline-formula><mml:math id="inf238"><mml:mrow><mml:mi>λ</mml:mi><mml:mo>→</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:math></inline-formula>, the solution incurs in the well known stability issues for large <inline-formula><mml:math id="inf239"><mml:mi>σ</mml:mi></mml:math></inline-formula> and overfitting issues for small <inline-formula><mml:math id="inf240"><mml:mi>σ</mml:mi></mml:math></inline-formula>. We note that ideally one would want to choose these hyper-parameters minimizing the test error; however, this would lead to overoptimistic estimates of the prediction properties of the obtained model. Hence, we consider a hold-out cross validation protocol, where the training data are further split in a training and a validation sets. The new training set is used to compute solutions corresponding to different hyper-parameters. The validation set is used as a proxy for the text error to select the hyper-parameters with small corresponding error. The prediction properties of the model thus tuned is then assessed on the test set.</p></sec><sec id="s4-4"><title>Dataset</title><p>To compose the dataset for regression we first extract two-dimensional snapshots of odor at fixed height from the 3D simulation. Each snapshot from the simulation has dimensions <inline-formula><mml:math id="inf241"><mml:mrow><mml:mn>1600</mml:mn><mml:mo>×</mml:mo><mml:mn>320</mml:mn></mml:mrow></mml:math></inline-formula> (number of points in the downwind direction × crosswind direction). The initial evolution up to <inline-formula><mml:math id="inf242"><mml:mrow><mml:mpadded width="+1.7pt"><mml:mn>300</mml:mn></mml:mpadded><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is excluded from the analysis as odor has not yet reached a stationary state. At stationary state we save 2700 frames at frequency <inline-formula><mml:math id="inf243"><mml:mrow><mml:mi>ω</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> per simulation. Thus at each spatial location we have the entire time evolution composed of 2700 time points at regular intervals of <inline-formula><mml:math id="inf244"><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:math></inline-formula>. We partition each simulation in fragments with <inline-formula><mml:math id="inf245"><mml:mi>M</mml:mi></mml:math></inline-formula> snapshots (duration <inline-formula><mml:math id="inf246"><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>). Most simulations are shown for <inline-formula><mml:math id="inf247"><mml:mrow><mml:mi>M</mml:mi><mml:mo>=</mml:mo><mml:mn>100</mml:mn></mml:mrow></mml:math></inline-formula>, thus for each spatial location we have 27 time series of the same duration (except for results leading to <xref ref-type="fig" rid="fig3">Figure 3a</xref>, where we vary memory from <inline-formula><mml:math id="inf248"><mml:mrow><mml:mn>10</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> to <inline-formula><mml:math id="inf249"><mml:mrow><mml:mn>250</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> resulting in 270 to 10 time series per location respectively).</p><p>The characteristic shape of the odor plume is a cone (<xref ref-type="fig" rid="fig2">Figure 2</xref>), that we defined as the region where the probability of detection computed over the entire simulation is larger than 0.35. The training set and test set are obtained by extracting <inline-formula><mml:math id="inf250"><mml:mrow><mml:mi>N</mml:mi><mml:mo>=</mml:mo><mml:mn>5000</mml:mn></mml:mrow></mml:math></inline-formula> (unless otherwise stated) and <inline-formula><mml:math id="inf251"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mtext>t</mml:mtext></mml:msub><mml:mo>=</mml:mo><mml:mn>13500</mml:mn></mml:mrow></mml:math></inline-formula> time series portions of duration <inline-formula><mml:math id="inf252"><mml:mrow><mml:mi>M</mml:mi><mml:mo>⁢</mml:mo><mml:msub><mml:mi>τ</mml:mi><mml:mi>η</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. To select these <inline-formula><mml:math id="inf253"><mml:mi>M</mml:mi></mml:math></inline-formula>-long time series, we extract random locations <inline-formula><mml:math id="inf254"><mml:msub><mml:mi mathvariant="bold">z</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> to cover homogeneously the cone, i.e. with flat probability within the cone, and random initial times <italic>t</italic><sub><italic>i</italic></sub>, with the training in the first half of the time history and the test in the second half of the time history. Time series that remain entirely under threshold are excluded.</p><p>Each odor time series is further processed by computing five features, two of which quantify intensity of the odor and rely on a precise representation of odor concentration (average concentration and average peak slope) and three of which quantify timing of odor encounters and are computed after binarizing the odor (average whiff and blank duration and intermittency factor). The threshold <inline-formula><mml:math id="inf255"><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>⁢</mml:mo><mml:mi>h</mml:mi><mml:mo>⁢</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> used for binarization is adaptive i.e. <inline-formula><mml:math id="inf256"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn><mml:mo fence="false" stretchy="false">⟨</mml:mo><mml:mi>c</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>c</mml:mi><mml:mo>&gt;</mml:mo><mml:mn>0</mml:mn><mml:mo fence="false" stretchy="false">⟩</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>, where the average is computed over each time series separately. The threshold thus varies from <inline-formula><mml:math id="inf257"><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>⁢</mml:mo><mml:mi>h</mml:mi><mml:mo>⁢</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mn>0.5</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> at the source to <inline-formula><mml:math id="inf258"><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>⁢</mml:mo><mml:mi>h</mml:mi><mml:mo>⁢</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msup><mml:mo>⁢</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> at the farthest edges of the cone, where <italic>c</italic><sub>0</sub> is the concentration at the source. The choice of an adaptive threshold was suggested in <xref ref-type="bibr" rid="bib19">Gorur-Shandilya et al., 2017</xref>. The precise value of the relative threshold has little effect on the results as shown in <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>, left. Fixed thresholds were tested and discarded because results depend sensibly on the threshold and the optimal threshold varies with the dataset in non-trivial ways (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>, right). Finally, adaptive thresholds that are defined based on purely local information appear more plausible for a biological system that has no information on the intensity of the source.</p><p>The parameters <inline-formula><mml:math id="inf259"><mml:mi>λ</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="inf260"><mml:mi>σ</mml:mi></mml:math></inline-formula> are obtained through 4-folds cross validation: the training set is split in 4 equal parts, 3 are used for training and 1 for validation. The empirical risk is computed on the validation set and averaged over the 4 possible permutations, systematically varying the hyperparameters <inline-formula><mml:math id="inf261"><mml:mrow><mml:mi>λ</mml:mi><mml:mo>,</mml:mo><mml:mi>σ</mml:mi></mml:mrow></mml:math></inline-formula>. The couple of hyperparameters that minimize the empirical risk over the validation set is selected through grid search using an <inline-formula><mml:math id="inf262"><mml:mrow><mml:mn>8</mml:mn><mml:mo>×</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:math></inline-formula> regular grid and further refined with a <inline-formula><mml:math id="inf263"><mml:mrow><mml:mn>4</mml:mn><mml:mo>×</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:math></inline-formula> subgrid. Results are insensitive to further refinement because there is a large plateau around the minimum, as shown in <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>. The optimal hyperparameters are used to compute the solution (4). The error <inline-formula><mml:math id="inf264"><mml:mi>χ</mml:mi></mml:math></inline-formula> used throughout the manuscript is simply the normalized test error <inline-formula><mml:math id="inf265"><mml:mrow><mml:mi>χ</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi mathvariant="bold">x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>/</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:msubsup><mml:msup><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. For most of the figures, we used 5000 training points and up to 13,500 testing points (we removed from the dataset all the points with null entries for the entire time span), we implemented Kernel ridge regression using FALKON <xref ref-type="bibr" rid="bib43">Rudi et al., 2018</xref>, a fast algorithm for matrix inversion (the number of iterations is set to 5 and the number of Nystrom centers is equal to the number of points in the training set) and we used it both for training and test.</p></sec><sec id="s4-5"><title>Modeling the expected performance of individual features</title><p>The importance of a feature is quantified by the corresponding optimal test error<disp-formula id="equ6"><label>(5)</label><mml:math id="m6"><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi mathvariant="normal">∞</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi>R</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mpadded width="+1.7pt"><mml:mi>y</mml:mi></mml:mpadded></mml:mrow><mml:mo>⁢</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>-</mml:mo><mml:mrow><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>⁢</mml:mo><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi>R</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>⁢</mml:mo><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:mrow></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf266"><mml:mi>R</mml:mi></mml:math></inline-formula> is the length of the cone. The target, or regression, function is given by <inline-formula><mml:math id="inf267"><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:math></inline-formula>:<disp-formula id="equ7"><label>(6)</label><mml:math id="m7"><mml:mrow><mml:mrow><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">⟨</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">|</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">⟩</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi>R</mml:mi></mml:msubsup><mml:mrow><mml:mi>y</mml:mi><mml:mo>⁢</mml:mo><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>and can be shown to minimize the expected error among all possible functions. In practice, the optimal expected error cannot be computed exactly, since neither the target nor the data distribution are known. In this paper, we choose to estimate the regression function from data using Kernel ridge regression with a Gaussian kernel. The choice of this latter approach is due to its nonparametric nature, which ensures that any target function <inline-formula><mml:math id="inf268"><mml:msup><mml:mi>f</mml:mi><mml:mo>*</mml:mo></mml:msup></mml:math></inline-formula> can be recovered provided large enough samples, and more generally that accurate estimates can be derived when only finite data are given <xref ref-type="bibr" rid="bib21">Hastie et al., 2001</xref>; <xref ref-type="bibr" rid="bib50">Steinwart, 2002</xref>. Provided with a kernel ridge regression estimator, an approximation to the optimal test error can then be computed on a hold-out set of data.</p><p>Next, we are interested into developing a clearer connection between the above statistical approach and quantities with a direct physical meaning. Towards this end, note that the joint, marginal and posterior distributions, given the prior on <inline-formula><mml:math id="inf269"><mml:mi>y</mml:mi></mml:math></inline-formula> and the likelihood <inline-formula><mml:math id="inf270"><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> are given by,<disp-formula id="equ8"> <label>(7)</label><mml:math id="m8"><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula><disp-formula id="equ9"> <label>(8)</label><mml:math id="m9"><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mo largeop="true" symmetric="true">∫</mml:mo><mml:mn>0</mml:mn><mml:mi mathvariant="normal">∞</mml:mi></mml:msubsup></mml:mstyle><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>⁢</mml:mo><mml:mrow><mml:mo rspace="0pt">d</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula><disp-formula id="equ10"><label>(9)</label><mml:math id="m10"><mml:mrow><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Assuming that points are sampled uniformly within the cone of detection, the prior is <inline-formula><mml:math id="inf271"><mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>⁢</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>/</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mrow></mml:math></inline-formula> (thus <inline-formula><mml:math id="inf272"><mml:mrow><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>⁢</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mo>/</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mrow></mml:math></inline-formula> and the denominator in <xref ref-type="disp-formula" rid="equ6">Equation 5</xref> is <inline-formula><mml:math id="inf273"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>/</mml:mo><mml:mn>18</mml:mn></mml:mrow></mml:math></inline-formula>). Importantly, the likelihood <inline-formula><mml:math id="inf274"><mml:mrow><mml:mi>p</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo lspace="2.5pt" rspace="2.5pt" stretchy="false">|</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is dictated by the fluid dynamics of odor plumes from a concentrated source. Our features are sample averages of intensity or timing attributes of the odor signal, and in the limit of large samples, they are normally distributed, that is<disp-formula id="equ11"><label>(10)</label><mml:math id="m11"><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>−</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Assuming the asymptotic limit to be well approximated, we can find empirical estimates for <inline-formula><mml:math id="inf275"><mml:mrow><mml:mi>g</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf276"><mml:mrow><mml:mi>s</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> given the data. This provides the desired characterization of the expected error and hence of the importance of each given feature in term of predictive power. Indeed, by numerically solving <xref ref-type="disp-formula" rid="equ6 equ7 equ8 equ9 equ10">Equations 5–9</xref> with the assumption (<xref ref-type="disp-formula" rid="equ11">Equation 10</xref>) and using the empirical estimates for <inline-formula><mml:math id="inf277"><mml:mrow><mml:mi>g</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="inf278"><mml:mrow><mml:mi>s</mml:mi><mml:mo>⁢</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, we reproduce the predictive power of individual features showed in <xref ref-type="fig" rid="fig5">Figure 5c</xref> (see <xref ref-type="fig" rid="fig5s3">Figure 5—figure supplement 3</xref>). To move beyond empirical estimates of the likelihood and generalize predictions to other kinds of flows one could generalize asymptotic models of turbulent plumes developed in <xref ref-type="bibr" rid="bib9">Celani et al., 2014</xref> to account for z-variations of the sampling locations. Similarly, extension are needed to consider combination of possibly non Gaussian features.</p></sec><sec id="s4-6"><title>Data availability</title><p>Simulations of odor transport are generated through NEK5000 <xref ref-type="bibr" rid="bib16">Fischer et al., 2008</xref>, freely available from Argonne National Laboratory (<ext-link ext-link-type="uri" xlink:href="https://nek5000.mcs.anl.gov/">https://nek5000.mcs.anl.gov/</ext-link>). Outputs from DNS presented in <xref ref-type="fig" rid="fig1">Figure 1</xref> are processed to extract time series and compute the five features described in the text (average concentration, slope, blank duration, whiff duration and intermittency factor). These data are available at the online repository <ext-link ext-link-type="uri" xlink:href="https://osf.io/ja9xr/">https://osf.io/ja9xr/</ext-link>. The results of kernel ridge regression perfomed on these data are presented in <xref ref-type="fig" rid="fig2">Figures 2</xref>—<xref ref-type="fig" rid="fig6">6</xref>. We perform kernel ridge regression on these data with the freely available code FALKON <xref ref-type="bibr" rid="bib43">Rudi et al., 2018</xref> (<ext-link ext-link-type="uri" xlink:href="https://github.com/LCSL/FALKON_paper">https://github.com/LCSL/FALKON_paper</ext-link>, <xref ref-type="bibr" rid="bib42">Rigolli, 2022</xref> copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:b152d94d711b88f99c206601afb1235de15321eb;origin=https://github.com/LCSL/FALKON_paper;visit=swh:1:snp:5032915c66d96288fedec074afe8d025600fca3b;anchor=swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17">swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17</ext-link>).</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf2"><p>No competing interests declared</p></fn><fn fn-type="COI-statement" id="conf3"><p>Reviewing editor, <italic>eLife</italic></p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Software, Formal analysis, Validation, Investigation, Visualization, Methodology, Writing – original draft, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Supervision, Investigation, Project administration, Writing – review and editing</p></fn><fn fn-type="con" id="con3"><p>Software, Supervision, Funding acquisition, Methodology, Project administration, Writing – review and editing</p></fn><fn fn-type="con" id="con4"><p>Conceptualization, Software, Formal analysis, Supervision, Funding acquisition, Investigation, Visualization, Methodology, Writing – original draft, Project administration, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-72196-mdarchecklist1-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>Simulations of odor transport are generated through NEK5000 (43), freely available from Argonne National Laboratory (<ext-link ext-link-type="uri" xlink:href="https://nek5000.mcs.anl.gov/">https://nek5000.mcs.anl.gov/</ext-link>). Outputs from DNS presented in Figure 1 are processed to extract time series and compute the five features described in the text (average concentration, slope, blank duration, whiff duration and intermittency factor). These data are available at the online repository <ext-link ext-link-type="uri" xlink:href="https://osf.io/ja9xr/">https://osf.io/ja9xr/</ext-link>. The results of kernel ridge regression performed on these data are presented in Figures 2-6. We perform kernel ridge regression on these data with the freely available code FALKON (<ext-link ext-link-type="uri" xlink:href="https://github.com/LCSL/FALKON_paper">https://github.com/LCSL/FALKON_paper</ext-link>, copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:b152d94d711b88f99c206601afb1235de15321eb;origin=https://github.com/LCSL/FALKON_paper;visit=swh:1:snp:5032915c66d96288fedec074afe8d025600fca3b;anchor=swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17">swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17</ext-link>).</p></sec><ack id="ack"><title>Acknowledgements</title><p>This work was supported by: the European Research Council (ERC) under the European Union’s Horizon 2020 research and innovation programme (grant agreement No 101002724 RIDING); the Air Force Office of Scientific Research under award number FA8655-20-1-7028; the National Institutes of Health (NIH) under award number R01DC018789; the French government, through the UCAJEDI Investments in the Future project managed by the National Research Agency (ANR) under reference number #ANR-15-IDEX-01. The authors are grateful to the OPAL infrastructure from Université Côte d’Azur and the Université Côte d’Azur’s Center for High-Performance Computing for providing resources and support. N.M. and N.R. are thankful for the support of Instituto Nazionale di Fisica Nucleare (INFN) Scientific Initiative SFT: Statistical Field Theory, Low-Dimensional Systems, Integrable Models and Applications.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ache</surname><given-names>BW</given-names></name><name><surname>Hein</surname><given-names>AM</given-names></name><name><surname>Bobkov</surname><given-names>YV</given-names></name><name><surname>Principe</surname><given-names>JC</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Smelling time: a neural basis for olfactory scene analysis</article-title><source>Trends in Neurosciences</source><volume>39</volume><fpage>649</fpage><lpage>655</lpage><pub-id pub-id-type="doi">10.1016/j.tins.2016.08.002</pub-id><pub-id pub-id-type="pmid">27594700</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ackels</surname><given-names>T</given-names></name><name><surname>Erskine</surname><given-names>A</given-names></name><name><surname>Dasgupta</surname><given-names>D</given-names></name><name><surname>Marin</surname><given-names>AC</given-names></name><name><surname>Warner</surname><given-names>TPA</given-names></name><name><surname>Tootoonian</surname><given-names>S</given-names></name><name><surname>Fukunaga</surname><given-names>I</given-names></name><name><surname>Harris</surname><given-names>JJ</given-names></name><name><surname>Schaefer</surname><given-names>AT</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Fast odour dynamics are encoded in the olfactory system and guide behaviour</article-title><source>Nature</source><volume>593</volume><fpage>558</fpage><lpage>563</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03514-2</pub-id><pub-id pub-id-type="pmid">33953395</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Atema</surname><given-names>J</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>Eddy chemotaxis and odor landscapes: exploration of nature with animal sensors</article-title><source>The Biological Bulletin</source><volume>191</volume><fpage>129</fpage><lpage>138</lpage><pub-id pub-id-type="doi">10.2307/1543074</pub-id><pub-id pub-id-type="pmid">29220222</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Baker</surname><given-names>KL</given-names></name><name><surname>Dickinson</surname><given-names>M</given-names></name><name><surname>Findley</surname><given-names>TM</given-names></name><name><surname>Gire</surname><given-names>DH</given-names></name><name><surname>Louis</surname><given-names>M</given-names></name><name><surname>Suver</surname><given-names>MP</given-names></name><name><surname>Verhagen</surname><given-names>JV</given-names></name><name><surname>Nagel</surname><given-names>KI</given-names></name><name><surname>Smear</surname><given-names>MC</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Algorithms for olfactory search across species</article-title><source>The Journal of Neuroscience</source><volume>38</volume><fpage>9383</fpage><lpage>9389</lpage><pub-id pub-id-type="doi">10.1523/JNEUROSCI.1668-18.2018</pub-id><pub-id pub-id-type="pmid">30381430</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Basil</surname><given-names>J</given-names></name><name><surname>Atema</surname><given-names>J</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Lobster orientation in turbulent odor plumes: simultaneous measurement of tracking behavior and temporal odor patterns</article-title><source>The Biological Bulletin</source><volume>187</volume><fpage>272</fpage><lpage>273</lpage><pub-id pub-id-type="doi">10.1086/BBLv187n2p272</pub-id><pub-id pub-id-type="pmid">7811821</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boie</surname><given-names>SD</given-names></name><name><surname>Connor</surname><given-names>EG</given-names></name><name><surname>McHugh</surname><given-names>M</given-names></name><name><surname>Nagel</surname><given-names>KI</given-names></name><name><surname>Ermentrout</surname><given-names>GB</given-names></name><name><surname>Crimaldi</surname><given-names>JP</given-names></name><name><surname>Victor</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Information-theoretic analysis of realistic odor plumes: What cues are useful for determining location?</article-title><source>PLOS Computational Biology</source><volume>14</volume><elocation-id>e1006275</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1006275</pub-id><pub-id pub-id-type="pmid">29990365</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname><given-names>SL</given-names></name><name><surname>Joseph</surname><given-names>J</given-names></name><name><surname>Stopfer</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Encoding a temporally structured stimulus with a temporally structured neural representation</article-title><source>Nature Neuroscience</source><volume>8</volume><fpage>1568</fpage><lpage>1576</lpage><pub-id pub-id-type="doi">10.1038/nn1559</pub-id><pub-id pub-id-type="pmid">16222230</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Catania</surname><given-names>KC</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Stereo and serial sniffing guide navigation to an odour source in a mammal</article-title><source>Nature Communications</source><volume>4</volume><elocation-id>1441</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms2444</pub-id><pub-id pub-id-type="pmid">23385586</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Celani</surname><given-names>A</given-names></name><name><surname>Villermaux</surname><given-names>E</given-names></name><name><surname>Vergassola</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Odor landscapes in turbulent environments</article-title><source>Physical Review X</source><volume>4</volume><elocation-id>041015</elocation-id><pub-id pub-id-type="doi">10.1103/PhysRevX.4.041015</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Demir</surname><given-names>M</given-names></name><name><surname>Kadakia</surname><given-names>N</given-names></name><name><surname>Anderson</surname><given-names>HD</given-names></name><name><surname>Clark</surname><given-names>DA</given-names></name><name><surname>Emonet</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Walking <italic>Drosophila</italic> navigate complex plumes using stochastic decisions biased by the timing of odor encounters</article-title><source>eLife</source><volume>9</volume><elocation-id>e57524</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.57524</pub-id><pub-id pub-id-type="pmid">33140723</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Duplat</surname><given-names>J</given-names></name><name><surname>Jouary</surname><given-names>A</given-names></name><name><surname>Villermaux</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Entanglement rules for random mixtures</article-title><source>Physical Review Letters</source><volume>105</volume><elocation-id>034504</elocation-id><pub-id pub-id-type="doi">10.1103/PhysRevLett.105.034504</pub-id><pub-id pub-id-type="pmid">20867769</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fackrell</surname><given-names>JE</given-names></name><name><surname>Robins</surname><given-names>AG</given-names></name></person-group><year iso-8601-date="1982">1982</year><article-title>Concentration fluctuations and fluxes in plumes from point sources in a turbulent boundary layer</article-title><source>Journal of Fluid Mechanics</source><volume>117</volume><fpage>1</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.1017/S0022112082001499</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Falkovich</surname><given-names>G</given-names></name><name><surname>Gawȩdzki</surname><given-names>K</given-names></name><name><surname>Vergassola</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Particles and fields in fluid turbulence</article-title><source>Reviews of Modern Physics</source><volume>73</volume><fpage>913</fpage><lpage>975</lpage><pub-id pub-id-type="doi">10.1103/RevModPhys.73.913</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Findley</surname><given-names>TM</given-names></name><name><surname>Wyrick</surname><given-names>DG</given-names></name><name><surname>Cramer</surname><given-names>JL</given-names></name><name><surname>Brown</surname><given-names>MA</given-names></name><name><surname>Holcomb</surname><given-names>B</given-names></name><name><surname>Attey</surname><given-names>R</given-names></name><name><surname>Yeh</surname><given-names>D</given-names></name><name><surname>Monasevitch</surname><given-names>E</given-names></name><name><surname>Nouboussi</surname><given-names>N</given-names></name><name><surname>Cullen</surname><given-names>I</given-names></name><name><surname>Songco</surname><given-names>JO</given-names></name><name><surname>King</surname><given-names>JF</given-names></name><name><surname>Ahmadian</surname><given-names>Y</given-names></name><name><surname>Smear</surname><given-names>MC</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Sniff-synchronized, gradient-guided olfactory search by freely moving mice</article-title><source>eLife</source><volume>10</volume><elocation-id>e58523</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.58523</pub-id><pub-id pub-id-type="pmid">33942713</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fischer</surname><given-names>PF</given-names></name><name><surname>Loth</surname><given-names>F</given-names></name><name><surname>Lee</surname><given-names>SE</given-names></name><name><surname>Lee</surname><given-names>SW</given-names></name><name><surname>Smith</surname><given-names>DS</given-names></name><name><surname>Bassiouny</surname><given-names>HS</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Simulation of high-Reynolds number vascular flows</article-title><source>Computer Methods in Applied Mechanics and Engineering</source><volume>196</volume><fpage>3049</fpage><lpage>3060</lpage><pub-id pub-id-type="doi">10.1016/j.cma.2006.10.015</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Fischer</surname><given-names>PF</given-names></name><name><surname>Lottes</surname><given-names>JW</given-names></name><name><surname>Kerkemeier</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2008">2008</year><data-title>Nek5000</data-title><version designator="v19.0">v19.0</version><source>NEK</source><ext-link ext-link-type="uri" xlink:href="http://nek5000.mcs.anl.gov">http://nek5000.mcs.anl.gov</ext-link></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gardiner</surname><given-names>JM</given-names></name><name><surname>Atema</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>The function of bilateral odor arrival time differences in olfactory orientation of sharks</article-title><source>Current Biology</source><volume>20</volume><fpage>1187</fpage><lpage>1191</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2010.04.053</pub-id><pub-id pub-id-type="pmid">20541411</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gire</surname><given-names>DH</given-names></name><name><surname>Kapoor</surname><given-names>V</given-names></name><name><surname>Arrighi-Allisan</surname><given-names>A</given-names></name><name><surname>Seminara</surname><given-names>A</given-names></name><name><surname>Murthy</surname><given-names>VN</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Mice develop efficient strategies for foraging and navigation using complex natural stimuli</article-title><source>Current Biology</source><volume>26</volume><fpage>1261</fpage><lpage>1273</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2016.03.040</pub-id><pub-id pub-id-type="pmid">27112299</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gorur-Shandilya</surname><given-names>S</given-names></name><name><surname>Demir</surname><given-names>M</given-names></name><name><surname>Long</surname><given-names>J</given-names></name><name><surname>Clark</surname><given-names>DA</given-names></name><name><surname>Emonet</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Olfactory receptor neurons use gain control and complementary kinetics to encode intermittent odorant stimuli</article-title><source>eLife</source><volume>6</volume><elocation-id>e27670</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.27670</pub-id><pub-id pub-id-type="pmid">28653907</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gorur-Shandilya</surname><given-names>S</given-names></name><name><surname>Martelli</surname><given-names>C</given-names></name><name><surname>Demir</surname><given-names>M</given-names></name><name><surname>Emonet</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Controlling and measuring dynamic odorant stimuli in the laboratory</article-title><source>The Journal of Experimental Biology</source><volume>222</volume><elocation-id>207787</elocation-id><pub-id pub-id-type="doi">10.1242/jeb.207787</pub-id><pub-id pub-id-type="pmid">31672728</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Hastie</surname><given-names>T</given-names></name><name><surname>Friedman</surname><given-names>J</given-names></name><name><surname>Tibshirani</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2001">2001</year><source>The Elements of Statistical Learning</source><publisher-loc>New York, NY</publisher-loc><publisher-name>Springer</publisher-name><pub-id pub-id-type="doi">10.1007/978-0-387-21606-5</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="thesis"><person-group person-group-type="author"><name><surname>Ho</surname><given-names>L</given-names></name></person-group><year iso-8601-date="1989">1989</year><article-title>A Legendre spectral element method for simulation of incompressible unsteady viscous free-surface flows</article-title><publisher-loc>Cambridge, USA</publisher-loc><publisher-name>Massachusetts Institute of Technology</publisher-name></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jacob</surname><given-names>V</given-names></name><name><surname>Monsempès</surname><given-names>C</given-names></name><name><surname>Rospars</surname><given-names>J-P</given-names></name><name><surname>Masson</surname><given-names>J-B</given-names></name><name><surname>Lucas</surname><given-names>P</given-names></name><name><surname>Morozov</surname><given-names>AV</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Olfactory coding in the turbulent realm</article-title><source>PLOS Computational Biology</source><volume>13</volume><elocation-id>e1005870</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1005870</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Justus</surname><given-names>KA</given-names></name><name><surname>Murlis</surname><given-names>J</given-names></name><name><surname>Jones</surname><given-names>C</given-names></name><name><surname>Cardé</surname><given-names>RT</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Measurement of odor-plume structure in a wind tunnel using a photoionization detector and a tracer gas</article-title><source>Environmental Fluid Mechanics</source><volume>2</volume><fpage>115</fpage><lpage>142</lpage><pub-id pub-id-type="doi">10.1023/A:1016227601019</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kennedy</surname><given-names>JS</given-names></name><name><surname>Marsh</surname><given-names>D</given-names></name></person-group><year iso-8601-date="1974">1974</year><article-title>Pheromone-regulated anemotaxis in flying moths</article-title><source>Science</source><volume>184</volume><fpage>999</fpage><lpage>1001</lpage><pub-id pub-id-type="doi">10.1126/science.184.4140.999</pub-id><pub-id pub-id-type="pmid">4826172</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leathers</surname><given-names>KW</given-names></name><name><surname>Michaelis</surname><given-names>BT</given-names></name><name><surname>Reidenbach</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Interpreting the spatial-temporal structure of turbulent chemical plumes utilized in odor tracking by lobsters</article-title><source>Fluids</source><volume>5</volume><elocation-id>82</elocation-id><pub-id pub-id-type="doi">10.3390/fluids5020082</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lewis</surname><given-names>SM</given-names></name><name><surname>Xu</surname><given-names>L</given-names></name><name><surname>Rigolli</surname><given-names>N</given-names></name><name><surname>Tariq</surname><given-names>MF</given-names></name><name><surname>Suarez</surname><given-names>LM</given-names></name><name><surname>Stern</surname><given-names>M</given-names></name><name><surname>Seminara</surname><given-names>A</given-names></name><name><surname>Gire</surname><given-names>DH</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Plume dynamics structure the spatiotemporal activity of mitral/tufted cell networks in the mouse olfactory bulb</article-title><source>Frontiers in Cellular Neuroscience</source><volume>15</volume><elocation-id>633757</elocation-id><pub-id pub-id-type="doi">10.3389/fncel.2021.633757</pub-id><pub-id pub-id-type="pmid">34012385</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mafra-Neto</surname><given-names>A</given-names></name><name><surname>Cardé</surname><given-names>RT</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Fine-scale structure of pheromone plumes modulates upwind orientation of flying moths</article-title><source>Nature</source><volume>369</volume><fpage>142</fpage><lpage>144</lpage><pub-id pub-id-type="doi">10.1038/369142a0</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martelli</surname><given-names>C</given-names></name><name><surname>Carlson</surname><given-names>JR</given-names></name><name><surname>Emonet</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Intensity invariant dynamics and odor-specific latencies in olfactory receptor neuron response</article-title><source>The Journal of Neuroscience</source><volume>33</volume><fpage>6285</fpage><lpage>6297</lpage><pub-id pub-id-type="doi">10.1523/JNEUROSCI.0426-12.2013</pub-id><pub-id pub-id-type="pmid">23575828</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Michaelis</surname><given-names>BT</given-names></name><name><surname>Leathers</surname><given-names>KW</given-names></name><name><surname>Bobkov</surname><given-names>YV</given-names></name><name><surname>Ache</surname><given-names>BW</given-names></name><name><surname>Principe</surname><given-names>JC</given-names></name><name><surname>Baharloo</surname><given-names>R</given-names></name><name><surname>Park</surname><given-names>IM</given-names></name><name><surname>Reidenbach</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Odor tracking in aquatic organisms: the importance of temporal and spatial intermittency of the turbulent plume</article-title><source>Scientific Reports</source><volume>10</volume><elocation-id>7961</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-020-64766-y</pub-id><pub-id pub-id-type="pmid">32409665</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moore</surname><given-names>P</given-names></name><name><surname>Crimaldi</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Odor landscapes and animal behavior: tracking odor plumes in different physical worlds</article-title><source>Journal of Marine Systems</source><volume>49</volume><fpage>55</fpage><lpage>64</lpage><pub-id pub-id-type="doi">10.1016/j.jmarsys.2003.05.005</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Murlis</surname><given-names>J</given-names></name><name><surname>Elkinton</surname><given-names>JS</given-names></name><name><surname>Cardé</surname><given-names>RT</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>Odor plumes and how insects use them</article-title><source>Annual Review of Entomology</source><volume>37</volume><fpage>505</fpage><lpage>532</lpage><pub-id pub-id-type="doi">10.1146/annurev.en.37.010192.002445</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nagel</surname><given-names>KI</given-names></name><name><surname>Wilson</surname><given-names>RI</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Biophysical mechanisms underlying olfactory receptor neuron dynamics</article-title><source>Nature Neuroscience</source><volume>14</volume><fpage>208</fpage><lpage>216</lpage><pub-id pub-id-type="doi">10.1038/nn.2725</pub-id><pub-id pub-id-type="pmid">21217763</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Orszag</surname><given-names>SA</given-names></name></person-group><year iso-8601-date="1980">1980</year><article-title>Spectral methods for problems in complex geometries</article-title><source>Journal of Computational Physics</source><volume>37</volume><fpage>70</fpage><lpage>92</lpage><pub-id pub-id-type="doi">10.1016/0021-9991(80)90005-4</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parabucki</surname><given-names>A</given-names></name><name><surname>Bizer</surname><given-names>A</given-names></name><name><surname>Morris</surname><given-names>G</given-names></name><name><surname>Munoz</surname><given-names>AE</given-names></name><name><surname>Bala</surname><given-names>ADS</given-names></name><name><surname>Smear</surname><given-names>M</given-names></name><name><surname>Shusterman</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Odor concentration change coding in the olfactory bulb</article-title><source>ENeuro</source><volume>6</volume><elocation-id>ENEURO.0396-18.2019</elocation-id><pub-id pub-id-type="doi">10.1523/ENEURO.0396-18.2019</pub-id><pub-id pub-id-type="pmid">30834303</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Park</surname><given-names>IM</given-names></name><name><surname>Bobkov</surname><given-names>YV</given-names></name><name><surname>Ache</surname><given-names>BW</given-names></name><name><surname>Príncipe</surname><given-names>JC</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Intermittency coding in the primary olfactory system: A neural substrate for olfactory scene analysis</article-title><source>The Journal of Neuroscience</source><volume>34</volume><fpage>941</fpage><lpage>952</lpage><pub-id pub-id-type="doi">10.1523/JNEUROSCI.2204-13.2014</pub-id><pub-id pub-id-type="pmid">24431452</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Park</surname><given-names>IJ</given-names></name><name><surname>Hein</surname><given-names>AM</given-names></name><name><surname>Bobkov</surname><given-names>YV</given-names></name><name><surname>Reidenbach</surname><given-names>MA</given-names></name><name><surname>Ache</surname><given-names>BW</given-names></name><name><surname>Principe</surname><given-names>JC</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Neurally encoding time for olfactory navigation</article-title><source>PLOS Computational Biology</source><volume>12</volume><elocation-id>e1004682</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1004682</pub-id><pub-id pub-id-type="pmid">26730727</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patera</surname><given-names>AT</given-names></name></person-group><year iso-8601-date="1984">1984</year><article-title>A spectral element method for fluid dynamics: Laminar flow in A channel expansion</article-title><source>Journal of Computational Physics</source><volume>54</volume><fpage>468</fpage><lpage>488</lpage><pub-id pub-id-type="doi">10.1016/0021-9991(84)90128-1</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Pope</surname><given-names>SB</given-names></name></person-group><year iso-8601-date="1984">1984</year><source>Turbulent Flows</source><publisher-loc>Cambridge</publisher-loc><publisher-name>Cambridge University Press</publisher-name><pub-id pub-id-type="doi">10.1017/CBO9780511840531</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Reddy</surname><given-names>G</given-names></name><name><surname>Shraiman</surname><given-names>BI</given-names></name><name><surname>Vergassola</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Sector search strategies for odor trail tracking</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.03.03.433838</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Riffell</surname><given-names>JA</given-names></name><name><surname>Shlizerman</surname><given-names>E</given-names></name><name><surname>Sanders</surname><given-names>E</given-names></name><name><surname>Abrell</surname><given-names>L</given-names></name><name><surname>Medina</surname><given-names>B</given-names></name><name><surname>Hinterwirth</surname><given-names>AJ</given-names></name><name><surname>Kutz</surname><given-names>JN</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Sensory biology flower discrimination by pollinators in a dynamic chemical environment</article-title><source>Science</source><volume>344</volume><fpage>1515</fpage><lpage>1518</lpage><pub-id pub-id-type="doi">10.1126/science.1251041</pub-id><pub-id pub-id-type="pmid">24970087</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Rigolli</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>FALKON_paper</data-title><version designator="swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17">swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:b152d94d711b88f99c206601afb1235de15321eb;origin=https://github.com/LCSL/FALKON_paper;visit=swh:1:snp:5032915c66d96288fedec074afe8d025600fca3b;anchor=swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17">https://archive.softwareheritage.org/swh:1:dir:b152d94d711b88f99c206601afb1235de15321eb;origin=https://github.com/LCSL/FALKON_paper;visit=swh:1:snp:5032915c66d96288fedec074afe8d025600fca3b;anchor=swh:1:rev:480741cf1e7da0d1d7415309cd6f254080a6ca17</ext-link></element-citation></ref><ref id="bib43"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Rudi</surname><given-names>A</given-names></name><name><surname>Carratino</surname><given-names>L</given-names></name><name><surname>Rosasco</surname><given-names>L</given-names></name></person-group><article-title>Neural information processing systems</article-title><conf-name>Proceedings of the 31st International Conference on Neural Information Processing Systems</conf-name><year iso-8601-date="2018">2018</year><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/3294996.3295145">https://dl.acm.org/doi/10.5555/3294996.3295145</ext-link></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Saddoughi</surname><given-names>SG</given-names></name><name><surname>Veeravalli</surname><given-names>SV</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Local isotropy in turbulent boundary layers at high Reynolds number</article-title><source>Journal of Fluid Mechanics</source><volume>268</volume><fpage>333</fpage><lpage>372</lpage><pub-id pub-id-type="doi">10.1017/S0022112094001370</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schmuker</surname><given-names>M</given-names></name><name><surname>Bahr</surname><given-names>V</given-names></name><name><surname>Huerta</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Exploiting plume structure to decode gas source distance using metal-oxide gas sensors</article-title><source>Sensors and Actuators B</source><volume>235</volume><fpage>636</fpage><lpage>646</lpage><pub-id pub-id-type="doi">10.1016/j.snb.2016.05.098</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Schölkopf</surname><given-names>B</given-names></name><name><surname>Smola</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2002">2002</year><source>Learning with Kernels</source><publisher-loc>Cambridge, MA, USA</publisher-loc><publisher-name>MIT Press</publisher-name></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shraiman</surname><given-names>B</given-names></name><name><surname>Siggia</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Scalar turbulence</article-title><source>Nature</source><volume>405</volume><fpage>639</fpage><lpage>646</lpage><pub-id pub-id-type="doi">10.1038/35015000</pub-id><pub-id pub-id-type="pmid">10864314</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smear</surname><given-names>M</given-names></name><name><surname>Resulaj</surname><given-names>A</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Bozza</surname><given-names>T</given-names></name><name><surname>Rinberg</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Multiple perceptible signals from a single olfactory glomerulus</article-title><source>Nature Neuroscience</source><volume>16</volume><fpage>1687</fpage><lpage>1691</lpage><pub-id pub-id-type="doi">10.1038/nn.3519</pub-id><pub-id pub-id-type="pmid">24056698</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steck</surname><given-names>K</given-names></name><name><surname>Veit</surname><given-names>D</given-names></name><name><surname>Grandy</surname><given-names>R</given-names></name><name><surname>Badia</surname><given-names>SBI</given-names></name><name><surname>Badia</surname><given-names>SBI</given-names></name><name><surname>Mathews</surname><given-names>Z</given-names></name><name><surname>Verschure</surname><given-names>P</given-names></name><name><surname>Hansson</surname><given-names>BS</given-names></name><name><surname>Knaden</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A high-throughput behavioral paradigm for <italic>Drosophila</italic> olfaction - The Flywalk</article-title><source>Scientific Reports</source><volume>2</volume><elocation-id>361</elocation-id><pub-id pub-id-type="doi">10.1038/srep00361</pub-id><pub-id pub-id-type="pmid">22511996</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steinwart</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Support vector machines are universally consistent</article-title><source>Journal of Complexity</source><volume>18</volume><fpage>768</fpage><lpage>791</lpage><pub-id pub-id-type="doi">10.1006/jcom.2002.0642</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Breugel</surname><given-names>F</given-names></name><name><surname>Dickinson</surname><given-names>MH</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Plume-tracking behavior of flying <italic>Drosophila</italic> emerges from a set of distinct sensory-motor reflexes</article-title><source>Current Biology</source><volume>24</volume><fpage>274</fpage><lpage>286</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2013.12.023</pub-id><pub-id pub-id-type="pmid">24440395</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vergassola</surname><given-names>M</given-names></name><name><surname>Villermaux</surname><given-names>E</given-names></name><name><surname>Shraiman</surname><given-names>BI</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>“Infotaxis” as a strategy for searching without gradients</article-title><source>Nature</source><volume>445</volume><fpage>406</fpage><lpage>409</lpage><pub-id pub-id-type="doi">10.1038/nature05464</pub-id><pub-id pub-id-type="pmid">17251974</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vickers</surname><given-names>NJ</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Mechanisms of animal navigation in odor plumes</article-title><source>The Biological Bulletin</source><volume>198</volume><fpage>203</fpage><lpage>212</lpage><pub-id pub-id-type="doi">10.2307/1542524</pub-id><pub-id pub-id-type="pmid">10786941</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vickers</surname><given-names>NJ</given-names></name><name><surname>Christensen</surname><given-names>TA</given-names></name><name><surname>Baker</surname><given-names>TC</given-names></name><name><surname>Hildebrand</surname><given-names>JG</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Odour-plume dynamics influence the brain’s olfactory code</article-title><source>Nature</source><volume>410</volume><fpage>466</fpage><lpage>470</lpage><pub-id pub-id-type="doi">10.1038/35068559</pub-id><pub-id pub-id-type="pmid">11260713</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Victor</surname><given-names>JD</given-names></name><name><surname>Boie</surname><given-names>SD</given-names></name><name><surname>Connor</surname><given-names>EG</given-names></name><name><surname>Crimaldi</surname><given-names>JP</given-names></name><name><surname>Ermentrout</surname><given-names>GB</given-names></name><name><surname>Nagel</surname><given-names>KI</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Olfactory navigation and the receptor nonlinearity</article-title><source>The Journal of Neuroscience</source><volume>39</volume><fpage>3713</fpage><lpage>3727</lpage><pub-id pub-id-type="doi">10.1523/JNEUROSCI.2512-18.2019</pub-id><pub-id pub-id-type="pmid">30846614</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.72196.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Goldstein</surname><given-names>Raymond E</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/013meh722</institution-id><institution>University of Cambridge</institution></institution-wrap><country>United Kingdom</country></aff></contrib></contrib-group></front-stub><body><p>This paper explores the question of the optimum strategy for odor detection in a turbulent environment. The authors use high-resolution simulations of turbulent flow to investigate the transport and detection of odors advected by the flow, comparing machine learning strategies based on the temporal dynamics of the signal with those based on intensity. The work should be of interest to researchers working on a broad range of problems in sensation and navigation across scales.</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.72196.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Goldstein</surname><given-names>Raymond E</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/013meh722</institution-id><institution>University of Cambridge</institution></institution-wrap><country>United Kingdom</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Learning to predict target location with turbulent odor plumes&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 2 peer reviewers, one of whom is a member of our Board of Reviewing Editors, and the evaluation has been overseen by Aleksandra Walczak as the Senior Editor. The reviewers have opted to remain anonymous.</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission.</p><p>Essential revisions:</p><p>1) It would be good if the authors could provide evidence that their conclusions do not depend on the precise location of the odor source relative to the cylinder. We do not require an exhaustive study, but some evidence would be very helpful.</p><p>2) We think that more could have been understood from these data, had the authors tried to focus on dimensionless quantities. It would be important to understand what sets the distance from the source where the intensity-based search strategy becomes less effective than the time-based one, and how the results depend on the Schmidt number.</p><p>3) The authors should carefully specify how dimensionless parameters are introduced. They do it in places but not systematically. For instance, how are wavenumbers in Figure 1(d) are made dimensionless?</p><p>4) On a similar note, how does the range of k's in Figure 1(d) compare to the height of the channel? Are these k's taken along the streamwise direction only? What is the vertical axis in that graph? Would it integral over all k's correspond to the odour density variance? These need to be specified.</p><p>5) What is the value of odour diffusivity \kapppa_\theta? What is the Schmidt number in these simulations? Visual inspection of Figure 1(d) suggests that Sc&gt;&gt;1, which might explain why the -5/3 slope is so far from representing the data. A reference to passive scalar advection by turbulence is in order here. We would suggest K. Sreenivasan, Turbulent mixing: A perspective, PNAS (2019).</p><p>6) Perhaps T in line 335 should be replaced by \theta.</p><p>7) Please provide figure numbers in lines 377, 412, and 414.<italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>In this work, the authors combine high resolution numerical simulations of turbulent airflow that advects a passive scalar (an &quot;odor&quot;) and machine learning algorithms to investigate the question of how a navigation strategy based on the temporal dynamics of odor detection compares with one based on the intensity of the plume. In the simulations, the odor is introduced downstream of a cylinder that induces turbulence in the flow and the machine learning algorithm is trained and evaluated at various points further downstream. The authors conclude that intensity/gradient based measurements work closer to the source, while temporal schemes are good throughout the range. The study is done to a very high standard and presented clearly, although there are general questions about the data analysis (see below) that should be improved. That said, the work should have significant impact on a broad range of fields, from sensing to navigation, across a range of organism length scales.</p><p>One concern is the nature of the turbulent profile and its advection of the passive scalar. It would help if the characteristic dimensionless numbers of the problem were specified (Peclet, Schmidt numbers), and it was made clear how the choice of odor release point affects the conclusions. The turbulence itself spreads and diffuses with distance from the source, and one would presume the location of the odor release can matter substantially.</p><p>Also, as mentioned briefly in the Discussion, this work examines algorithms based on evaluating accuracy of prediction at a given measurement point. Unless I have missed something in the presentation, the issue of navigation is left unexamined. As in bacterial chemotaxis or any related problem, no navigation strategy is perfect, especially in the face of such a fluctuating source of information, so the full problem involves making estimates of where the source is and then moving to a new location, estimating again, moving, etc. The authors should clarify under what circumstances an evaluation at a fixed point is sufficiently predictive of &quot;learning&quot;.<italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>In their manuscript, Rigolli et al., studied how measurements of intensity of a passive scalar (odour), its spatial gradients, and its time variations can be used to efficiently find the spatial location of its source. To this end, the authors performed direct numerical simulations of high-Reynolds number pressure-driven channel flow, partially blocked by a cylinder to generate turbulent velocity fluctuations. At a fixed position downstream the cylinder, they introduce a source of odour that diffuses through the fluid and is advected by its turbulent motion. The authors then trained a supervised machine learning algorithm on a collection of spatial and temporal odour profiles thus obtained. The algorithm was subsequently tested on various odour signals originating from the same pool of simulations. By varying the spatial position of the measurement point and the temporal exposure, the authors concluded that detection strategies based on the odour concentration and its spatial gradient work best at small separations from the source (and also close to the zero-odour concentration boundaries), while time-based strategies work reasonably well everywhere within the domain selected for detection.</p><p>Within the geometry set by the authors, the conclusions of the paper are supported by data. However, this setup leads to a natural question whether the search strategies determined here pertain to the distance to the odour source or to the distance to the turbulence source (the half-cylinder). Turbulence generated by an obstacle has a particular spatial profile; its temporal profile also depends on the distance to the obstacle. It is therefore possible that the machine learning algorithm has indirectly picked up these features rather than the odour profile itself. This can be settled by feeding the existing algorithm signals from simulations with various distances between the source and the obstacle. In the absence of this check, it is impossible to make general statements about applicability of these search strategies in other situations.</p><p>Another weak point of this study is the use of a cone of detection as it dramatically reduces the search complexity to almost a quasi one-dimensional problem. The authors appreciate it and point out that their choice is forced by a very limited by a very small number of detections outside the cone. This shortcoming should be addressed in future work.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.72196.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>1) It would be good if the authors could provide evidence that their conclusions do not depend on the precise location of the odor source relative to the cylinder. We do not require an exhaustive study, but some evidence would be very helpful.</p></disp-quote><p>We expect results to be robust to changes in source location, as long as this is within the bulk of the channel and not within the viscous layer, which generates substantially less intermittent odor plumes (see e.g. Fackrell and Robins, J Fluid Mech 117:1, 1982). We performed additional computational fluid dynamics simulations, moving the source to two further locations downstream of the obstacle, at the same height and performed inference over an identical conical domain shifted to match the source location (see Figure 2, Figure Supplement 5 left). Performance at source height varies little over the three locations, demonstrating that the algorithm learns from the dynamics of the scalar and not of the underlying velocity field (we chose one individual feature per class and their combination, see Figure 2, Figure Supplement 5 center). For all three sources, performance of the average degrades with height, whereas performance of intermittency improves with height (Figure 2, Figure Supplement 5 right), consistent with the results of the Figure 5c and despite using a different cone. Finally, we trained the algorithm with a dataset generated by one source and tested it on dataset obtained from the other sources: we found that pairs of features are more sensitive to the details of the dataset, whereas individual features maintain their full predictive power (Figure 2 Figure Suppl 6).</p><p>If and how animals adapt their models to deal with sources at different heights is a fascinating question that is relevant for animal behavior; we believe this point deserves an in depth investigation on its own and we are currently addressing these questions in a different study.</p><fig id="sa2fig1" position="float"><label>Author response image 1.</label><caption><title>Estimated test error of individual features at different heights using the theoretical framework Equation (1)-(5) outlined in the text with the empirical likelihood estimated from data (not shown).</title><p>Symbols as in Figure 5c of the manuscript (black / grey represent timing /intensity features; dashed lines: whiffs).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-72196-sa2-fig1-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>2) We think that more could have been understood from these data, had the authors tried to focus on dimensionless quantities. It would be important to understand what sets the distance from the source where the intensity-based search strategy becomes less effective than the time-based one, and how the results depend on the Schmidt number.</p></disp-quote><p>We thank the reviewers for this important observation that led us to develop a theoretical frame work that we believe deserves further attention. We highlight below the conceptual building blocks, which we have detailed in the manuscript. The predictive power χ of an individual feature x is its expected error: <inline-formula><mml:math id="sa2m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mo>∫</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>−</mml:mo><mml:msup><mml:mi>f</mml:mi><mml:mrow><mml:mo>∗</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mtext>dy</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mo>∫</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>−</mml:mo><mml:mover><mml:mi>y</mml:mi><mml:mo accent="false">¯</mml:mo></mml:mover></mml:mrow><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mi>d</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula> where <italic>p</italic>(<italic>x,y</italic>) is the joint probability distribution of the input output pair (<italic>x,y</italic>)and <italic>p</italic>(<italic>y</italic>) is the prior on the output <italic>y.f*(x)</italic> is the optimal predictor and can be shown to be <inline-formula><mml:math id="sa2m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msup><mml:mi>f</mml:mi><mml:mrow><mml:mo>∗</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mo>∫</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mtext>yp</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>y</mml:mi><mml:mo>,</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> where R is the length of the cone (see Hastie et al., The elements of statistical learning: Datamining, inference, and prediction, 2001 and Steinwart, J of Complexity 18:768, 2002). The expected error is generally unavailable because p&lt;milestone-start /&gt;(&lt;milestone-end /&gt;yx&lt;milestone-start /&gt;)&lt;milestone-end /&gt; and p(x,y) are unknown. However, we can connect equation (1) to the physics of the odor plume by computing the joint, marginal and posterior distributions using the likelihood p(x│y):<inline-formula><mml:math id="sa2m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mrow/><mml:mo>|</mml:mo><mml:mrow/><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> <inline-formula><mml:math id="sa2m4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mo>∫</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi mathvariant="normal">∞</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mtext>dy</mml:mtext></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> <inline-formula><mml:math id="sa2m5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mrow/><mml:mo>|</mml:mo><mml:mrow/><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula> where the prior is <italic>p(y)</italic>=2y/<italic>R<sup>2</sup></italic> (thus <inline-formula><mml:math id="sa2m6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo accent="false">¯</mml:mo></mml:mover></mml:mrow></mml:mstyle></mml:math></inline-formula>= 2<italic>R</italic>/3 and the denominator in equation (1) is R<sup>2</sup>/18). Importantly, the likelihood is dictated by the fluid dynamics of odor plumes from a concentrated source. Because our individual features are sample averages, their statistics is approximately Gaussian: <inline-formula><mml:math id="sa2m7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mrow/><mml:mo>|</mml:mo><mml:mrow/><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mi>N</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>−</mml:mo><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>s</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> and we can then simply characterize the likelihood by data fitting to estimate their average and standard deviation. By solving equations (1) to (4) with the assumption (5) and the empirical estimates for g(y) and s(y) we recover the predictive power of individual features showed in Figure 5c of the manuscript (see Figure 1). (The argument needs to be better adapted to whiffs which deviate considerably from a normal distribution)</p><p>We believe this theoretical framework deserves further attention. Indeed, one can move</p><p>beyond empirical estimates of the likelihood and generalize predictions to pairs of features and other kinds of flows by leveraging the asymptotic arguments recently proposed by Celani et al., Phys Rev 4:041015, 2014. We do not further elaborate on this point here, but we are currently working to fully develop these asymptotic arguments not only to obtain scaling behaviors for g(y) and s(y) but also the prefactor, which is crucial to understand what non-dimensional quantities dictate the switch from timing to intensity. Note that in this regime the Schmidt number plays a minor role, as we further elaborate on below.</p><disp-quote content-type="editor-comment"><p>3) The authors should carefully specify how dimensionless parameters are introduced. They do it in places but not systematically. For instance, how are wavenumbers in Figure 1(d) are made dimensionless?</p></disp-quote><p>Thank you for the comment, we have provided information on the Schmidt number used in the simulations (Sc = 1) and the definition of wavenumbers in Figure 1d (see next answer for more details on this point), as well as dimensional parameters that were either only mentioned in the text (viscosity) or not mentioned (diffusivity). All information is now summarized in Table 1.</p><disp-quote content-type="editor-comment"><p>4) On a similar note, how does the range of k’s in Figure 1(d) compare to the height of the channel? Are these k’s taken along the streamwise direction only? What is the vertical axis in that graph? Would it integral over all k’s correspond to the odour density variance? These need to be specified.</p></disp-quote><p>Thank you for the comment, we have included the missing information. Figure 1d shows the two dimensional spectra <inline-formula><mml:math id="sa2m8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>E</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mi>d</mml:mi><mml:mtext>dk</mml:mtext></mml:mfrac><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mo>∫</mml:mo><mml:mrow><mml:mrow><mml:mo>│</mml:mo></mml:mrow><mml:mi>k</mml:mi><mml:mi>x</mml:mi><mml:mi>y</mml:mi><mml:mrow><mml:mo>│</mml:mo></mml:mrow><mml:mo>&lt;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>│</mml:mo></mml:mrow><mml:mrow><mml:mover><mml:mi>c</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>k</mml:mi><mml:mrow><mml:mtext>xy</mml:mtext></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mo>│</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:msup><mml:mi>d</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> normalized with the scalar variance <italic>σ</italic>c2, where ĉ(κ<sub>xy</sub>) is the two dimensional Fourier transform of the scalar concentration at the height of the source. The integral of the spectra is indeed the scalar variance. We non-dimensionalize the wavenumber with the inverse Kolmogorov scale <italic>η</italic> −1. We corrected a small discrepancy coming from the fact that the Legendre polynomials used by Nek5000 have sampling points that are not perfectly equally spaced. The k−5∕3 scaling holds for k <italic>η</italic> 0.1, consistent with previous experimental results in channel flow (see e.g. S.G. Saddoughi and S.V. Veeravalli J. Fluid Mech.(1994) 268:333-372). To further characterize the flow we add a panel to figure 1, showing that the mean flow follows the well known law of the wall, recovering classical statistics for channel turbulence.</p><disp-quote content-type="editor-comment"><p>5) What is the value of odour diffusivity \kapppa_\theta? What is the Schmidt number in these simulations? Visual inspection of Figure 1(d) suggests that Sc&gt;&gt;1, which might explain why the -5/3 slope is so far from representing the data. A reference to passive scalar advection by turbulence is in order here. We would suggest K. Sreenivasan, Turbulent mixing: A perspective, PNAS (2019).</p></disp-quote><p>Here we work at Schmidt = 1, κ<italic>θ</italic> = ν = 1.5 × 10−5m2∕s. The inertial range is short but consistent with previous literature on channel flow (see previous answer). Varying the Schmidt number affects the fine details between the Batchelor and Kolmogorov scales (Falkovich et al., Rev Mod Phys 105 73:913, 2001). These are below the size of the source in our simulations, as is also the case in 106 many situations of practical relevance. In this regime, the dynamics below the Kolmogorov scale 107 affects little the large scale statistics of odor plumes from concentrated sources which are dictated 108 by the separation of Lagrangian particles (as argued in Celani et al., Phys Rev X, 2014 and confirmed experimentally in Duplat et al., Phys Fluids, 22:035104, 2010). We included this comment at page 12.</p><disp-quote content-type="editor-comment"><p>6) Perhaps T in line 335 should be replaced by \theta.</p></disp-quote><p>Thank you, this is indeed a typo and should read c</p><disp-quote content-type="editor-comment"><p>7) Please provide figure numbers in lines 377, 412, and 414.</p></disp-quote><p>Thank you, we have added the references to the figures</p><disp-quote content-type="editor-comment"><p>Reviewer #1 (Recommendations for the authors):</p><p>In this work, the authors combine high resolution numerical simulations of turbulent airflow that advects a passive scalar (an &quot;odor&quot;) and machine learning algorithms to investigate the question of how a navigation strategy based on the temporal dynamics of odor detection compares with one based on the intensity of the plume. In the simulations, the odor is introduced downstream of a cylinder that induces turbulence in the flow and the machine learning algorithm is trained and evaluated at various points further downstream. The authors conclude that intensity/gradient based measurements work closer to the source, while temporal schemes are good throughout the range. The study is done to a very high standard and presented clearly, although there are general questions about the data analysis (see below) that should be improved. That said, the work should have significant impact on a broad range of fields, from sensing to navigation, across a range of organism length scales.</p><p>One concern is the nature of the turbulent profile and its advection of the passive scalar. It would help if the characteristic dimensionless numbers of the problem were specified (Peclet, Schmidt numbers), and it was made clear how the choice of odor release point affects the conclusions. The turbulence itself spreads and diffuses with distance from the source, and one would presume the location of the odor release can matter substantially.</p></disp-quote><p>We thank the reviewer for the comment. We have now stated the non-dimensional numbers and remarked that in our regime the Schmidt number is expected to play a minor role (see response to general comments above). To analyze how the results depend on source location, we have conducted a series of numerical simulations with the source located further downstream from its original location, and this does not affect our conclusion. We did not move the source laterally, as this cannot affect the results because the flow is homogeneous in the crosswind direction z2. We expect that moving the source to the ground will affect the results because intermittency of an odor plume from a source located at ground is much less intermittent (see response to general comment 1). Animals that would target both sources in air and on ground would clearly have to use qualitatively different predictive models. Ethological studies on dogs suggest that they do track both sources on ground and in air, and that they do this by alternating sniffing in air and on ground.</p><p>We are currently investigating this fascinating behavior.</p><disp-quote content-type="editor-comment"><p>Also, as mentioned briefly in the Discussion, this work examines algorithms based on evaluating accuracy of prediction at a given measurement point. Unless I have missed something in the presentation, the issue of navigation is left unexamined. As in bacterial chemotaxis or any related problem, no navigation strategy is perfect, especially in the face of such a fluctuating source of information, so the full problem involves making estimates of where the source is and then moving to a new location, estimating again, moving, etc. The authors should clarify under what circumstances an evaluation at a fixed point is sufficiently predictive of &quot;learning&quot;.</p></disp-quote><p>We agree entirely with the referee on this point, which we now stress more in the Discussion. We are currently working to test whether the best predictors are also the best features to be used in navigation. Although it is often implicitly assumed that this is indeed the case, up until now this question is entirely unexplored.</p><disp-quote content-type="editor-comment"><p>Reviewer #2 (Recommendations for the authors):</p><p>In their manuscript, Rigolli et al., studied how measurements of intensity of a passive scalar (odour), its spatial gradients, and its time variations can be used to efficiently find the spatial location of its source. To this end, the authors performed direct numerical simulations of high-Reynolds number pressure-driven channel flow, partially blocked by a cylinder to generate turbulent velocity fluctuations. At a fixed position downstream the cylinder, they introduce a source of odour that diffuses through the fluid and is advected by its turbulent motion. The authors then trained a supervised machine learning algorithm on a collection of spatial and temporal odour profiles thus obtained. The algorithm was subsequently tested on various odour signals originating from the same pool of simulations. By varying the spatial position of the measurement point and the temporal exposure, the authors concluded that detection strategies based on the odour concentration and its spatial gradient work best at small separations from the source (and also close to the zero-odour concentration boundaries), while time-based strategies work reasonably well everywhere within the domain selected for detection.</p><p>Within the geometry set by the authors, the conclusions of the paper are supported by data. However, this setup leads to a natural question whether the search strategies determined here pertain to the distance to the odour source or to the distance to the turbulence source (the half-cylinder). Turbulence generated by an obstacle has a particular spatial profile; its temporal profile also depends on the distance to the obstacle. It is therefore possible that the machine learning algorithm has indirectly picked up these features rather than the odour profile itself. This can be settled by feeding the existing algorithm signals from simulations with various distances between the source and the obstacle. In the absence of this check, it is impossible to make general statements about applicability of these search strategies in other situations.</p></disp-quote><p>Thank you for your comment. To test robustness of the algorithm, we have performed two additional sets of simulations moving the source further downstream of the obstacle: the results are consistent with the results of the main simulation; source location may affect details about where exactly the transition between the timing vs intensity regimes occurs. First, we verified that performance of 2 individual features and their pair does not depend on source location, confirming that the algorithm learns solely from odor statistics. Second, we trained the algorithm with data from one simulation and tested its performance over datasets from the remaining simulations.</p><p>We found that performance of individual features is preserved even when training and test are performed over different dataset. Pairs of features are more sensitive to consistency between training and test dataset (see Figure 2 Figure supplement 6). Third, we analysed performance for each source location as a function of height of the sampling plane. We recovered that timing features improve with height whereas intensity features degrade with height, as showed in Figure 5c (Figure 2 Figure Supplement 5, right). We note that the transition between timing vs intensity may occur at different heights for the three source locations, we did not investigate this comparison in more detail.</p><disp-quote content-type="editor-comment"><p>Another weak point of this study is the use of a cone of detection as it dramatically reduces the search complexity to almost a quasi one-dimensional problem. The authors appreciate it and point out that their choice is forced by a very limited by a very small number of detections outside the cone. This shortcoming should be addressed in future work.</p></disp-quote><p>We entirely agree with the referee, in fact olfactory searches first need to locate the plume and subsequently locate the source within the plume. Our manuscript defines the most informative features once the agent is within the plume. We believe that our results are not relevant for the previous stage when the agent searches for the plume. This is because there is hardly any detection at all outside of the cone. In fact, our ongoing work (currently under review) demonstrates a model based navigation strategy that uses absence of detection to narrow down the possible locations to search for the plume.</p></body></sub-article></article>