<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing with OASIS Tables v3.0 20080202//EN" "https://jats.nlm.nih.gov/nlm-dtd/publishing/3.0/journalpub-oasis3.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:oasis="http://docs.oasis-open.org/ns/oasis-exchange/table" xml:lang="en" dtd-version="3.0" article-type="research-article"><?xmltex \makeatother\@nolinetrue\makeatletter?>
  <front>
    <journal-meta><journal-id journal-id-type="publisher">AMT</journal-id><journal-title-group>
    <journal-title>Atmospheric Measurement Techniques</journal-title>
    <abbrev-journal-title abbrev-type="publisher">AMT</abbrev-journal-title><abbrev-journal-title abbrev-type="nlm-ta">Atmos. Meas. Tech.</abbrev-journal-title>
  </journal-title-group><issn pub-type="epub">1867-8548</issn><publisher>
    <publisher-name>Copernicus Publications</publisher-name>
    <publisher-loc>Göttingen, Germany</publisher-loc>
  </publisher></journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.5194/amt-17-1251-2024</article-id><title-group><article-title>A novel probabilistic source apportionment approach: Bayesian auto-correlated matrix factorization</article-title><alt-title>Bayesian auto-correlated matrix factorization</alt-title>
      </title-group><?xmltex \runningtitle{Bayesian auto-correlated matrix factorization}?><?xmltex \runningauthor{A. Rusanen et al.}?>
      <contrib-group>
        <contrib contrib-type="author" corresp="yes" rid="aff1 aff2">
          <name><surname>Rusanen</surname><given-names>Anton</given-names></name>
          <email>anton.rusanen@helsinki.fi</email>
        <ext-link>https://orcid.org/0000-0002-4523-9889</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff3">
          <name><surname>Björklund</surname><given-names>Anton</given-names></name>
          
        <ext-link>https://orcid.org/0000-0002-7749-2918</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff4">
          <name><surname>Manousakas</surname><given-names>Manousos I.</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff4 aff5">
          <name><surname>Jiang</surname><given-names>Jianhui</given-names></name>
          
        <ext-link>https://orcid.org/0000-0003-3557-3311</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff1 aff6 aff7">
          <name><surname>Kulmala</surname><given-names>Markku T.</given-names></name>
          
        <ext-link>https://orcid.org/0000-0003-3464-7825</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff1 aff3">
          <name><surname>Puolamäki</surname><given-names>Kai</given-names></name>
          
        <ext-link>https://orcid.org/0000-0003-1819-1047</ext-link></contrib>
        <contrib contrib-type="author" corresp="yes" rid="aff4">
          <name><surname>Daellenbach</surname><given-names>Kaspar R.</given-names></name>
          <email>kaspar.daellenbach@psi.ch</email>
        </contrib>
        <aff id="aff1"><label>1</label><institution>Institute for Atmospheric and Earth System Research (INAR)/Physics, Faculty of Science, University of Helsinki,<?xmltex \hack{\break}?> 00014 University of Helsinki, Finland</institution>
        </aff>
        <aff id="aff2"><label>2</label><institution>Atmospheric Composition Research, Finnish Meteorological Institute, 00101 Helsinki, Finland</institution>
        </aff>
        <aff id="aff3"><label>3</label><institution>Department of Computer Science, Faculty of Science, University of Helsinki, 00014 University of Helsinki, Finland</institution>
        </aff>
        <aff id="aff4"><label>4</label><institution>Laboratory of Atmospheric Chemistry, Paul Scherrer Institute (PSI), 5232 Villigen-PSI, Switzerland</institution>
        </aff>
        <aff id="aff5"><label>5</label><institution>Shanghai Key Lab for Urban Ecological Processes and Eco-Restoration, School of Ecological and Environmental Sciences, East China Normal University, 200241 Shanghai, China</institution>
        </aff>
        <aff id="aff6"><label>6</label><institution>Aerosol and Haze Laboratory, Beijing Advanced Innovation Center for Soft Matter Sciences and Engineering,<?xmltex \hack{\break}?> Beijing University of Chemical Technology (BUCT), 100029 Beijing, China</institution>
        </aff>
        <aff id="aff7"><label>7</label><institution>Joint International Research Laboratory of Atmospheric and Earth System Sciences, School of Atmospheric Sciences, Nanjing University, 210023 Nanjing, China</institution>
        </aff>
      </contrib-group>
      <author-notes><corresp id="corr1">Anton Rusanen (anton.rusanen@helsinki.fi) and Kaspar R. Daellenbach (kaspar.daellenbach@psi.ch)</corresp></author-notes><pub-date><day>22</day><month>February</month><year>2024</year></pub-date>
      
      <volume>17</volume>
      <issue>4</issue>
      <fpage>1251</fpage><lpage>1277</lpage>
      <history>
        <date date-type="received"><day>5</day><month>April</month><year>2023</year></date>
           <date date-type="rev-request"><day>12</day><month>May</month><year>2023</year></date>
           <date date-type="rev-recd"><day>23</day><month>November</month><year>2023</year></date>
           <date date-type="accepted"><day>6</day><month>December</month><year>2023</year></date>
      </history>
      <permissions>
        <copyright-statement>Copyright: © 2024 Anton Rusanen et al.</copyright-statement>
        <copyright-year>2024</copyright-year>
      <license license-type="open-access"><license-p>This work is licensed under the Creative Commons Attribution 4.0 International License. To view a copy of this licence, visit <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link></license-p></license></permissions><self-uri xlink:href="https://amt.copernicus.org/articles/amt-17-1251-2024.html">This article is available from https://amt.copernicus.org/articles/amt-17-1251-2024.html</self-uri><self-uri xlink:href="https://amt.copernicus.org/articles/amt-17-1251-2024.pdf">The full text article is available as a PDF file from https://amt.copernicus.org/articles/amt-17-1251-2024.pdf</self-uri>
      <abstract><title>Abstract</title>

      <p id="d1e182">The concentrations of atmospheric particulate matter and many of its constituents are temporally auto-correlated. However, this information has not been utilized in source apportionment methods. Here, we present a Bayesian matrix factorization model (BAMF) that considers the temporal auto-correlation of the components (sources) and provides a direct error estimation. The performance of BAMF is compared with positive matrix factorization (PMF) using synthetic Time-of-Flight Aerosol Chemical Speciation Monitor data, representing different urban environments from typical European towns to megacities. We find that BAMF resolves sources with overall higher factorization performance (temporal behavior and bias) than PMF on all datasets with temporally auto-correlated components. Highly correlated components continue to be challenging and ancillary information is still required to reach good factorizations. However, we demonstrate that adding even partial prior information about the chemical composition of the components to BAMF improves the factorization. Overall, BAMF-type models are promising tools for source apportionment and merit further research.</p>
  </abstract>
    
<funding-group>
<award-group id="gs1">
<funding-source>Research Council of Finland</funding-source>
<award-id>337549</award-id>
<award-id>345704</award-id>
<award-id>346376</award-id>
</award-group>
<award-group id="gs2">
<funding-source>Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung</funding-source>
<award-id>PZPGP2_201992</award-id>
</award-group>
<award-group id="gs3">
<funding-source>Science and Technology Commission of Shanghai Municipality</funding-source>
<award-id>21PJ1402800</award-id>
</award-group>
</funding-group>
</article-meta>
  </front>
<body>
      

<sec id="Ch1.S1" sec-type="intro">
  <label>1</label><title>Introduction</title>
      <p id="d1e194">Air pollution in the form of particulate matter (PM) has a substantial impact on the earth's climate <xref ref-type="bibr" rid="bib1.bibx22" id="paren.1"/> and severe adverse effects on human health <xref ref-type="bibr" rid="bib1.bibx27 bib1.bibx13" id="paren.2"/>. The noxiousness of PM could strongly depend on the chemical composition of the particles, which is governed by their origin <xref ref-type="bibr" rid="bib1.bibx2 bib1.bibx13" id="paren.3"/>. PM is affected by many emission sources and dynamic atmospheric processes, making PM a poorly understood complex mixture, especially the organic aerosol (OA) fraction of PM. Typically, directly emitted OA (primary OA – POA) is distinguished from OA formed in the atmosphere from emitted vapors by nucleation or condensation (secondary OA – SOA). Identifying and quantifying the sources of PM is, therefore, essential for designing effective and efficient air pollution reduction strategies. Such analyses (called “source apportionment”) combine chemical characterization data with non-negative matrix factorization methods. The idea is to use the variation in the chemical composition of a set of measurements, such as outputs from mass spectrometers, to decompose the measurements into “source terms” using non-negative matrix<?pagebreak page1252?> factorization. The underlying assumption is that the measurement is a linear combination of strictly non-negative source terms.</p>
      <p id="d1e206">Multiple methods for weighted non-negative matrix factorization exist <xref ref-type="bibr" rid="bib1.bibx38" id="paren.4"/>. A widely used method in atmospheric sciences is positive matrix factorization (PMF) <xref ref-type="bibr" rid="bib1.bibx30" id="paren.5"/>, which has been used in over 1000 papers <xref ref-type="bibr" rid="bib1.bibx20" id="paren.6"/>. In earlier studies, chemical mass balance (CMB) was a popular method, but it has the drawback that factor profiles must be defined beforehand (see, e.g., <xref ref-type="bibr" rid="bib1.bibx39" id="altparen.7"/>). This introduced significant uncertainty since these factor profiles are usually not known beforehand or only with considerable uncertainty. PMF improved on this by optimizing the source profiles <xref ref-type="bibr" rid="bib1.bibx4" id="paren.8"/>.</p>
      <p id="d1e224">Previous studies have revealed that chemical data from the Aerosol Mass Spectrometer family (Aerodyne Aerosol Mass Spectrometer, <xref ref-type="bibr" rid="bib1.bibx3" id="altparen.9"/>; Aerosol Chemical Speciation Monitor, <xref ref-type="bibr" rid="bib1.bibx29 bib1.bibx15" id="altparen.10"/>) retain sufficient information for the resolution of some sources <xref ref-type="bibr" rid="bib1.bibx41 bib1.bibx42 bib1.bibx12" id="paren.11"/>. However, distinguishing factors with chemical or temporal similarities or accurately resolving low-concentration factors is often challenging <xref ref-type="bibr" rid="bib1.bibx37 bib1.bibx4 bib1.bibx42 bib1.bibx17" id="paren.12"/>. Several studies have shown that utilizing a priori information to constrain the chemical composition or time series of POA sources is usually required to accurately estimate their contribution to OA <xref ref-type="bibr" rid="bib1.bibx4 bib1.bibx11 bib1.bibx31 bib1.bibx35 bib1.bibx43 bib1.bibx21 bib1.bibx44 bib1.bibx7" id="paren.13"/>. In addition, different statistical data reduction methods applied to mass spectrometry data extract different components <xref ref-type="bibr" rid="bib1.bibx23" id="paren.14"/>. This demonstrates that the problem does not have one unique solution, and the choice of method can emphasize different features of the resolved components.</p>
      <p id="d1e246">While developments related to source apportionment, in atmospheric science, focused on different ways to pre- and post-process data <xref ref-type="bibr" rid="bib1.bibx40" id="paren.15"/>, the underlying solver algorithm mainly remained the same: PMF. Rolling PMF <xref ref-type="bibr" rid="bib1.bibx5" id="paren.16"/> refers to a pre-processing strategy feeding only subsets of data (e.g., 7 or 14 d) to the PMF solver. This allows for a temporal variation of the chemical composition of sources (particularly relevant for SOA), even if their profiles remain static within each PMF run <xref ref-type="bibr" rid="bib1.bibx5" id="paren.17"/>.</p>
      <p id="d1e259">The commonly used optimization goal <inline-formula><mml:math id="M1" display="inline"><mml:mi>Q</mml:mi></mml:math></inline-formula> in PMF only accounts for reconstruction of the data <xref ref-type="bibr" rid="bib1.bibx38 bib1.bibx30" id="paren.18"/>. It lacks time information, which is a drawback considering some atmospheric measurements exhibit strong temporal auto-correlation (see, e.g., Fig. <xref ref-type="fig" rid="Ch1.F1"/>). Earlier studies have also found sources with longer cycles due to emissions, such as traffic, and meteorological conditions <xref ref-type="bibr" rid="bib1.bibx13 bib1.bibx9" id="paren.19"/>. Here, we present a probabilistic matrix factorization method that accounts for auto-correlation. We evaluate the model performance in resolving air pollution sources based on realistic synthetic chemical data.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F1" specific-use="star"><?xmltex \currentcnt{1}?><?xmltex \def\figurename{Figure}?><label>Figure 1</label><caption><p id="d1e279">The auto-correlations of the hourly means of several aerosol constituents measured at 19 different sites in Europe. The auto-correlation is the Pearson correlation coefficient between the original and the delayed time series. Auto-correlations at a lag of 1 and 2 h are very high in most cases. These data show that particulate matter constituents exhibit strong lag-1 auto-correlation, consistent with earlier such statements in the literature (e.g., <xref ref-type="bibr" rid="bib1.bibx18" id="altparen.20"/>). Auto-correlations calculated on data from <xref ref-type="bibr" rid="bib1.bibx8" id="text.21"/>.</p></caption>
        <?xmltex \igopts{width=426.791339pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f01.png"/>

      </fig>

</sec>
<sec id="Ch1.S2">
  <label>2</label><title>Methods</title>
      <p id="d1e302">In this section we discuss the methods used in the paper, starting with notation in Sect. <xref ref-type="sec" rid="Ch1.S2.SS1"/>. We define the BAMF model in Sect. <xref ref-type="sec" rid="Ch1.S2.SS2"/> and describe how we find solutions in Sect. <xref ref-type="sec" rid="Ch1.S2.SS3"/> using the pre- and post-processing steps in Sect. <xref ref-type="sec" rid="Ch1.S2.SS4"/>. In Sect. <xref ref-type="sec" rid="Ch1.S2.SS5"/> we describe how we compare BAMF with PMF, which is defined in Sect. <xref ref-type="sec" rid="Ch1.S2.SS6"/>.</p><?xmltex \hack{\vspace{2.5mm}}?>
<sec id="Ch1.S2.SS1">
  <label>2.1</label><title>Notation</title>
      <p id="d1e326">In this paper, we describe the data by <inline-formula><mml:math id="M2" display="inline"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi><mml:mo>∈</mml:mo><mml:msup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>×</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, where the rows <inline-formula><mml:math id="M3" display="inline"><mml:mrow><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo>[</mml:mo><mml:mi>n</mml:mi><mml:mo>]</mml:mo><mml:mo>=</mml:mo><mml:mo mathvariant="italic">{</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula> correspond to measurements taken at consecutive times <inline-formula><mml:math id="M4" display="inline"><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. The columns <inline-formula><mml:math id="M5" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:mo>[</mml:mo><mml:mi>m</mml:mi><mml:mo>]</mml:mo><mml:mo>=</mml:mo><mml:mo mathvariant="italic">{</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula> correspond to the different dimensions of the measurement. Our objective is to find a lower dimensional non-negative decomposition <inline-formula><mml:math id="M6" display="inline"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi><mml:mo>≈</mml:mo><mml:mi mathvariant="bold">GF</mml:mi></mml:mrow></mml:math></inline-formula> with <inline-formula><mml:math id="M7" display="inline"><mml:mi>p</mml:mi></mml:math></inline-formula> factors such that <inline-formula><mml:math id="M8" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mo>≥</mml:mo><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mn mathvariant="normal">0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>×</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M9" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mo>≥</mml:mo><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mn mathvariant="normal">0</mml:mn></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>×</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="M10" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mo>≪</mml:mo><mml:mo>min⁡</mml:mo><mml:mo>(</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. In other words, the objective is to present the data as a multiplication of two much smaller matrices. The rows of <inline-formula><mml:math id="M11" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="bold">F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>⋅</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> contain the time-independent components of the decomposition, which we call <italic>factor profiles</italic>. Factor profiles are defined to sum to unity in order to facilitate comparisons between different datasets and models. The columns of <inline-formula><mml:math id="M12" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="bold">G</mml:mi><mml:mrow><mml:mo>⋅</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> contain the time dependency of each of the rows of <inline-formula><mml:math id="M13" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>; we will call these the <italic>factor time series</italic>. Simply put, the factor profiles represent the concentration and time-independent chemical composition of the sources, and the factor time series describes the time-dependent concentration of the sources. Note that the ordering of these profiles is arbitrary for the overall solution.</p><?xmltex \hack{\vspace{2.5mm}}?>
</sec>
<sec id="Ch1.S2.SS2">
  <label>2.2</label><title>Bayesian auto-correlated matrix factorization, BAMF</title>
      <?pagebreak page1253?><p id="d1e571">We define a Bayesian probabilistic model that captures our prior assumptions of the process that generated the measurements. The only observed variables in our model are the data matrix <inline-formula><mml:math id="M14" display="inline"><mml:mrow><mml:mi mathvariant="bold">X</mml:mi><mml:mo>∈</mml:mo><mml:msup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>×</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> and the uncertainty estimate <inline-formula><mml:math id="M15" display="inline"><mml:mrow><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mo>≥</mml:mo><mml:mn mathvariant="normal">0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>×</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>. Together they define the probability distribution of the observed concentration of each <inline-formula><mml:math id="M16" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> at any point in time with the data matrix being the average and the uncertainty matrix the standard deviation of the distribution, here the Gaussian distribution. In addition to these observed variables, there are several latent variables. These include the matrices <inline-formula><mml:math id="M17" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M18" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> mentioned above, vectors <inline-formula><mml:math id="M19" display="inline"><mml:mrow><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mo>≥</mml:mo><mml:mn mathvariant="normal">0</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M20" display="inline"><mml:mrow><mml:mi mathvariant="bold-italic">β</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mo>≥</mml:mo><mml:mn mathvariant="normal">0</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> which determine the auto-correlation behavior of the model, as well the “noise-free data matrix” <inline-formula><mml:math id="M21" display="inline"><mml:mrow><mml:mi mathvariant="bold">Z</mml:mi><mml:mo>∈</mml:mo><mml:msubsup><mml:mi mathvariant="double-struck">R</mml:mi><mml:mrow><mml:mo>≥</mml:mo><mml:mn mathvariant="normal">0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>×</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>. We define the probabilistic model as

                <disp-formula specific-use="gather" content-type="numbered"><mml:math id="M22" display="block"><mml:mtable displaystyle="true"><mml:mlabeledtr id="Ch1.E1"><mml:mtd><mml:mtext>1</mml:mtext></mml:mtd><mml:mtd><mml:mrow><mml:mstyle displaystyle="true" class="stylechange"/><?xmltex \hack{\hspace*{7mm}}?><mml:mi mathvariant="bold">Z</mml:mi><mml:mo>=</mml:mo><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mi mathvariant="bold">F</mml:mi></mml:mrow></mml:mtd></mml:mlabeledtr><mml:mlabeledtr id="Ch1.E2"><mml:mtd><mml:mtext>2</mml:mtext></mml:mtd><mml:mtd><mml:mrow><mml:mstyle class="stylechange" displaystyle="true"/><mml:mtable rowspacing="0.2ex" class="split" displaystyle="true" columnalign="right left"><mml:mtr><mml:mtd><mml:mrow><?xmltex \hack{\hspace*{4mm}}?><mml:msub><mml:mi mathvariant="bold-italic">X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>∼</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mtext>Normal</mml:mtext><mml:mfenced close=")" open="("><mml:mrow><mml:mtext>location</mml:mtext><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">Z</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>scale</mml:mtext><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mtext>for all</mml:mtext><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo>[</mml:mo><mml:mi>n</mml:mi><mml:mo>]</mml:mo><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mtext>and</mml:mtext><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:mo>[</mml:mo><mml:mi>m</mml:mi><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mtd></mml:mlabeledtr><mml:mlabeledtr id="Ch1.E3"><mml:mtd><mml:mtext>3</mml:mtext></mml:mtd><mml:mtd><mml:mrow><mml:mstyle class="stylechange" displaystyle="true"/><?xmltex \hack{\hspace*{5mm}}?><mml:msub><mml:mi mathvariant="bold">F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>⋅</mml:mo></mml:mrow></mml:msub><mml:mo>∼</mml:mo><mml:mtext>Dirichlet</mml:mtext><mml:mo mathsize="1.1em">(</mml:mo><?xmltex \igopts{height=7.397717pt}?><mml:mstyle background="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-g01.png"/><mml:mo mathsize="1.1em">)</mml:mo><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mtext>for all</mml:mtext><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo>[</mml:mo><mml:mi>p</mml:mi><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mlabeledtr><mml:mlabeledtr id="Ch1.E4"><mml:mtd><mml:mtext>4</mml:mtext></mml:mtd><mml:mtd><mml:mrow><mml:mstyle class="stylechange" displaystyle="true"/><mml:mtable class="split" rowspacing="0.2ex" displaystyle="true" columnalign="right left"><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi mathvariant="bold">G</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>∼</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mspace width="0.125em" linebreak="nobreak"/><?xmltex \hack{\hbox\bgroup\fontsize{9.5}{9.5}\selectfont$\displaystyle}?><mml:mtext mathvariant="normal">Cauchy</mml:mtext><mml:mfenced close=")" open="("><mml:mrow><mml:mtext>location</mml:mtext><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="bold">G</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mtext>scale</mml:mtext><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">β</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:mfenced><?xmltex \hack{$\egroup}?></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mtext>for all</mml:mtext><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mi>k</mml:mi><mml:mo>∈</mml:mo><mml:mo>[</mml:mo><mml:mi>p</mml:mi><mml:mo>]</mml:mo><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mtext>and</mml:mtext><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mo>[</mml:mo><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>]</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mtd></mml:mlabeledtr></mml:mtable></mml:math></disp-formula>

            where Normal corresponds to the normal probability distribution with a given mean and standard deviation, and Dirichlet to the Dirichlet distribution parameterized by a unit vector <bold><italic>1</italic></bold><inline-formula><mml:math id="M23" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="bold-italic">m</mml:mi></mml:msub></mml:math></inline-formula>. The model specification implies that the components of <inline-formula><mml:math id="M24" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="bold">F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>⋅</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> can have values from <inline-formula><mml:math id="M25" display="inline"><mml:mrow><mml:mo>[</mml:mo><mml:mn mathvariant="normal">0</mml:mn><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> with equal likelihood, but all rows must sum to unity. Cauchy is the Cauchy probability distribution, with the width depending on the time difference between the <inline-formula><mml:math id="M26" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula>th and <inline-formula><mml:math id="M27" display="inline"><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>th observation (<inline-formula><mml:math id="M28" display="inline"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>). Essentially our model describes the data as a non-negative matrix decomposition (NMF) with a lag-1 auto-correlation term and a Gaussian error term for the reconstruction of <inline-formula><mml:math id="M29" display="inline"><mml:mi mathvariant="bold">X</mml:mi></mml:math></inline-formula>.</p>
      <p id="d1e1081">We chose the Cauchy distribution for the auto-correlation term because the long tails make large jumps between the <inline-formula><mml:math id="M30" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula>th and <inline-formula><mml:math id="M31" display="inline"><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>th observation more probable than, e.g., a Gaussian distribution. Other choices are possible, but the experiments in this paper suggest the Cauchy distribution works as an approximation for real data. Choosing the distribution shape also implicitly influences the weight of the <inline-formula><mml:math id="M32" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> auto-correlation. The <inline-formula><mml:math id="M33" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi mathvariant="bold-italic">β</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> determines the scale of the Cauchy distribution. The <inline-formula><mml:math id="M34" display="inline"><mml:mi mathvariant="italic">α</mml:mi></mml:math></inline-formula> terms allow the model to deal with time steps of different lengths and missing data, since it forms a simple linear model for the scale. It has a minimum width <inline-formula><mml:math id="M35" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">β</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and increases linearly as <inline-formula><mml:math id="M36" display="inline"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:math></inline-formula> increases (with <inline-formula><mml:math id="M37" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">α</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>). Thus, for arbitrarily large time steps the Cauchy approaches a uniform distribution. In physical terms this means that at short time steps we expect the values of <inline-formula><mml:math id="M38" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> to stay close to the previous value, and at large time steps larger deviations have higher probability. It is possible to use other formulations for the width, which would be appropriate if one wishes to include a more complex and computationally intensive description of auto-correlation. It is also possible to consider more than lag-1 auto-correlation.</p>
<sec id="Ch1.S2.SS2.SSS1">
  <label>2.2.1</label><title>Uncorrelated Bayesian matrix factorization, BAMF-0</title>
      <p id="d1e1193">For comparison, we created a version of the BAMF model without the lag-1 auto-correlation terms of Eq. (<xref ref-type="disp-formula" rid="Ch1.E4"/>). In other words, the model consist entirely of Eqs. (<xref ref-type="disp-formula" rid="Ch1.E1"/>)–(<xref ref-type="disp-formula" rid="Ch1.E3"/>). The model is otherwise identical to the BAMF model. This variation is<?pagebreak page1254?> essentially a probabilistic weighted NMF model, making it possible to assess the impact of the auto-correlation terms on the solution.</p>
</sec>
<sec id="Ch1.S2.SS2.SSS2">
  <label>2.2.2</label><title>Bayesian matrix factorization with additional constraints, BAMF-C</title>
      <p id="d1e1210">In source apportionment analyses, it is common to utilize reference spectra as boundary conditions for the factor analysis – to find components with, e.g., previously observed chemical compositions. We include this scenario in another model, called “BAMF-C”, by adding peak intensity ratios to the BAMF model,
              <disp-formula id="Ch1.E5" content-type="numbered"><label>5</label><mml:math id="M39" display="block"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mo>[</mml:mo><mml:mi>i</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>]</mml:mo><mml:mo>/</mml:mo><mml:mi mathvariant="bold">F</mml:mi><mml:mo>[</mml:mo><mml:mi>k</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>]</mml:mo><mml:mo>∼</mml:mo><mml:mtext>Normal</mml:mtext><mml:mfenced close=")" open="("><mml:mrow><mml:mtext>ratio</mml:mtext><mml:mo>,</mml:mo><mml:mtext>width</mml:mtext></mml:mrow></mml:mfenced><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
            where <inline-formula><mml:math id="M40" display="inline"><mml:mrow><mml:mi>i</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M41" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M42" display="inline"><mml:mrow><mml:mi>k</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M43" display="inline"><mml:mrow><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:math></inline-formula> are the indices of matrix <inline-formula><mml:math id="M44" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> indicating the peak pair to constrain. The ratio is the desired intensity ratio, and width is a free parameter describing the width of the distribution, i.e., the uncertainty of the intensity ratio. This approach makes it possible to constrain the range of <inline-formula><mml:math id="M45" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> for arbitrarily many <inline-formula><mml:math id="M46" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> pairs, which is currently not done in the other models. A similar concept could constrain <inline-formula><mml:math id="M47" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> to have a similar time behavior as an external ancillary measurement, e.g., of a source tracer. The constraint is similar to the widely used <inline-formula><mml:math id="M48" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula> value (anchor value) approach in Source Finder coupled to PMF (<xref ref-type="bibr" rid="bib1.bibx4" id="altparen.22"/>, SoFi/PMF) with two notable differences. Firstly, the intensity ratio of the peaks is constrained in BAMF-C, while the <inline-formula><mml:math id="M49" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula>-value constraint approach in SoFi/PMF uses the peak intensity. Secondly, BAMF-C has a soft-boundary, with increasing penalty term as distance from the anchor grows. SoFi/PMF employs a hard boundary (defined as a relative deviation from the <inline-formula><mml:math id="M50" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula> value), without additional penalty on the object function <inline-formula><mml:math id="M51" display="inline"><mml:mi>Q</mml:mi></mml:math></inline-formula> for deviation from the anchor. In the present study, we evaluate the performance of the PMF and BAMF algorithm itself without discarding sub-optimal solutions (PMF) or samples (BAMF) during post-processing. Appendix <xref ref-type="sec" rid="App1.Ch1.S6"/> lists the profiles used for constraints in this work.</p>
</sec>
</sec>
<sec id="Ch1.S2.SS3">
  <label>2.3</label><title>Solver</title>
      <p id="d1e1386">We use Stan <xref ref-type="bibr" rid="bib1.bibx6" id="paren.23"/> to compile and run the probabilistic models. Stan solves the probabilistic inference problem with a Markov chain Monte Carlo (MCMC) method. Instead of obtaining a single solution for the latent variables, e.g., by finding the model with the highest likelihood, we get an empirical distribution of possible solutions from which we can infer, e.g., confidence intervals. Stan takes our model and observed variables as input and outputs samples from the posterior distribution of the latent variables. We run multiple MCMC chains, starting from different initial conditions and usually extract a few thousand posterior samples per chain.</p>
      <p id="d1e1392">The standard way to initialize the model in Stan is by randomly sampling from the prior distributions. However, our model has many parameters with fairly strict distributions. Consequently, we found this starting point to be poor, sometimes causing Stan to markedly slow down, or even fail. Hence, we initialize the model with a point solution. We utilize Stan's capability to find a single maximum a posteriori (MAP) point solution for the parameters, which we use as the initialization. Note, however, that the solutions typically have several local optima, in which case the point solution is only one such local optimum.</p>
<sec id="Ch1.S2.SS3.SSSx1" specific-use="unnumbered">
  <title>Hamiltonian Markov chain Monte Carlo</title>
      <p id="d1e1400">Stan <xref ref-type="bibr" rid="bib1.bibx6" id="paren.24"/> uses a Hamiltonian MCMC method to draw samples from the posterior distribution of our model (Eqs. <xref ref-type="disp-formula" rid="Ch1.E1"/>–<xref ref-type="disp-formula" rid="Ch1.E4"/>) given the data. We go through the basic idea here, but direct readers to <xref ref-type="bibr" rid="bib1.bibx6" id="text.25"/>, <xref ref-type="bibr" rid="bib1.bibx16" id="text.26"/> and references therein for a more detailed explanation.</p>
      <p id="d1e1416">The samples are drawn in proportion to the posterior probability of each sample. Obtaining samples from a multidimensional posterior distribution is a non-trivial task. For effective sampling, we use Hamiltonian Monte Carlo (HMC), a method where the gradient of the distribution and an ancillary variable called momentum are used to direct the chain to explore the typical set <xref ref-type="bibr" rid="bib1.bibx16" id="paren.27"/>. Specifically, we use the HMC based No-U-Turn Sampler (NUTS) <xref ref-type="bibr" rid="bib1.bibx19" id="paren.28"/> from Stan. Stan uses warm-up iterations to estimate the parameters the NUTS sampler needs before drawing the posterior samples used in the computations <xref ref-type="bibr" rid="bib1.bibx6" id="paren.29"/>.</p>
      <p id="d1e1428">The MAP estimate is used as a starting point for the sampling. It is found with an optimization method on the same probability distribution used by the sampling. We use the LBFGS <xref ref-type="bibr" rid="bib1.bibx28" id="paren.30"/> gradient-based optimization algorithm included in Stan for MAP estimates <xref ref-type="bibr" rid="bib1.bibx6" id="paren.31"/>.</p>
</sec>
</sec>
<sec id="Ch1.S2.SS4">
  <label>2.4</label><title>Pre- and post-processing</title>
      <p id="d1e1446">Before running our model, we normalize the data such that the mean of the data (<inline-formula><mml:math id="M52" display="inline"><mml:mi mathvariant="bold">X</mml:mi></mml:math></inline-formula>) is 1. The error estimate is scaled with the same scaling factor. The equations to do this are:

                <disp-formula id="Ch1.Ex1"><mml:math id="M53" display="block"><mml:mrow><mml:mstyle displaystyle="true" class="stylechange"/><mml:mtable class="array" columnalign="right center left"><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi mathvariant="normal">norm</mml:mi></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>=</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mi mathvariant="bold">X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:mo>(</mml:mo><mml:mi>n</mml:mi><mml:mo>×</mml:mo><mml:mi>m</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi mathvariant="bold">X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:mtd><mml:mtd><mml:mo>=</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi mathvariant="bold">X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi mathvariant="normal">norm</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:mtd><mml:mtd><mml:mo>=</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mi mathvariant="normal">norm</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>

          where <inline-formula><mml:math id="M54" display="inline"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi mathvariant="normal">norm</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is a scalar normalization factor, <inline-formula><mml:math id="M55" display="inline"><mml:mrow><mml:msubsup><mml:mi mathvariant="bold">X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M56" display="inline"><mml:mrow><mml:msubsup><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> are the scaled data and error, respectively, which we use as model inputs. This normalization is done so that we can use a consistent scale for priors and posteriors, making the modeling easier, and the denormalization is performed to return the results to familiar units. The normalization is optional, a user can also choose to use non-scaled values.</p>
      <?pagebreak page1255?><p id="d1e1618">Stan outputs posterior samples from the two matrices <inline-formula><mml:math id="M57" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M58" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, representing the time-independent chemical composition of the factors and their time-dependent concentration. Since all the magnitude information is in <inline-formula><mml:math id="M59" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="M60" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> needs to be renormalized by simply multiplying with the normalization factor. The rows of <inline-formula><mml:math id="M61" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> are constrained to sum to unity, and are, thus, directly comparable to mass spectrometric references, which are normalized similarly <xref ref-type="bibr" rid="bib1.bibx10 bib1.bibx36" id="paren.32"/>.</p>
<sec id="Ch1.S2.SS4.SSSx1" specific-use="unnumbered">
  <title>Sorting the components</title>
      <p id="d1e1665">The order of the components in <inline-formula><mml:math id="M62" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M63" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> is arbitrary in our samples. The problem is not unique to BAMF but inherent to all such matrix decompositions. To be able to compare solutions, we need to be able to sort the components. The contribution of the same component to <inline-formula><mml:math id="M64" display="inline"><mml:mi mathvariant="bold">Z</mml:mi></mml:math></inline-formula> should be similar between two samples. To calculate the contribution for each component, we multiply the row of <inline-formula><mml:math id="M65" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> with the corresponding column of <inline-formula><mml:math id="M66" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> and use this to sort the components.</p>
      <p id="d1e1703">To select the ordering of the components, we take a small number of representative samples, usually the last five, and compute the optimal permutation using the Hungarian algorithm <xref ref-type="bibr" rid="bib1.bibx25" id="paren.33"/>, which is a cost minimization algorithm that minimizes the cost of assigning values. In this case we are minimizing the Manhattan distances, which is the sum of absolute differences between the <inline-formula><mml:math id="M67" display="inline"><mml:mi mathvariant="bold">Z</mml:mi></mml:math></inline-formula> contributions in the samples. We then select the most common permutation as the ordering of the factors for all samples.</p>
      <p id="d1e1716">We use this approach for sorting the outputs of all models (BAMF-0/C, BAMF, PMF) to ensure the most direct comparability of the results. Finally, median, 25 %, and 75 % percentiles are computed using the sorted samples. In the comparisons, we use medians for all models, but in some figures, we also show 25 % and 75 % percentiles. The median, or any central estimate, is not guaranteed to be the “best” optimized solution in any metric (probability or root mean square sum of residuals). Still, we use it to represent a reasonable solution inferred from the samples.</p>
</sec>
</sec>
<sec id="Ch1.S2.SS5">
  <label>2.5</label><title>Evaluating model performance</title>
      <p id="d1e1728">The first metric to check is if the model explains the data well (<italic>reconstruction performance</italic>). If <inline-formula><mml:math id="M68" display="inline"><mml:mi mathvariant="bold">X</mml:mi></mml:math></inline-formula> is not reconstructed appropriately, the solution is not acceptable. This can mean either that the data cannot be factorized this way (the model assumptions are wrong) or that the solver failed to find a solution. In such cases, the number of iterations should be increased, or solver parameters must be adjusted, such as the number of warm-up samples and parameters influencing the step size.</p>
      <p id="d1e1741">Even at moderate data sizes, assessing whether the original data falls within the model confidence bounds for every variable individually is not practical. Therefore, we summarized this information by computing the model residual (difference between the model input and output, mean of all samples) normalized with the uncertainty of the model output (standard deviation of all samples); essentially observing whether the original data are inside the sample standard deviation.
            <disp-formula id="Ch1.E6" content-type="numbered"><label>6</label><mml:math id="M69" display="block"><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">S</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:msub><mml:mi mathvariant="bold">X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>E</mml:mi><mml:mfenced open="[" close="]"><mml:mrow><mml:msub><mml:mi mathvariant="bold">X</mml:mi><mml:mi mathvariant="normal">samples</mml:mi></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mfenced><mml:mo>/</mml:mo><mml:mi mathvariant="italic">σ</mml:mi><mml:mfenced open="(" close=")"><mml:mrow><mml:msub><mml:mi mathvariant="bold">X</mml:mi><mml:mi mathvariant="normal">samples</mml:mi></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:math></disp-formula>
          Data in <inline-formula><mml:math id="M70" display="inline"><mml:mi mathvariant="bold-italic">S</mml:mi></mml:math></inline-formula> should be centered at 0 and have a standard deviation below 1, which means the data are often less than 1 model standard deviation away from the model mean. In addition, we also use a common evaluation metric in PMF analyses (<inline-formula><mml:math id="M71" display="inline"><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">exp</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>), where <inline-formula><mml:math id="M72" display="inline"><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is defined as (<xref ref-type="bibr" rid="bib1.bibx4" id="altparen.34"/>, notation adapted):
            <disp-formula id="Ch1.E7" content-type="numbered"><label>7</label><mml:math id="M73" display="block"><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:mo mathsize="2.0em">(</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:msub><mml:mfenced close=")" open="("><mml:mrow><mml:mi mathvariant="bold">X</mml:mi><mml:mo>-</mml:mo><mml:mi mathvariant="bold">Z</mml:mi></mml:mrow></mml:mfenced><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle><mml:msup><mml:mo mathsize="2.0em">)</mml:mo><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
          Essentially <inline-formula><mml:math id="M74" display="inline"><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> describes the sum of squared model residuals normalized to the input error. We use the same reduction as <xref ref-type="bibr" rid="bib1.bibx42" id="text.35"/> where <inline-formula><mml:math id="M75" display="inline"><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">exp</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is approximated as data size and denote <inline-formula><mml:math id="M76" display="inline"><mml:mrow><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">exp</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> as <inline-formula><mml:math id="M77" display="inline"><mml:mrow><mml:msubsup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>.</p>
      <p id="d1e1951">For synthetic data – with a known ground truth – it is possible to assess how well the methods resolve the actual components in addition to the reconstruction performance. We call this evaluation <italic>factorization performance</italic>. We compare the median solutions with the corresponding actual components by calculating the average distance, Pearson and Spearman (nonlinear) correlations. Optimal matching between median solutions and true components is obtained using the Hungarian algorithm <xref ref-type="bibr" rid="bib1.bibx25" id="paren.36"/>. The approach is similar to the sorting above but with the true components defining the order. Direct comparison with true components is only possible in cases where the number of true components matches the number of modeled components. Otherwise, the model must combine multiple components or create additional ones.</p>
</sec>
<sec id="Ch1.S2.SS6">
  <label>2.6</label><title>PMF</title>
      <p id="d1e1968">We use PMF, specifically the multilinear engine 2 (ME-2) controlled by the user interface SoFi <xref ref-type="bibr" rid="bib1.bibx4 bib1.bibx30" id="paren.37"/>, as a baseline comparison. PMF solves the decomposition in Eq. (<xref ref-type="disp-formula" rid="Ch1.E1"/>) by minimizing the sum of the squared residuals normalized with the input error – see Eq. (<xref ref-type="disp-formula" rid="Ch1.E7"/>) (object function) – given the boundary condition that all values must be positive <xref ref-type="bibr" rid="bib1.bibx4" id="paren.38"/>. Since SoFi finds local optima, we ran it with different random seeds to get multiple solutions for all comparisons. As the runs have varying starting points, they often lead to different local optima, especially in cases with high rotational ambiguity. Thus, PMF provides a collection of local minima, while BAMF tries to sample the posterior distribution of the model, including plausible answers that are not minima. For simplicity, we will refer to the group of PMF solution sets as “samples”, even though PMF is not a sampler.</p>
      <p id="d1e1981">A priori information in the form of known rows of factor profiles or of known columns of factor time series can be added to the model to reduce the rotational ambiguity. By adding this external information, the user can reduce the<?pagebreak page1256?> space PMF searches for the optimized solution, reducing the rotational ambiguity of the solution. Using external data to run PMF is usually referred to as “constraining the solution”, and external information is used as constraints. Here we used two approaches, (a) entirely unconstrained PMF runs and (b) constrained runs using external source profiles. We rely on the commonly used <inline-formula><mml:math id="M78" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula>-value approach to constrain the PMF runs. In the <inline-formula><mml:math id="M79" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula>-value approach, the user inputs one or more factor profiles or factor time series and defines a relative tolerated deviation from the anchor (termed “<inline-formula><mml:math id="M80" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula> value”) <xref ref-type="bibr" rid="bib1.bibx4" id="paren.39"/>. Constraint strengths are not directly comparable between the hard cut-off approach used in PMF and the softer Gaussian error term approach used in BAMF-C. For the best possible comparability, we first ran BAMF-C (constraint strength 0.001) and determined the equivalent <inline-formula><mml:math id="M81" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula> value by taking the maximum deviation from the anchor value on a constrained component in <inline-formula><mml:math id="M82" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> (<inline-formula><mml:math id="M83" display="inline"><mml:mi>a</mml:mi></mml:math></inline-formula> value of 18 %). See Appendix <xref ref-type="sec" rid="App1.Ch1.S6"/> for the profiles used to make this comparison.</p>
</sec>
</sec>
<sec id="Ch1.S3">
  <label>3</label><title>Datasets</title>
      <p id="d1e2041">We generated synthetic datasets mimicking the OA sources in different urban environments. These synthetic datasets mimic mass spectral OA analyses of a Time-of-Flight Aerosol Chemical Speciation Monitor (ToF-ACSM, <xref ref-type="bibr" rid="bib1.bibx15" id="altparen.40"/>), a ubiquitous instrument in measuring PM composition with a focus on OA. The instrument measures a signal for several mass-to-charge ratio channels. We model these mass spectra as a sum of 2–5 different mixed sources. The sources are constructed of time-independent chemical fingerprints, in our notation <inline-formula><mml:math id="M84" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, from the AMS Spectral Database <xref ref-type="bibr" rid="bib1.bibx37 bib1.bibx36" id="paren.41"/> combined with their time behavior and magnitude, <inline-formula><mml:math id="M85" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>. The noiseless spectra, <inline-formula><mml:math id="M86" display="inline"><mml:mi mathvariant="bold">Z</mml:mi></mml:math></inline-formula>, are then acquired by matrix multiplication of <inline-formula><mml:math id="M87" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M88" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>. We then generate <inline-formula><mml:math id="M89" display="inline"><mml:mi mathvariant="bold">X</mml:mi></mml:math></inline-formula>, as Eq. (<xref ref-type="disp-formula" rid="Ch1.E2"/>), by applying random Gaussian noise to each data point. The errors are applied to <inline-formula><mml:math id="M90" display="inline"><mml:mi mathvariant="bold">X</mml:mi></mml:math></inline-formula>, which is a sum of all the components, so that individual component error in the original <inline-formula><mml:math id="M91" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> is undefined.</p>
      <p id="d1e2109">The ToF-ACSM alternates between measuring particles and air together, called “open signal” (<inline-formula><mml:math id="M92" display="inline"><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">open</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>), and measuring only air, called “closed signal” (<inline-formula><mml:math id="M93" display="inline"><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">closed</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>). The difference signal (<inline-formula><mml:math id="M94" display="inline"><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">diff</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">open</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">closed</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>) represents the signal caused by the measured particles. For computing the error of <inline-formula><mml:math id="M95" display="inline"><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">diff</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, we use an error function based on the signal strength, according to <xref ref-type="bibr" rid="bib1.bibx1" id="text.42"/>, and <xref ref-type="bibr" rid="bib1.bibx37" id="text.43"/>:
          <disp-formula id="Ch1.E8" content-type="numbered"><label>8</label><mml:math id="M96" display="block"><mml:mrow><mml:msub><mml:mtext>Error</mml:mtext><mml:mi mathvariant="normal">open</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:mfenced open="(" close=")"><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">open</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">baseline</mml:mi></mml:msub></mml:mrow></mml:mfenced><mml:mo>×</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi mathvariant="normal">open</mml:mi></mml:msub></mml:mrow><mml:msqrt><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mn mathvariant="normal">28</mml:mn><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:mfrac></mml:mstyle></mml:msqrt></mml:mfrac></mml:mstyle></mml:msqrt></mml:mrow></mml:math></disp-formula>

          <disp-formula id="Ch1.E9" content-type="numbered"><label>9</label><mml:math id="M97" display="block"><mml:mrow><mml:msub><mml:mtext>Error</mml:mtext><mml:mi mathvariant="normal">closed</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:mfenced close=")" open="("><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">closed</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">baseline</mml:mi></mml:msub></mml:mrow></mml:mfenced><mml:mo>×</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi mathvariant="normal">closed</mml:mi></mml:msub></mml:mrow><mml:msqrt><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mn mathvariant="normal">28</mml:mn><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:mfrac></mml:mstyle></mml:msqrt></mml:mfrac></mml:mstyle></mml:msqrt></mml:mrow></mml:math></disp-formula>

          <disp-formula id="Ch1.E10" content-type="numbered"><label>10</label><mml:math id="M98" display="block"><mml:mrow><?xmltex \hack{\hbox\bgroup\fontsize{9.5}{9.5}\selectfont$\displaystyle}?><mml:mtext mathvariant="normal">Error</mml:mtext><mml:mo>=</mml:mo><mml:mo movablelimits="false">max⁡</mml:mo><mml:mo mathsize="2.0em">(</mml:mo><mml:msub><mml:mtext>Error</mml:mtext><mml:mo>min⁡</mml:mo></mml:msub><mml:mo>,</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:mn mathvariant="normal">1.2</mml:mn><mml:mo>×</mml:mo><mml:msqrt><mml:mrow><mml:msubsup><mml:mtext>Error</mml:mtext><mml:mi mathvariant="normal">open</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mtext>Error</mml:mtext><mml:mi mathvariant="normal">closed</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msubsup></mml:mrow></mml:msqrt></mml:mrow><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mi mathvariant="normal">open</mml:mi></mml:msub><mml:mo>×</mml:mo><mml:msqrt><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mn mathvariant="normal">28</mml:mn><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:mfrac></mml:mstyle></mml:msqrt></mml:mrow></mml:mfrac></mml:mstyle><mml:mo mathsize="2.0em">)</mml:mo><mml:mo>,</mml:mo><?xmltex \hack{$\egroup}?></mml:mrow></mml:math></disp-formula>
        where <inline-formula><mml:math id="M99" display="inline"><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mi mathvariant="normal">open</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M100" display="inline"><mml:mrow><mml:msub><mml:mi>t</mml:mi><mml:mi mathvariant="normal">closed</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> are the open and closed signal measurement times, respectively, <inline-formula><mml:math id="M101" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> is the mass charge ratio of the measurement, Error<inline-formula><mml:math id="M102" display="inline"><mml:msub><mml:mi/><mml:mo>min⁡</mml:mo></mml:msub></mml:math></inline-formula> is a lower limit set on the measurement error, and <inline-formula><mml:math id="M103" display="inline"><mml:mrow><mml:msub><mml:mi>I</mml:mi><mml:mi mathvariant="normal">baseline</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is the baseline signal in the mass spectrometer. Since the organic fragment ions at the <inline-formula><mml:math id="M104" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> values 16, 17, 18, and 28 are computed based on the measurement at <inline-formula><mml:math id="M105" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> 44 and thus contain duplicate information, they are removed before running any of the models and only later reintroduced in the results.</p>
<sec id="Ch1.S3.SS1">
  <label>3.1</label><title>Synthetic data representing a polluted megacity</title>
      <p id="d1e2435">First, we generated a synthetic ToF-ACSM OA mass spectral dataset mimicking a polluted megacity environment affected by multiple OA sources. The synthetic datasets used here are based on observations from Beijing, as it is a relatively well-studied environment. In our case, the modeled sources are traffic exhaust (HOA); cooking (COA); biomass burning (BBOA); coal combustion (CCOA); and secondary OA (OOA). In addition, we also constructed more simple datasets generated with fewer factors (two factors: HOA <inline-formula><mml:math id="M106" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> OOA; three factors: HOA<inline-formula><mml:math id="M107" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>COA<inline-formula><mml:math id="M108" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>OOA; and four factors: HOA<inline-formula><mml:math id="M109" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>COA<inline-formula><mml:math id="M110" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>BBOA<inline-formula><mml:math id="M111" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>OOA). <inline-formula><mml:math id="M112" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> were chemical fingerprints from the literature <xref ref-type="bibr" rid="bib1.bibx14 bib1.bibx37 bib1.bibx36" id="paren.44"/>. <inline-formula><mml:math id="M113" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> was created as a mix of Gaussian and Cauchy (BBOA Gaussian, others Cauchy), biased, positive random walks, with added typical diurnal concentration cycles <xref ref-type="bibr" rid="bib1.bibx26" id="paren.45"/> corresponding to the matching OA sources (based on <inline-formula><mml:math id="M114" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>). A mix of different random walk distributions was chosen to test whether the model approximation of Cauchy auto-correlation works with them. The random walk aims to introduce variability such as one would get from varying transport and mixing. The diurnal cycle was simply summed to the random walk to produce the time series. The overall order of magnitude and diurnal concentration variability of the components (OA sources) were estimated based on previous literature on OA sources in Beijing <xref ref-type="bibr" rid="bib1.bibx26" id="paren.46"/>. For each number of factors, we constructed 10 different datasets amounting to a total of 40 datasets. For reference, one 5-component dataset in its component form is shown in Fig. <xref ref-type="fig" rid="Ch1.F2"/>. CCOA and BBOA are very similar in <inline-formula><mml:math id="M115" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M116" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> (see Appendix <xref ref-type="sec" rid="App1.Ch1.S2"/>), making it a challenging dataset. The time series of CCOA and BBOA are also similar in magnitude and have similar diurnal behavior. While the short-term auto-correlation is high, the random walks of the megacity dataset are not as highly auto-correlated as the PM data in Fig. <xref ref-type="fig" rid="Ch1.F1"/>. The added diurnal peaks can be seen as the periodic peaks in the correlograms in Fig. <xref ref-type="fig" rid="Ch1.F2"/>c and f. See Appendix <xref ref-type="sec" rid="App1.Ch1.S3"/> for an example of the <inline-formula><mml:math id="M117" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> dependence of the measurement errors on this dataset.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F2" specific-use="star"><?xmltex \currentcnt{2}?><?xmltex \def\figurename{Figure}?><label>Figure 2</label><caption><p id="d1e2551">Characteristics of the synthetic ToF-ACSM OA datasets. Panels <bold>(a)</bold> and <bold>(d)</bold> show the factor profiles (<inline-formula><mml:math id="M118" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>) used to construct the synthetic datasets. The solid black bars are ions derived from <inline-formula><mml:math id="M119" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> 44 and are only used for converting concentrations. Panels <bold>(b)</bold> and <bold>(e)</bold> show the factor time series (<inline-formula><mml:math id="M120" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>) used to construct each dataset, and the unit for them is <inline-formula><mml:math id="M121" display="inline"><mml:mrow class="unit"><mml:mi mathvariant="normal">µ</mml:mi><mml:mi mathvariant="normal">g</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:msup><mml:mi mathvariant="normal">m</mml:mi><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">3</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, and panels <bold>(c)</bold> and <bold>(f)</bold> show the temporal auto-correlation for each factor. Auto-correlation refers to the Pearson correlation coefficient of the component with the same component time-shifted by the number of hours. Panels <bold>(a)</bold>, <bold>(b)</bold>, and <bold>(c)</bold> are for megacity data and panels <bold>(d)</bold>, <bold>(e)</bold>, and <bold>(f)</bold> are for the European urban environment.</p></caption>
          <?xmltex \igopts{width=492.232677pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f02.png"/>

        </fig>

</sec>
<?pagebreak page1257?><sec id="Ch1.S3.SS2">
  <label>3.2</label><title>Synthetic dataset representing a typical European urban environment</title>
      <p id="d1e2651">As another test, we used chemical transport model data from <xref ref-type="bibr" rid="bib1.bibx24" id="text.47"/> representing approximately 2 weeks of simulated measurements in Zurich, Switzerland. Zurich represents a typical European city with low pollution levels. In this case, the <inline-formula><mml:math id="M122" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> time series come from the transport model and <inline-formula><mml:math id="M123" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> is taken from the literature (HOA and BBOA from an ambient analysis presented by <xref ref-type="bibr" rid="bib1.bibx14 bib1.bibx37 bib1.bibx36" id="altparen.48"/>, biological SOA (SOA<inline-formula><mml:math id="M124" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">bio</mml:mi></mml:msub></mml:math></inline-formula>) from an ambient analysis presented by <xref ref-type="bibr" rid="bib1.bibx12" id="altparen.49"/>, anthropogenic SOA, SOA<inline-formula><mml:math id="M125" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">anthro</mml:mi></mml:msub></mml:math></inline-formula>, is represented by laboratory Diesel generator SOA presented by <xref ref-type="bibr" rid="bib1.bibx34" id="altparen.50"/>).</p>
      <?pagebreak page1258?><p id="d1e2699">This dataset differs from those in Sect. <xref ref-type="sec" rid="Ch1.S3.SS1"/> in two important ways. Firstly, the concentrations are lower since the environment is less polluted, which affects the error estimation as larger relative measurement errors, median 1.0 % of data magnitude for this dataset and median 0.6 % for one of the datasets described in Sect. <xref ref-type="sec" rid="Ch1.S3.SS1"/>. Secondly, the sources exhibit high correlation in the time series (Appendix <xref ref-type="sec" rid="App1.Ch1.S2"/> shows that correlations in <inline-formula><mml:math id="M126" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> are higher than in the dataset described in Sect. <xref ref-type="sec" rid="Ch1.S3.SS1"/>), possibly indicating that meteorological conditions and transport of pollutants are important drivers of the concentration of the components. From a source apportionment analysis perspective, this simulates the worst-case situation with the data having poor separability in <inline-formula><mml:math id="M127" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>. The components of this dataset compared to the megacity data can be seen in Fig. <xref ref-type="fig" rid="Ch1.F2"/>. The auto-correlation behavior of the two datasets is very similar, indicating that our fully synthetic data behave as realistically as the transport model.</p>
</sec>
</sec>
<sec id="Ch1.S4">
  <label>4</label><title>Results and discussion</title>
      <p id="d1e2736">In this section we compare the factorizations from BAMF and PMF on simulated megacity data, in Sect. <xref ref-type="sec" rid="Ch1.S4.SS1"/>, and synthetic European data, in Sect. <xref ref-type="sec" rid="Ch1.S4.SS2"/>. We also investigate what happens when we do not know the true number of sources, in Sect. <xref ref-type="sec" rid="Ch1.S4.SS3"/>, and how additional prior information improves the factorization, in Sect. <xref ref-type="sec" rid="Ch1.S4.SS4"/>.</p>
<sec id="Ch1.S4.SS1">
  <label>4.1</label><title>Simulated megacity source apportionment</title>
      <p id="d1e2754">In the first experiment, we assess the performance of BAMF, BAMF-0, and PMF on synthetic data mimicking the conditions in a polluted megacity described in Sect. <xref ref-type="sec" rid="Ch1.S3.SS1"/>. First, we assess the reconstruction of the input by the different models. In addition to minimizing the residuals, BAMF also includes a penalty for deviations from auto-correlation in <inline-formula><mml:math id="M128" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>. Due to this and BAMF being a sampled model instead of an optimizer as described in Sect. <xref ref-type="sec" rid="Ch1.S2.SS3"/>, PMF would be expected to give answers with lower absolute measurement error-weighted residuals compared to BAMF. In other words, PMF is expected to have a better reconstruction performance.</p>
      <p id="d1e2768">Figure <xref ref-type="fig" rid="Ch1.F3"/> shows the median solution reconstruction across the 10 different datasets as the number of components increases. All models reconstruct the input data well within the error estimate. The similar reconstruction metrics for BAMF and BAMF-0 suggest that the inclusion of the auto-correlation term does not substantially deteriorate the reconstruction accuracy. On the other hand, PMF has, as expected, marginally lower absolute measurement error-weighted residuals than BAMF and BAMF-0. However, they are below unity and within the error estimate, judging by the normalized residuals. Therefore, PMF most likely finds solutions that fit the noise in the data better. All other reconstruction metrics (<inline-formula><mml:math id="M129" display="inline"><mml:mi mathvariant="bold-italic">S</mml:mi></mml:math></inline-formula> mean, <inline-formula><mml:math id="M130" display="inline"><mml:mi mathvariant="bold-italic">S</mml:mi></mml:math></inline-formula> standard deviation, <inline-formula><mml:math id="M131" display="inline"><mml:mrow><mml:msubsup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) are comparable for all models.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F3" specific-use="star"><?xmltex \currentcnt{3}?><?xmltex \def\figurename{Figure}?><label>Figure 3</label><caption><p id="d1e2802">Reconstruction metrics for BAMF, BAMF-0, and PMF for synthetic megacity data. Panel <bold>(a)</bold> shows the relative error of <inline-formula><mml:math id="M132" display="inline"><mml:mi mathvariant="bold">X</mml:mi></mml:math></inline-formula> of the median solution as a function of the number of components for the synthetic megacity data (BAMF and BAMF-0: results from 10 different datasets for each number of factors,  PMF: only for the 5-factor cases). Panel <bold>(b)</bold> shows the <inline-formula><mml:math id="M133" display="inline"><mml:mrow><mml:msubsup><mml:mi>Q</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mo>*</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> statistic, where all models are very similar. Panel <bold>(c)</bold> shows the mean and panel <bold>(d)</bold> shows the standard deviation of <inline-formula><mml:math id="M134" display="inline"><mml:mi mathvariant="bold-italic">S</mml:mi></mml:math></inline-formula>, as a function of the number of factors for the synthetic megacity data (the ideal value for the mean is 0 and the ideal value for standard deviation is smaller than 1). <inline-formula><mml:math id="M135" display="inline"><mml:mi mathvariant="bold">S</mml:mi></mml:math></inline-formula> refers to the difference between the original data and the samples normalized with the standard deviation of the samples (uncertainty of the model).</p></caption>
          <?xmltex \igopts{width=497.923228pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f03.png"/>

        </fig>

      <p id="d1e2859">Data reconstruction is essential to get within the error limits. However, source apportionment aims to accurately and precisely resolve the actual components in <inline-formula><mml:math id="M136" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M137" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, i.e., factorization performance. The solution of each model to the example dataset is shown in Fig. <xref ref-type="fig" rid="Ch1.F4"/>. For this example, we observe that all models resolve all five components. However, BAMF has a better factorization performance both in <inline-formula><mml:math id="M138" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M139" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> than the models not accounting for auto-correlation (BAMF-0, PMF) in Table <xref ref-type="table" rid="Ch1.T1"/>. Similarly, for the diurnal cycles in the example in Fig. <xref ref-type="fig" rid="Ch1.F4"/>c, the  diurnal concentration of OA sources, as identified by BAMF, closely resembles the ground truth components. While all models capture the time behavior of the diurnals, the absolute magnitude has a bias, with BAMF having a substantially smaller bias than the other models for four out of five components (Table <xref ref-type="table" rid="Ch1.T1"/>).</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F4" specific-use="star"><?xmltex \currentcnt{4}?><?xmltex \def\figurename{Figure}?><label>Figure 4</label><caption><p id="d1e2901">Illustration of <inline-formula><mml:math id="M140" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M141" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> reconstruction of all models for one of the synthetic megacity ToF-ACSM OA datasets with five components. Panel <bold>(a)</bold> is <inline-formula><mml:math id="M142" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, <bold>(b)</bold> is <inline-formula><mml:math id="M143" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, <bold>(c)</bold> is the diurnal concentration variation, and <bold>(d)</bold> is auto-correlation measured as Pearson correlation coefficient. Here, we display the median and the 0.25 and 0.75 quantiles. For all BAMF-type models, these quantities are computed based on the samples and for PMF, based on the 100 solutions.</p></caption>
          <?xmltex \igopts{width=426.791339pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f04.png"/>

        </fig>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T1" specific-use="star"><?xmltex \currentcnt{1}?><label>Table 1</label><caption><p id="d1e2954">Reconstruction and factorization performance of all three models for the synthetic megacity ToF-ACSM OA dataset in Fig. <xref ref-type="fig" rid="Ch1.F4"/>. The reconstruction metrics measure the residuals divided by the error estimate. A value closer to 0 is better and a value below 1 is smaller than the error estimate given to the model. The factorization performance is assessed via three metrics: <inline-formula><mml:math id="M144" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth is the average ratio of each factor time series, <inline-formula><mml:math id="M145" display="inline"><mml:mi>r</mml:mi></mml:math></inline-formula> is the Pearson correlation coefficient between the factor time series, and <inline-formula><mml:math id="M146" display="inline"><mml:mi mathvariant="italic">ρ</mml:mi></mml:math></inline-formula> is the Spearman correlation coefficient between the factor profiles. For the factorization performance, a value closer to 1 is better. Nonlinear correlation coefficient is used for factor profiles, since they have an additional constraint of summing to unity and thus linear correlation is penalized for small errors disproportionately. For each metric, the best value is highlighted in bold.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="5">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:colspec colnum="3" colname="col3" align="center"/>
     <oasis:colspec colnum="4" colname="col4" align="center"/>
     <oasis:colspec colnum="5" colname="col5" align="center"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">BAMF</oasis:entry>
         <oasis:entry colname="col4">BAMF-0</oasis:entry>
         <oasis:entry colname="col5">PMF</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">Reconstruction performance:</oasis:entry>
         <oasis:entry colname="col2">Median<inline-formula><mml:math id="M147" display="inline"><mml:mrow><mml:mo>(</mml:mo><mml:mo>|</mml:mo><mml:mi mathvariant="bold">Z</mml:mi><mml:mo>-</mml:mo><mml:mi mathvariant="bold">X</mml:mi><mml:mo>|</mml:mo><mml:mo>/</mml:mo><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">0.68</oasis:entry>
         <oasis:entry colname="col4">0.68</oasis:entry>
         <oasis:entry colname="col5"><bold>0.64</bold></oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Max<inline-formula><mml:math id="M148" display="inline"><mml:mrow><mml:mo>(</mml:mo><mml:mo>|</mml:mo><mml:mi mathvariant="bold">Z</mml:mi><mml:mo>-</mml:mo><mml:mi mathvariant="bold">X</mml:mi><mml:mo>|</mml:mo><mml:mo>/</mml:mo><mml:mi mathvariant="bold-italic">σ</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">5.85</oasis:entry>
         <oasis:entry colname="col4">5.07</oasis:entry>
         <oasis:entry colname="col5"><bold>3.72</bold></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Factorization performance:</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M149" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth OOA</oasis:entry>
         <oasis:entry colname="col3"><bold>0.94</bold></oasis:entry>
         <oasis:entry colname="col4">0.83</oasis:entry>
         <oasis:entry colname="col5">0.78</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M150" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula> OOA</oasis:entry>
         <oasis:entry colname="col3"><bold>1.00</bold></oasis:entry>
         <oasis:entry colname="col4"><bold>1.00</bold></oasis:entry>
         <oasis:entry colname="col5"><bold>1.00</bold></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M151" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula> OOA</oasis:entry>
         <oasis:entry colname="col3"><bold>0.99</bold></oasis:entry>
         <oasis:entry colname="col4">0.90</oasis:entry>
         <oasis:entry colname="col5">0.93</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2">Diurnal <inline-formula><mml:math id="M152" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth OOA</oasis:entry>
         <oasis:entry rowsep="1" colname="col3"><bold>0.94</bold></oasis:entry>
         <oasis:entry rowsep="1" colname="col4">0.83</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">0.78</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M153" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth HOA</oasis:entry>
         <oasis:entry colname="col3"><bold>0.99</bold></oasis:entry>
         <oasis:entry colname="col4">1.16</oasis:entry>
         <oasis:entry colname="col5">1.51</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M154" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula> HOA</oasis:entry>
         <oasis:entry colname="col3"><bold>0.99</bold></oasis:entry>
         <oasis:entry colname="col4">0.92</oasis:entry>
         <oasis:entry colname="col5">0.86</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M155" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula> HOA</oasis:entry>
         <oasis:entry colname="col3">0.99</oasis:entry>
         <oasis:entry colname="col4"><bold>1.00</bold></oasis:entry>
         <oasis:entry colname="col5">0.97</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2">Diurnal <inline-formula><mml:math id="M156" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth HOA</oasis:entry>
         <oasis:entry rowsep="1" colname="col3"><bold>1.01</bold></oasis:entry>
         <oasis:entry rowsep="1" colname="col4">1.21</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">1.57</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M157" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth COA</oasis:entry>
         <oasis:entry colname="col3"><bold>0.99</bold></oasis:entry>
         <oasis:entry colname="col4">1.44</oasis:entry>
         <oasis:entry colname="col5">1.39</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M158" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula> COA</oasis:entry>
         <oasis:entry colname="col3"><bold>0.98</bold></oasis:entry>
         <oasis:entry colname="col4">0.78</oasis:entry>
         <oasis:entry colname="col5">0.95</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M159" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula> COA</oasis:entry>
         <oasis:entry colname="col3">0.99</oasis:entry>
         <oasis:entry colname="col4"><bold>1.00</bold></oasis:entry>
         <oasis:entry colname="col5"><bold>1.00</bold></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2">Diurnal <inline-formula><mml:math id="M160" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth COA</oasis:entry>
         <oasis:entry rowsep="1" colname="col3"><bold>1.03</bold></oasis:entry>
         <oasis:entry rowsep="1" colname="col4">1.53</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">1.47</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M161" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth BBOA</oasis:entry>
         <oasis:entry colname="col3"><bold>1.08</bold></oasis:entry>
         <oasis:entry colname="col4">1.26</oasis:entry>
         <oasis:entry colname="col5">1.28</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M162" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.25em" linebreak="nobreak"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula> BBOA</oasis:entry>
         <oasis:entry colname="col3">0.98</oasis:entry>
         <oasis:entry colname="col4"><bold>1.00</bold></oasis:entry>
         <oasis:entry colname="col5">0.66</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M163" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula> BBOA</oasis:entry>
         <oasis:entry colname="col3">0.99</oasis:entry>
         <oasis:entry colname="col4"><bold>1.00</bold></oasis:entry>
         <oasis:entry colname="col5">0.94</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2">Diurnal <inline-formula><mml:math id="M164" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth BBOA</oasis:entry>
         <oasis:entry rowsep="1" colname="col3"><bold>1.10</bold></oasis:entry>
         <oasis:entry rowsep="1" colname="col4">1.26</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">1.37</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M165" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth CCOA</oasis:entry>
         <oasis:entry colname="col3">1.24</oasis:entry>
         <oasis:entry colname="col4">1.20</oasis:entry>
         <oasis:entry colname="col5"><bold>1.17</bold></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M166" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula> CCOA</oasis:entry>
         <oasis:entry colname="col3"><bold>0.99</bold></oasis:entry>
         <oasis:entry colname="col4">0.98</oasis:entry>
         <oasis:entry colname="col5">0.86</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M167" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace linebreak="nobreak" width="0.25em"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula> CCOA</oasis:entry>
         <oasis:entry colname="col3"><bold>1.00</bold></oasis:entry>
         <oasis:entry colname="col4">0.99</oasis:entry>
         <oasis:entry colname="col5">0.96</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Diurnal <inline-formula><mml:math id="M168" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> Truth CCOA</oasis:entry>
         <oasis:entry colname="col3">1.27</oasis:entry>
         <oasis:entry colname="col4"><bold>1.19</bold></oasis:entry>
         <oasis:entry colname="col5">1.27</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table><?xmltex \gdef\@currentlabel{1}?></table-wrap>

      <p id="d1e3677">All models slightly underestimate OOA, which results in overestimating the other components (Table <xref ref-type="table" rid="Ch1.T1"/>). For the final component, CCOA, PMF has less bias, but it correlates worse with the truth. BAMF-0 and PMF significantly underestimate OOA, which results in an overestimation of other components to get the reconstruction correct. Some factors have a more pronounced cyclical temporal behavior, as witnessed by high auto-correlation at higher temporal lag (e.g., Fig. <xref ref-type="fig" rid="Ch1.F4"/>c CCOA, BBOA, COA). Based on Table <xref ref-type="table" rid="Ch1.T1"/>, some of the factors with pronounced cyclical temporal behavior are better represented by BAMF than PMF (e.g., BBOA, COA), while this is not necessarily the case for others (e.g., CCOA).</p>

<?xmltex \floatpos{p}?><table-wrap id="Ch1.T2" specific-use="star" orientation="landscape"><?xmltex \currentcnt{2}?><label>Table 2</label><caption><p id="d1e3690">Reconstruction and factorization performance for the synthetic European city ToF-ACSM OA dataset. <inline-formula><mml:math id="M169" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> True is the average ratio of the component to the true one, <inline-formula><mml:math id="M170" display="inline"><mml:mi>r</mml:mi></mml:math></inline-formula> is the Pearson correlation coefficient, and <inline-formula><mml:math id="M171" display="inline"><mml:mi mathvariant="italic">ρ</mml:mi></mml:math></inline-formula> is the Spearman correlation coefficient. Diurn is a factor's average ratio of the diurnal of <inline-formula><mml:math id="M172" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> to the true diurnal. POA <inline-formula><mml:math id="M173" display="inline"><mml:mo>/</mml:mo></mml:math></inline-formula> SOA is the ratio of primary (HOA, BBOA) and secondary organic aerosol (anthropogenic SOA, biogenic SOA); for the ground truth, this ratio is 1.55. Unidentified components were not included in this ratio. For the reconstruction metrics, values closer to 0 are better, and values below 1 are within the error given to the model. For factorization metrics, the ideal value is 1.</p></caption><oasis:table frame="topbot"><?xmltex \begin{scaleboxenv}{.85}[.85]?><oasis:tgroup cols="22">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:colspec colnum="3" colname="col3" align="center"/>
     <oasis:colspec colnum="4" colname="col4" align="center"/>
     <oasis:colspec colnum="5" colname="col5" align="right" colsep="1"/>
     <oasis:colspec colnum="6" colname="col6" align="center"/>
     <oasis:colspec colnum="7" colname="col7" align="center"/>
     <oasis:colspec colnum="8" colname="col8" align="center"/>
     <oasis:colspec colnum="9" colname="col9" align="center" colsep="1"/>
     <oasis:colspec colnum="10" colname="col10" align="center"/>
     <oasis:colspec colnum="11" colname="col11" align="center"/>
     <oasis:colspec colnum="12" colname="col12" align="center"/>
     <oasis:colspec colnum="13" colname="col13" align="center" colsep="1"/>
     <oasis:colspec colnum="14" colname="col14" align="center"/>
     <oasis:colspec colnum="15" colname="col15" align="center"/>
     <oasis:colspec colnum="16" colname="col16" align="center"/>
     <oasis:colspec colnum="17" colname="col17" align="center" colsep="1"/>
     <oasis:colspec colnum="18" colname="col18" align="center"/>
     <oasis:colspec colnum="19" colname="col19" align="center"/>
     <oasis:colspec colnum="20" colname="col20" align="center"/>
     <oasis:colspec colnum="21" colname="col21" align="center" colsep="1"/>
     <oasis:colspec colnum="22" colname="col22" align="center"/>
     <oasis:thead>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3"/>
         <oasis:entry rowsep="1" namest="col4" nameend="col5" colsep="1"><inline-formula><mml:math id="M174" display="inline"><mml:mrow><mml:mo>|</mml:mo><mml:mi mathvariant="bold">X</mml:mi><mml:mo>-</mml:mo><mml:mi mathvariant="bold">Z</mml:mi><mml:mo>|</mml:mo><mml:mo>/</mml:mo><mml:mi mathvariant="bold-italic">σ</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry rowsep="1" namest="col6" nameend="col9" colsep="1">HOA </oasis:entry>
         <oasis:entry rowsep="1" namest="col10" nameend="col13" colsep="1">SOA anthro </oasis:entry>
         <oasis:entry rowsep="1" namest="col14" nameend="col17" colsep="1">SOA bio </oasis:entry>
         <oasis:entry rowsep="1" namest="col18" nameend="col21" colsep="1">BBOA </oasis:entry>
         <oasis:entry rowsep="1" colname="col22">POA <inline-formula><mml:math id="M175" display="inline"><mml:mo>/</mml:mo></mml:math></inline-formula> SOA</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3"/>
         <oasis:entry colname="col4">Median</oasis:entry>
         <oasis:entry colname="col5">Max</oasis:entry>
         <oasis:entry colname="col6"><inline-formula><mml:math id="M176" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> True</oasis:entry>
         <oasis:entry colname="col7"><inline-formula><mml:math id="M177" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col8"><inline-formula><mml:math id="M178" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col9">Diurn</oasis:entry>
         <oasis:entry colname="col10"><inline-formula><mml:math id="M179" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> True</oasis:entry>
         <oasis:entry colname="col11"><inline-formula><mml:math id="M180" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col12"><inline-formula><mml:math id="M181" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col13">Diurn</oasis:entry>
         <oasis:entry colname="col14"><inline-formula><mml:math id="M182" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> True</oasis:entry>
         <oasis:entry colname="col15"><inline-formula><mml:math id="M183" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col16"><inline-formula><mml:math id="M184" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col17">Diurn</oasis:entry>
         <oasis:entry colname="col18"><inline-formula><mml:math id="M185" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula> True</oasis:entry>
         <oasis:entry colname="col19"><inline-formula><mml:math id="M186" display="inline"><mml:mrow><mml:mi mathvariant="bold">G</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mi>r</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col20"><inline-formula><mml:math id="M187" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mi mathvariant="italic">ρ</mml:mi></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col21">Diurn</oasis:entry>
         <oasis:entry colname="col22"/>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">BAMF</oasis:entry>
         <oasis:entry colname="col2">Full</oasis:entry>
         <oasis:entry colname="col3">3</oasis:entry>
         <oasis:entry colname="col4">1.18</oasis:entry>
         <oasis:entry colname="col5">54.56</oasis:entry>
         <oasis:entry colname="col6">0.78</oasis:entry>
         <oasis:entry colname="col7">0.98</oasis:entry>
         <oasis:entry colname="col8">0.97</oasis:entry>
         <oasis:entry colname="col9">0.80</oasis:entry>
         <oasis:entry colname="col10">1.55</oasis:entry>
         <oasis:entry colname="col11">0.95</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">1.53</oasis:entry>
         <oasis:entry colname="col14"/>
         <oasis:entry colname="col15"/>
         <oasis:entry colname="col16"/>
         <oasis:entry colname="col17"/>
         <oasis:entry colname="col18">0.85</oasis:entry>
         <oasis:entry colname="col19">0.97</oasis:entry>
         <oasis:entry colname="col20">0.99</oasis:entry>
         <oasis:entry colname="col21">0.86</oasis:entry>
         <oasis:entry colname="col22">1.01</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">4</oasis:entry>
         <oasis:entry colname="col4">0.67</oasis:entry>
         <oasis:entry colname="col5">5.13</oasis:entry>
         <oasis:entry colname="col6">0.72</oasis:entry>
         <oasis:entry colname="col7">0.91</oasis:entry>
         <oasis:entry colname="col8">1.00</oasis:entry>
         <oasis:entry colname="col9">0.74</oasis:entry>
         <oasis:entry colname="col10">1.39</oasis:entry>
         <oasis:entry colname="col11">1.00</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">1.39</oasis:entry>
         <oasis:entry colname="col14">0.91</oasis:entry>
         <oasis:entry colname="col15">1.00</oasis:entry>
         <oasis:entry colname="col16">0.96</oasis:entry>
         <oasis:entry colname="col17">0.88</oasis:entry>
         <oasis:entry colname="col18">0.86</oasis:entry>
         <oasis:entry colname="col19">0.97</oasis:entry>
         <oasis:entry colname="col20">1.00</oasis:entry>
         <oasis:entry colname="col21">0.87</oasis:entry>
         <oasis:entry colname="col22">0.96</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2"/>
         <oasis:entry rowsep="1" colname="col3">5</oasis:entry>
         <oasis:entry rowsep="1" colname="col4">0.67</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">5.14</oasis:entry>
         <oasis:entry rowsep="1" colname="col6">0.04</oasis:entry>
         <oasis:entry rowsep="1" colname="col7">0.16</oasis:entry>
         <oasis:entry rowsep="1" colname="col8">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col9">0.05</oasis:entry>
         <oasis:entry rowsep="1" colname="col10">1.41</oasis:entry>
         <oasis:entry rowsep="1" colname="col11">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col12">0.99</oasis:entry>
         <oasis:entry rowsep="1" colname="col13">1.42</oasis:entry>
         <oasis:entry rowsep="1" colname="col14">0.96</oasis:entry>
         <oasis:entry rowsep="1" colname="col15">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col16">0.95</oasis:entry>
         <oasis:entry rowsep="1" colname="col17">0.93</oasis:entry>
         <oasis:entry rowsep="1" colname="col18">0.50</oasis:entry>
         <oasis:entry rowsep="1" colname="col19">0.77</oasis:entry>
         <oasis:entry rowsep="1" colname="col20">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col21">0.48</oasis:entry>
         <oasis:entry rowsep="1" colname="col22">0.38</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2">Incomplete</oasis:entry>
         <oasis:entry rowsep="1" colname="col3">4</oasis:entry>
         <oasis:entry rowsep="1" colname="col4">0.67</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">5.06</oasis:entry>
         <oasis:entry rowsep="1" colname="col6">0.71</oasis:entry>
         <oasis:entry rowsep="1" colname="col7">0.91</oasis:entry>
         <oasis:entry rowsep="1" colname="col8">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col9">0.73</oasis:entry>
         <oasis:entry rowsep="1" colname="col10">1.41</oasis:entry>
         <oasis:entry rowsep="1" colname="col11">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col12">0.98</oasis:entry>
         <oasis:entry rowsep="1" colname="col13">1.41</oasis:entry>
         <oasis:entry rowsep="1" colname="col14">0.90</oasis:entry>
         <oasis:entry rowsep="1" colname="col15">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col16">0.96</oasis:entry>
         <oasis:entry rowsep="1" colname="col17">0.87</oasis:entry>
         <oasis:entry rowsep="1" colname="col18">0.85</oasis:entry>
         <oasis:entry rowsep="1" colname="col19">0.97</oasis:entry>
         <oasis:entry rowsep="1" colname="col20">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col21">0.87</oasis:entry>
         <oasis:entry rowsep="1" colname="col22">0.95</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Partial</oasis:entry>
         <oasis:entry colname="col3">3</oasis:entry>
         <oasis:entry colname="col4">1.18</oasis:entry>
         <oasis:entry colname="col5">53.08</oasis:entry>
         <oasis:entry colname="col6">0.96</oasis:entry>
         <oasis:entry colname="col7">0.90</oasis:entry>
         <oasis:entry colname="col8">0.96</oasis:entry>
         <oasis:entry colname="col9">1.00</oasis:entry>
         <oasis:entry colname="col10">1.10</oasis:entry>
         <oasis:entry colname="col11">0.93</oasis:entry>
         <oasis:entry colname="col12">0.98</oasis:entry>
         <oasis:entry colname="col13">1.11</oasis:entry>
         <oasis:entry colname="col14"/>
         <oasis:entry colname="col15"/>
         <oasis:entry colname="col16"/>
         <oasis:entry colname="col17"/>
         <oasis:entry colname="col18">0.96</oasis:entry>
         <oasis:entry colname="col19">0.91</oasis:entry>
         <oasis:entry colname="col20">0.97</oasis:entry>
         <oasis:entry colname="col21">0.95</oasis:entry>
         <oasis:entry colname="col22">1.66</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">4</oasis:entry>
         <oasis:entry colname="col4">0.67</oasis:entry>
         <oasis:entry colname="col5">4.98</oasis:entry>
         <oasis:entry colname="col6">0.73</oasis:entry>
         <oasis:entry colname="col7">0.96</oasis:entry>
         <oasis:entry colname="col8">1.00</oasis:entry>
         <oasis:entry colname="col9">0.74</oasis:entry>
         <oasis:entry colname="col10">1.32</oasis:entry>
         <oasis:entry colname="col11">0.99</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">1.33</oasis:entry>
         <oasis:entry colname="col14">1.59</oasis:entry>
         <oasis:entry colname="col15">0.91</oasis:entry>
         <oasis:entry colname="col16">0.98</oasis:entry>
         <oasis:entry colname="col17">1.96</oasis:entry>
         <oasis:entry colname="col18">0.80</oasis:entry>
         <oasis:entry colname="col19">0.97</oasis:entry>
         <oasis:entry colname="col20">1.00</oasis:entry>
         <oasis:entry colname="col21">0.75</oasis:entry>
         <oasis:entry colname="col22">0.88</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2"/>
         <oasis:entry rowsep="1" colname="col3">5</oasis:entry>
         <oasis:entry rowsep="1" colname="col4">0.67</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">5.03</oasis:entry>
         <oasis:entry rowsep="1" colname="col6">0.28</oasis:entry>
         <oasis:entry rowsep="1" colname="col7">0.93</oasis:entry>
         <oasis:entry rowsep="1" colname="col8">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col9">0.30</oasis:entry>
         <oasis:entry rowsep="1" colname="col10">1.40</oasis:entry>
         <oasis:entry rowsep="1" colname="col11">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col12">0.99</oasis:entry>
         <oasis:entry rowsep="1" colname="col13">1.40</oasis:entry>
         <oasis:entry rowsep="1" colname="col14">0.90</oasis:entry>
         <oasis:entry rowsep="1" colname="col15">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col16">0.96</oasis:entry>
         <oasis:entry rowsep="1" colname="col17">0.91</oasis:entry>
         <oasis:entry rowsep="1" colname="col18">0.10</oasis:entry>
         <oasis:entry rowsep="1" colname="col19">0.17</oasis:entry>
         <oasis:entry rowsep="1" colname="col20">0.99</oasis:entry>
         <oasis:entry rowsep="1" colname="col21">0.12</oasis:entry>
         <oasis:entry rowsep="1" colname="col22">0.21</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">None</oasis:entry>
         <oasis:entry colname="col3">3</oasis:entry>
         <oasis:entry colname="col4">1.18</oasis:entry>
         <oasis:entry colname="col5">52.80</oasis:entry>
         <oasis:entry colname="col6">1.91</oasis:entry>
         <oasis:entry colname="col7">0.91</oasis:entry>
         <oasis:entry colname="col8">0.95</oasis:entry>
         <oasis:entry colname="col9">2.00</oasis:entry>
         <oasis:entry colname="col10">1.21</oasis:entry>
         <oasis:entry colname="col11">0.96</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">1.22</oasis:entry>
         <oasis:entry colname="col14">2.24</oasis:entry>
         <oasis:entry colname="col15">0.35</oasis:entry>
         <oasis:entry colname="col16">0.88</oasis:entry>
         <oasis:entry colname="col17">3.12</oasis:entry>
         <oasis:entry colname="col18"/>
         <oasis:entry colname="col19"/>
         <oasis:entry colname="col20"/>
         <oasis:entry colname="col21"/>
         <oasis:entry colname="col22">0.83</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">4</oasis:entry>
         <oasis:entry colname="col4">0.67</oasis:entry>
         <oasis:entry colname="col5">4.96</oasis:entry>
         <oasis:entry colname="col6">1.70</oasis:entry>
         <oasis:entry colname="col7">0.91</oasis:entry>
         <oasis:entry colname="col8">0.95</oasis:entry>
         <oasis:entry colname="col9">1.77</oasis:entry>
         <oasis:entry colname="col10">1.39</oasis:entry>
         <oasis:entry colname="col11">1.00</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">1.39</oasis:entry>
         <oasis:entry colname="col14">0.91</oasis:entry>
         <oasis:entry colname="col15">1.00</oasis:entry>
         <oasis:entry colname="col16">0.96</oasis:entry>
         <oasis:entry colname="col17">0.89</oasis:entry>
         <oasis:entry colname="col18">0.24</oasis:entry>
         <oasis:entry colname="col19">0.74</oasis:entry>
         <oasis:entry colname="col20">0.95</oasis:entry>
         <oasis:entry colname="col21">0.24</oasis:entry>
         <oasis:entry colname="col22">0.97</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">5</oasis:entry>
         <oasis:entry colname="col4">0.67</oasis:entry>
         <oasis:entry colname="col5">5.07</oasis:entry>
         <oasis:entry colname="col6">1.70</oasis:entry>
         <oasis:entry colname="col7">0.91</oasis:entry>
         <oasis:entry colname="col8">0.95</oasis:entry>
         <oasis:entry colname="col9">1.77</oasis:entry>
         <oasis:entry colname="col10">1.36</oasis:entry>
         <oasis:entry colname="col11">1.00</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">1.36</oasis:entry>
         <oasis:entry colname="col14">0.92</oasis:entry>
         <oasis:entry colname="col15">1.00</oasis:entry>
         <oasis:entry colname="col16">0.96</oasis:entry>
         <oasis:entry colname="col17">0.91</oasis:entry>
         <oasis:entry colname="col18">0.24</oasis:entry>
         <oasis:entry colname="col19">0.75</oasis:entry>
         <oasis:entry colname="col20">0.96</oasis:entry>
         <oasis:entry colname="col21">0.23</oasis:entry>
         <oasis:entry colname="col22">0.99</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">BAMF-0</oasis:entry>
         <oasis:entry colname="col2">None</oasis:entry>
         <oasis:entry colname="col3">3</oasis:entry>
         <oasis:entry colname="col4">1.18</oasis:entry>
         <oasis:entry colname="col5">52.60</oasis:entry>
         <oasis:entry colname="col6">1.48</oasis:entry>
         <oasis:entry colname="col7">0.96</oasis:entry>
         <oasis:entry colname="col8">0.95</oasis:entry>
         <oasis:entry colname="col9">1.48</oasis:entry>
         <oasis:entry colname="col10">1.01</oasis:entry>
         <oasis:entry colname="col11">0.93</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">1.01</oasis:entry>
         <oasis:entry colname="col14"/>
         <oasis:entry colname="col15"/>
         <oasis:entry colname="col16"/>
         <oasis:entry colname="col17"/>
         <oasis:entry colname="col18">0.85</oasis:entry>
         <oasis:entry colname="col19">0.78</oasis:entry>
         <oasis:entry colname="col20">0.98</oasis:entry>
         <oasis:entry colname="col21">0.86</oasis:entry>
         <oasis:entry colname="col22">2.07</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">4</oasis:entry>
         <oasis:entry colname="col4">0.67</oasis:entry>
         <oasis:entry colname="col5">5.17</oasis:entry>
         <oasis:entry colname="col6">1.23</oasis:entry>
         <oasis:entry colname="col7">0.97</oasis:entry>
         <oasis:entry colname="col8">0.95</oasis:entry>
         <oasis:entry colname="col9">1.22</oasis:entry>
         <oasis:entry colname="col10">0.82</oasis:entry>
         <oasis:entry colname="col11">0.99</oasis:entry>
         <oasis:entry colname="col12">1.00</oasis:entry>
         <oasis:entry colname="col13">0.83</oasis:entry>
         <oasis:entry colname="col14">2.24</oasis:entry>
         <oasis:entry colname="col15">0.98</oasis:entry>
         <oasis:entry colname="col16">0.91</oasis:entry>
         <oasis:entry colname="col17">2.25</oasis:entry>
         <oasis:entry colname="col18">0.77</oasis:entry>
         <oasis:entry colname="col19">0.74</oasis:entry>
         <oasis:entry colname="col20">0.98</oasis:entry>
         <oasis:entry colname="col21">0.72</oasis:entry>
         <oasis:entry colname="col22">1.37</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">5</oasis:entry>
         <oasis:entry colname="col4">0.67</oasis:entry>
         <oasis:entry colname="col5">5.24</oasis:entry>
         <oasis:entry colname="col6">0.96</oasis:entry>
         <oasis:entry colname="col7">0.96</oasis:entry>
         <oasis:entry colname="col8">0.96</oasis:entry>
         <oasis:entry colname="col9">0.97</oasis:entry>
         <oasis:entry colname="col10">0.76</oasis:entry>
         <oasis:entry colname="col11">0.98</oasis:entry>
         <oasis:entry colname="col12">1.00</oasis:entry>
         <oasis:entry colname="col13">0.78</oasis:entry>
         <oasis:entry colname="col14">1.95</oasis:entry>
         <oasis:entry colname="col15">1.00</oasis:entry>
         <oasis:entry colname="col16">0.92</oasis:entry>
         <oasis:entry colname="col17">1.90</oasis:entry>
         <oasis:entry colname="col18">0.53</oasis:entry>
         <oasis:entry colname="col19">0.70</oasis:entry>
         <oasis:entry colname="col20">0.97</oasis:entry>
         <oasis:entry colname="col21">0.44</oasis:entry>
         <oasis:entry colname="col22">1.11</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">PMF</oasis:entry>
         <oasis:entry colname="col2">Full</oasis:entry>
         <oasis:entry colname="col3">3</oasis:entry>
         <oasis:entry colname="col4">1.10</oasis:entry>
         <oasis:entry colname="col5">82.93</oasis:entry>
         <oasis:entry colname="col6">0.82</oasis:entry>
         <oasis:entry colname="col7">0.97</oasis:entry>
         <oasis:entry colname="col8">1.00</oasis:entry>
         <oasis:entry colname="col9">0.85</oasis:entry>
         <oasis:entry colname="col10">1.79</oasis:entry>
         <oasis:entry colname="col11">0.97</oasis:entry>
         <oasis:entry colname="col12">0.98</oasis:entry>
         <oasis:entry colname="col13">1.78</oasis:entry>
         <oasis:entry colname="col14"/>
         <oasis:entry colname="col15"/>
         <oasis:entry colname="col16"/>
         <oasis:entry colname="col17"/>
         <oasis:entry colname="col18">0.62</oasis:entry>
         <oasis:entry colname="col19">0.80</oasis:entry>
         <oasis:entry colname="col20">1.00</oasis:entry>
         <oasis:entry colname="col21">0.65</oasis:entry>
         <oasis:entry colname="col22">0.74</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">4</oasis:entry>
         <oasis:entry colname="col4">0.63</oasis:entry>
         <oasis:entry colname="col5">4.01</oasis:entry>
         <oasis:entry colname="col6">0.73</oasis:entry>
         <oasis:entry colname="col7">0.97</oasis:entry>
         <oasis:entry colname="col8">1.00</oasis:entry>
         <oasis:entry colname="col9">0.74</oasis:entry>
         <oasis:entry colname="col10">1.12</oasis:entry>
         <oasis:entry colname="col11">1.00</oasis:entry>
         <oasis:entry colname="col12">1.00</oasis:entry>
         <oasis:entry colname="col13">1.13</oasis:entry>
         <oasis:entry colname="col14">2.78</oasis:entry>
         <oasis:entry colname="col15">0.93</oasis:entry>
         <oasis:entry colname="col16">0.92</oasis:entry>
         <oasis:entry colname="col17">3.15</oasis:entry>
         <oasis:entry colname="col18">0.71</oasis:entry>
         <oasis:entry colname="col19">0.90</oasis:entry>
         <oasis:entry colname="col20">1.00</oasis:entry>
         <oasis:entry colname="col21">0.70</oasis:entry>
         <oasis:entry colname="col22">0.78</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" colname="col2"/>
         <oasis:entry rowsep="1" colname="col3">5</oasis:entry>
         <oasis:entry rowsep="1" colname="col4">0.62</oasis:entry>
         <oasis:entry rowsep="1" colname="col5">4.29</oasis:entry>
         <oasis:entry rowsep="1" colname="col6">0.70</oasis:entry>
         <oasis:entry rowsep="1" colname="col7">0.97</oasis:entry>
         <oasis:entry rowsep="1" colname="col8">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col9">0.71</oasis:entry>
         <oasis:entry rowsep="1" colname="col10">0.57</oasis:entry>
         <oasis:entry rowsep="1" colname="col11">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col12">0.99</oasis:entry>
         <oasis:entry rowsep="1" colname="col13">0.58</oasis:entry>
         <oasis:entry rowsep="1" colname="col14">1.79</oasis:entry>
         <oasis:entry rowsep="1" colname="col15">0.83</oasis:entry>
         <oasis:entry rowsep="1" colname="col16">0.96</oasis:entry>
         <oasis:entry rowsep="1" colname="col17">2.10</oasis:entry>
         <oasis:entry rowsep="1" colname="col18">0.54</oasis:entry>
         <oasis:entry rowsep="1" colname="col19">0.88</oasis:entry>
         <oasis:entry rowsep="1" colname="col20">1.00</oasis:entry>
         <oasis:entry rowsep="1" colname="col21">0.52</oasis:entry>
         <oasis:entry rowsep="1" colname="col22">1.18</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">None</oasis:entry>
         <oasis:entry colname="col3">3</oasis:entry>
         <oasis:entry colname="col4">1.12</oasis:entry>
         <oasis:entry colname="col5">84.37</oasis:entry>
         <oasis:entry colname="col6">1.05</oasis:entry>
         <oasis:entry colname="col7">0.93</oasis:entry>
         <oasis:entry colname="col8">0.89</oasis:entry>
         <oasis:entry colname="col9">1.06</oasis:entry>
         <oasis:entry colname="col10">0.78</oasis:entry>
         <oasis:entry colname="col11">0.98</oasis:entry>
         <oasis:entry colname="col12">0.99</oasis:entry>
         <oasis:entry colname="col13">0.80</oasis:entry>
         <oasis:entry colname="col14"/>
         <oasis:entry colname="col15"/>
         <oasis:entry colname="col16"/>
         <oasis:entry colname="col17"/>
         <oasis:entry colname="col18">1.28</oasis:entry>
         <oasis:entry colname="col19">0.94</oasis:entry>
         <oasis:entry colname="col20">0.97</oasis:entry>
         <oasis:entry colname="col21">1.25</oasis:entry>
         <oasis:entry colname="col22">2.88</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">4</oasis:entry>
         <oasis:entry colname="col4">0.63</oasis:entry>
         <oasis:entry colname="col5">4.01</oasis:entry>
         <oasis:entry colname="col6">0.90</oasis:entry>
         <oasis:entry colname="col7">0.92</oasis:entry>
         <oasis:entry colname="col8">0.95</oasis:entry>
         <oasis:entry colname="col9">0.88</oasis:entry>
         <oasis:entry colname="col10">0.89</oasis:entry>
         <oasis:entry colname="col11">0.98</oasis:entry>
         <oasis:entry colname="col12">0.95</oasis:entry>
         <oasis:entry colname="col13">0.93</oasis:entry>
         <oasis:entry colname="col14">3.29</oasis:entry>
         <oasis:entry colname="col15">0.33</oasis:entry>
         <oasis:entry colname="col16">0.93</oasis:entry>
         <oasis:entry colname="col17">4.33</oasis:entry>
         <oasis:entry colname="col18">0.65</oasis:entry>
         <oasis:entry colname="col19">0.93</oasis:entry>
         <oasis:entry colname="col20">0.97</oasis:entry>
         <oasis:entry colname="col21">0.66</oasis:entry>
         <oasis:entry colname="col22">0.87</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">5</oasis:entry>
         <oasis:entry colname="col4">0.62</oasis:entry>
         <oasis:entry colname="col5">4.47</oasis:entry>
         <oasis:entry colname="col6">0.62</oasis:entry>
         <oasis:entry colname="col7">0.84</oasis:entry>
         <oasis:entry colname="col8">0.95</oasis:entry>
         <oasis:entry colname="col9">0.61</oasis:entry>
         <oasis:entry colname="col10">0.82</oasis:entry>
         <oasis:entry colname="col11">0.98</oasis:entry>
         <oasis:entry colname="col12">0.98</oasis:entry>
         <oasis:entry colname="col13">0.85</oasis:entry>
         <oasis:entry colname="col14">2.81</oasis:entry>
         <oasis:entry colname="col15">0.63</oasis:entry>
         <oasis:entry colname="col16">0.94</oasis:entry>
         <oasis:entry colname="col17">3.76</oasis:entry>
         <oasis:entry colname="col18">0.21</oasis:entry>
         <oasis:entry colname="col19">0.68</oasis:entry>
         <oasis:entry colname="col20">0.88</oasis:entry>
         <oasis:entry colname="col21">0.20</oasis:entry>
         <oasis:entry colname="col22">0.48</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup><?xmltex \end{scaleboxenv}?></oasis:table><?xmltex \gdef\@currentlabel{2}?></table-wrap>

      <p id="d1e5343">When considering all 10 synthetic datasets with five components mimicking a polluted megacity, BAMF consistently produces factors closer in magnitude to the truth and which correlate better with the actual factors than the other models (Fig. <xref ref-type="fig" rid="Ch1.F5"/>). BAMF is also better correlated with the time behavior of the components, while one of the components (CCOA, strongly correlated with BBOA in terms of <inline-formula><mml:math id="M188" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M189" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>) is difficult for all models. We hypothesize BAMF-0 shows better spread in Fig. <xref ref-type="fig" rid="Ch1.F5"/>, due to being constrained by the underestimation of OOA and having to include those peaks in the spectra. Figure <xref ref-type="fig" rid="Ch1.F5"/>a shows that BAMF can over- and underestimate both BBOA and CCOA depending on the dataset. Appendix <xref ref-type="sec" rid="App1.Ch1.S1"/> shows the inverse relationship between CCOA and BBOA mass concentration biases and how the profile reconstruction affects the CCOA mass concentration bias for the BAMF model. Overall, using auto-correlation in source apportionment markedly improves the quality of the resolved factors while keeping the overall reconstruction metrics similar.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F5" specific-use="star"><?xmltex \currentcnt{5}?><?xmltex \def\figurename{Figure}?><label>Figure 5</label><caption><p id="d1e5371">Summary of factorization performance of the three models for all synthetic megacity ToF-ACSM OA datasets with five components (10 datasets): Panel <bold>(a)</bold> shows the mean of the median value of the components of <inline-formula><mml:math id="M190" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> divided by the true value (ideal value is 1). Panel <bold>(b)</bold> shows the Spearman correlation coefficient median solution components of <inline-formula><mml:math id="M191" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> compared to the true value (a value of 1 refers to a perfect correlation). Panel <bold>(c)</bold> shows the Pearson correlation coefficient of the median solution components of <inline-formula><mml:math id="M192" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> compared to the true value (a value of 1 refers to a perfect linear correlation).</p></caption>
          <?xmltex \igopts{width=497.923228pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f05.png"/>

        </fig>

</sec>
<sec id="Ch1.S4.SS2">
  <label>4.2</label><title>Simulated European low-pollution city source apportionment</title>
      <p id="d1e5419">In a second exercise, we assessed the performance of BAMF, BAMF-0, and PMF on a synthetic dataset mimicking the conditions in a typical European city (Sect. <xref ref-type="sec" rid="Ch1.S3.SS2"/>). In contrast to the fully synthetic dataset in Sect. <xref ref-type="sec" rid="Ch1.S3.SS1"/>, here, the true components <inline-formula><mml:math id="M193" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> are OA source components computed by an air quality model <xref ref-type="bibr" rid="bib1.bibx24" id="paren.51"/>. This provides <inline-formula><mml:math id="M194" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> time series close to the atmosphere while still knowing the ground truth. While the three models show somewhat different components (Fig. <xref ref-type="fig" rid="Ch1.F6"/>), the reconstruction metrics indicate that all models have acceptable solutions (Table <xref ref-type="table" rid="Ch1.T2"/>). In fact, the metrics also show that the European dataset is reconstructed almost within the error limits with even only three components, i.e., one component less than is present in the synthetic dataset (HOA, BBOA, SOA<inline-formula><mml:math id="M195" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">anthro</mml:mi></mml:msub></mml:math></inline-formula>, SOA<inline-formula><mml:math id="M196" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">bio</mml:mi></mml:msub></mml:math></inline-formula>). This could explain why there is significant freedom in acceptable four-component solutions and variation between them.</p>

      <?xmltex \floatpos{p}?><fig id="Ch1.F6" specific-use="star"><?xmltex \currentcnt{6}?><?xmltex \def\figurename{Figure}?><label>Figure 6</label><caption><p id="d1e5468">Factorization performance of all three models for the synthetic European city ToF-ACSM OA dataset. The shaded area is the interquartile range (0.25–0.75 quantile). Panel <bold>(a)</bold> is <inline-formula><mml:math id="M197" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, <bold>(b)</bold> is <inline-formula><mml:math id="M198" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, <bold>(c)</bold> is the diurnal, and <bold>(d)</bold> is auto-correlation measured as Pearson correlation coefficient.</p></caption>
          <?xmltex \igopts{width=426.791339pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f06.png"/>

        </fig>

      <?pagebreak page1259?><p id="d1e5504">All models show signs of mixing between the components, most likely due to the correlation of the true <inline-formula><mml:math id="M199" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> components (time behavior is very similar) as well as similar chemical signatures <inline-formula><mml:math id="M200" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>. PMF mixes the SOA components while BAMF mixes the POA components. However, it is worth noting that BAMF has a significant bias on several components in <inline-formula><mml:math id="M201" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> as seen in Table <xref ref-type="table" rid="Ch1.T2"/>, but otherwise reflects their time behavior well. Given the recovered POA <inline-formula><mml:math id="M202" display="inline"><mml:mo>/</mml:mo></mml:math></inline-formula> SOA ratio, BAMF most likely mixes HOA and BBOA explaining that HOA is overestimated while BBOA is underestimated. PMF, on the other hand, produces two almost identical components (both in <inline-formula><mml:math id="M203" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M204" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>) for two of the four components, and PMF cannot thus distinguish the components present in the dataset. In Fig. <xref ref-type="fig" rid="Ch1.F6"/>d BAMF-0 and BAMF also seem to overestimate higher lag auto-correlation of BBOA in a similar fashion but have a higher bias. It should be noted that even BAMF only considers lag-1 auto-correlation in the model. For HOA, BAMF and PMF do not match the anti-correlated part between lags 5 and 20 and PMF cannot match the behavior of SOA bio. Overall, all models are challenged by the European dataset, with BAMF having the most consistent performance.</p>
</sec>
<sec id="Ch1.S4.SS3">
  <label>4.3</label><title>Resolving an unknown number of sources</title>
      <p id="d1e5562">For real-world source apportionment analyses, the true number of components, i.e., sources, to be resolved via matrix factorization is unknown yet crucial. Despite the importance, accurately determining and specifying the correct number of modeled components is not trivial (see, e.g., <xref ref-type="bibr" rid="bib1.bibx23 bib1.bibx37 bib1.bibx42" id="altparen.52"/>). Typical strategies rely on reconstructing <inline-formula><mml:math id="M205" display="inline"><mml:mi mathvariant="bold">X</mml:mi></mml:math></inline-formula> within the measurement error, the absence of structure in the measurement error-weighted residuals and the environmental interpretability of the resolved components. Here, we assess the behavior of the  BAMF, BAMF-0, and PMF models as the number of components are changed on the four-component chemical transport model dataset from Sect. <xref ref-type="sec" rid="Ch1.S3.SS2"/>. The model runs were performed both with an underspecified setting using three components and overspecified setting using five components (Figs. <xref ref-type="fig" rid="Ch1.F7"/> and <xref ref-type="fig" rid="Ch1.F8"/>). While underspecified, all models extract an SOA<inline-formula><mml:math id="M206" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">anthro</mml:mi></mml:msub></mml:math></inline-formula> component and merge the remaining three components into two. At the same time, BAMF extracts a component similar to SOA<inline-formula><mml:math id="M207" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">bio</mml:mi></mml:msub></mml:math></inline-formula>, while BAMF-0 and PMF extract components similar to HOA and BBOA but lack SOA<inline-formula><mml:math id="M208" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">bio</mml:mi></mml:msub></mml:math></inline-formula>.</p>

      <?xmltex \floatpos{p}?><fig id="Ch1.F7" specific-use="star"><?xmltex \currentcnt{7}?><?xmltex \def\figurename{Figure}?><label>Figure 7</label><caption><p id="d1e5611">Factorization performance of underspecified (three components) models for the synthetic European city ToF-ACSM OA dataset. Panel <bold>(a)</bold> is <inline-formula><mml:math id="M209" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, <bold>(b)</bold> is <inline-formula><mml:math id="M210" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, <bold>(c)</bold> is the diurnal, and <bold>(d)</bold> is auto-correlation measured as Pearson correlation coefficient. The models extract different components and they are shown next to the closest original component. This is why there are 4 components shown even though the models extract only 3 components each.</p></caption>
          <?xmltex \igopts{width=398.338583pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f07.png"/>

        </fig>

      <?xmltex \floatpos{p}?><fig id="Ch1.F8" specific-use="star"><?xmltex \currentcnt{8}?><?xmltex \def\figurename{Figure}?><label>Figure 8</label><caption><p id="d1e5649">Factorization performance of overspecified (five components) models for the synthetic European city ToF-ACSM OA dataset. Panel <bold>(a)</bold> is <inline-formula><mml:math id="M211" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, <bold>(b)</bold> is <inline-formula><mml:math id="M212" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, <bold>(c)</bold> is the diurnal, and <bold>(d)</bold> is auto-correlation measured as Pearson correlation coefficient. “?0” denotes an unidentified component.</p></caption>
          <?xmltex \igopts{width=398.338583pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f08.png"/>

        </fig>

      <p id="d1e5686">For the overspecified models (five instead of four components), the results differ (Fig. <xref ref-type="fig" rid="Ch1.F8"/>). While the models without the auto-correlation assumption (PMF, BAMF-0) split the true components into multiple sub-components (mostly the POA components, HOA and BBOA), BAMF produces an extra component that is easily identifiable as unnecessary in addition to the four components resolved with four<?pagebreak page1260?> factors. This unnecessary component is characterized by an extremely high auto-correlation and low magnitude, as seen in Fig. <xref ref-type="fig" rid="Ch1.F8"/>b, c, and d. For this component the time series is almost constant, and the composition is flat with large uncertainties. The extra component does not affect the factorization performance of the other components of BAMF substantially. At the same time, BAMF-0 and PMF have a reduced factorization performance with too many components (Fig. <xref ref-type="fig" rid="Ch1.F8"/>, Table <xref ref-type="table" rid="Ch1.T2"/>). In general use, one would prefer the model to indicate the limits of the factorization as BAMF does, instead of producing duplicate components. It should be noted that the behavior of BAMF can be changed by adding new model terms, such as constraints on <inline-formula><mml:math id="M213" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>. For example, the model minimizes the constrained component when it is run equally overspecified (five instead of four components) but with a priori information on <inline-formula><mml:math id="M214" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> as can be seen in Appendix <xref ref-type="sec" rid="App1.Ch1.S4"/>.</p>
</sec>
<?pagebreak page1261?><sec id="Ch1.S4.SS4">
  <label>4.4</label><title>Using ancillary information to improve resolving sources</title>
      <p id="d1e5723">As highlighted in Sect. <xref ref-type="sec" rid="Ch1.S4.SS2"/>, all models are challenged by the European dataset. Imperfect matrix factorization results are likewise often observed when using PMF for real-world chemical datasets (see, e.g., <xref ref-type="bibr" rid="bib1.bibx4 bib1.bibx12" id="altparen.53"/>). Often, information is available that could help resolve the sources, such as chemical fingerprints of specific components associated with different sources. In current practice, previously observed <inline-formula><mml:math id="M215" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> profiles are often used as boundary conditions in source apportionment analyses. This approach has significant uncertainty in the general case because true <inline-formula><mml:math id="M216" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> is unknown. Still, in our specific test case, the a priori information is precisely correct – the known information on <inline-formula><mml:math id="M217" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> of the true components.</p>
      <p id="d1e5752">We tested the performance of the models when using a priori information on <inline-formula><mml:math id="M218" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> using three different approaches on the European dataset (from Sect. <xref ref-type="sec" rid="Ch1.S3.SS2"/>): <list list-type="order"><list-item>
      <p id="d1e5766"><italic>Full constraint.</italic> For BAMF-C and PMF, the two POA components (HOA and BBOA) were fully (for all <inline-formula><mml:math id="M219" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula>s) constrained with a roughly 18 % deviation allowed from the anchor (see Sect. <xref ref-type="sec" rid="Ch1.S2.SS2.SSS2"/> and <xref ref-type="sec" rid="Ch1.S2.SS6"/> for the determination of this).</p></list-item><list-item>
      <p id="d1e5788"><italic>Incomplete constraint.</italic> Knowledge on the entire factor profile is not always available, e.g., not the same <inline-formula><mml:math id="M220" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> range is measured. With the incomplete constraint approach, we test whether a priori information on parts of the factor profile improves the factorization performance and thus whether such information is useful. For the BAMF model, a priori information was only used in a limited arbitrarily chosen <inline-formula><mml:math id="M221" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> range (12–60) instead of for all <inline-formula><mml:math id="M222" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula>s (<inline-formula><mml:math id="M223" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> 12–100), the same deviation allowed from the anchor for the constrained components.</p></list-item><list-item>
      <?pagebreak page1263?><p id="d1e5842"><italic>Partial constraint.</italic> Sometimes, only very little chemical information is available for a specific factor, and it is defined by few key tracers. With the partial constraint, we test whether a priori information on just a few peaks/variables improves the factorization performance. For the BAMF model, a priori information was only used for four arbitrarily chosen <inline-formula><mml:math id="M224" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> peaks out of 74 (<inline-formula><mml:math id="M225" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula>s 45, 57, and 60, <inline-formula><mml:math id="M226" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mo>[</mml:mo><mml:mi>i</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>, compared to <inline-formula><mml:math id="M227" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> 43, <inline-formula><mml:math id="M228" display="inline"><mml:mrow><mml:mi mathvariant="bold">F</mml:mi><mml:mo>[</mml:mo><mml:mi>k</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>) for HOA and BBOA defined with the same deviation allowed from the anchor for constrained components.</p></list-item></list></p>
      <p id="d1e5927">The reconstruction and factorization performance of the different models are compared in Table <xref ref-type="table" rid="Ch1.T2"/>. The fully constrained BAMF model (BAMF-C) performs substantially better in extracting BBOA both in <inline-formula><mml:math id="M229" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M230" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> compared to BAMF. In fact, the extracted components are very similar to the true components with very similar temporal behavior and reduced biases in <inline-formula><mml:math id="M231" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> (Fig. <xref ref-type="fig" rid="Ch1.F9"/>). Fully constrained PMF performs marginally better than unconstrained PMF but still mixes SOA<inline-formula><mml:math id="M232" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">bio</mml:mi></mml:msub></mml:math></inline-formula> with SOA<inline-formula><mml:math id="M233" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">anthro</mml:mi></mml:msub></mml:math></inline-formula> (Fig. <xref ref-type="fig" rid="Ch1.F9"/>). This can be expected since the a priori information is applied to HOA and BBOA, not to the SOA components. Figure <xref ref-type="fig" rid="Ch1.F9"/>d shows that PMF-C captures HOA time behavior very well, while BAMF still overestimates some auto-correlation, even though the bias in Fig. <xref ref-type="fig" rid="Ch1.F9"/>a is significantly reduced. PMF-C also has some solutions that start to approach correct SOA bio, with solutions falling between unconstrained PMF and BAMF-C. For BAMF-C and BAMF SOA bio is virtually unchanged (Figs. <xref ref-type="fig" rid="Ch1.F9"/>a, d  and <xref ref-type="fig" rid="Ch1.F6"/>a, d). The incompletely constrained BAMF model performs slightly worse than the fully constrained BAMF model. Partially constrained BAMF performs worse but is still on par with the fully constrained PMF (Table <xref ref-type="table" rid="Ch1.T2"/>). The partially constrained BAMF model reduces bias in BBOA but starts mixing SOA bio with the other components (Fig. <xref ref-type="fig" rid="Ch1.F10"/>, Table <xref ref-type="table" rid="Ch1.T2"/>). Using a priori information on <inline-formula><mml:math id="M234" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> improves the factorization performance of both PMF and especially BAMF, with more information leading to solutions closer to the ground truth. This is especially helpful when the additional information is on the components the model mixes when unconstrained. Such prior information is powerful, e.g., constrained PMF does better than unconstrained BAMF for all components except SOA bio. For this dataset and these constraints, BAMF-C fully constrained has the best factorization performance, fully constrained PMF and partially constrained BAMF-C perform slightly worse but about equally good, and unconstrained models fail to resolve one or more components correctly regardless of the model. A comparison of the results of fully constrained PMF and unconstrained BAMF can be seen in Appendix <xref ref-type="sec" rid="App1.Ch1.S7"/>.</p>

      <?xmltex \floatpos{p}?><fig id="Ch1.F9" specific-use="star"><?xmltex \currentcnt{9}?><?xmltex \def\figurename{Figure}?><label>Figure 9</label><caption><p id="d1e6003">Factorization performance of models using a priori information on <inline-formula><mml:math id="M235" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> for the synthetic European city ToF-ACSM OA dataset, fully constrained BAMF and fully constrained PMF results compared to unconstrained PMF. Panel <bold>(a)</bold> is <inline-formula><mml:math id="M236" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, <bold>(b)</bold> is <inline-formula><mml:math id="M237" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, <bold>(c)</bold> is the diurnal concentration, and <bold>(d)</bold> is the auto-correlation behavior.</p></caption>
          <?xmltex \igopts{width=398.338583pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f09.png"/>

        </fig>

      <?xmltex \floatpos{p}?><fig id="Ch1.F10" specific-use="star"><?xmltex \currentcnt{10}?><?xmltex \def\figurename{Figure}?><label>Figure 10</label><caption><p id="d1e6048">Factorization performance of models using a priori information on <inline-formula><mml:math id="M238" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> for the synthetic European city ToF-ACSM OA dataset, partial and incomplete constraints in BAMF compared to unconstrained BAMF. Panel <bold>(a)</bold> is <inline-formula><mml:math id="M239" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula>, <bold>(b)</bold> is <inline-formula><mml:math id="M240" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula>, <bold>(c)</bold> is the diurnal concentration, and <bold>(d)</bold> is the auto-correlation behavior. The results for SOA components overlap between BAMF and BAMF-C(incomplete) such that the BAMF results are not visible in panels <bold>(b)</bold>, <bold>(c)</bold>, and <bold>(d)</bold>.</p></caption>
          <?xmltex \igopts{width=398.338583pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f10.png"/>

        </fig>

</sec>
</sec>
<sec id="Ch1.S5" sec-type="conclusions">
  <label>5</label><title>Conclusions</title>
      <?pagebreak page1269?><p id="d1e6110">We present a Bayesian matrix factorization model that accounts for temporal auto-correlation of the components (BAMF) and provides direct error estimation. BAMF is built on top of Stan, a freely available, robust, actively developed, open-source framework for statistical modeling with the ability of full Bayesian statistical inference with MCMC sampling. Here, we characterize the BAMF performance on synthetic Time-of-Flight Aerosol Chemical Speciation Monitor mass spectral OA data compared to PMF. This approach allows us to assess the model performance based on input data reconstruction and the ability to accurately model the chemical composition and concentration time series of the components.</p>
      <p id="d1e6113">All models performed well in reconstruction performance regardless of factorization performance, indicating that reconstructing the data is insufficient for judging how good the extracted factors are.  Without strongly correlated components, BAMF resolves temporally auto-correlated components well (synthetic megacity dataset), while PMF performs considerably worse. Both BAMF and PMF are challenged by strongly correlated components (European data).</p>
      <p id="d1e6116">Further, we show that using a priori information on the chemical composition of the components improves BAMF factorization performance such that all components are well represented. Even adding a priori information for a few peaks significantly reduced component bias, and partially specifying the profile (for 56 % of the peaks) produced comparable results to fully constraining the profile with PMF. This opens up possibilities for using incomplete chemical composition information to improve factorizations.</p>
      <p id="d1e6119">While we tested BAMF on synthetic OA ToF-ACSM data in this paper, source apportionment analyses of other chemical PM data (e.g., trace elements from either Xact or offline filter analysis) could also profit from accounting for the auto-correlation of components, if the components are auto-correlated. Further testing is especially needed for datasets with temporally sparse sources, i.e., pollution sources occurring only during specific events, which are also challenging for PMF.</p>
      <p id="d1e6123">Overall, we believe BAMF-type models are promising tools for source apportionment and deserve further research, e.g., improving the separation of the chemical composition of components or the computational speed of BAMF. These models can also be used complementary to current source apportionment methods due to their different emphasis and advantages. One such research topic would be introducing rolling window methods as has been done with PMF, to allow the source profiles to change over time and to act as a basis for real-time source apportionment. Other possible topics are using BAMF with other time series instruments and with real-world data. Another area of development is computational speed – for the dataset sizes discussed here running BAMF takes a few hours on a modern computer (Intel Xeon Silver 4110), but the time increases as the data size increases.</p>
</sec>

      
      </body>
    <back><app-group>

<app id="App1.Ch1.S1">
  <?xmltex \currentcnt{A}?><label>Appendix A</label><title>The correlation of BAMF CCOA error in the five-component megacity datasets</title>
      <p id="d1e6137">The error in <inline-formula><mml:math id="M241" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> for CCOA seems to be correlated with BBOA error (Fig. <xref ref-type="fig" rid="App1.Ch1.S1.F11"/>) with correlation coefficient <inline-formula><mml:math id="M242" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>0.5 and the error quickly increases as the component is underestimated as shown in Fig. <xref ref-type="fig" rid="App1.Ch1.S1.F12"/>. Thus it seems that the variance in the reconstruction for CCOA in BAMF is due to the mixing of BBOA and CCOA components. This is probably due to these components having very similar <inline-formula><mml:math id="M243" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M244" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> profiles as seen in Tables <xref ref-type="table" rid="App1.Ch1.S2.T3"/> and <xref ref-type="table" rid="App1.Ch1.S2.T4"/>.</p>

      <?xmltex \floatpos{h!}?><fig id="App1.Ch1.S1.F11"><?xmltex \currentcnt{A1}?><?xmltex \def\figurename{Figure}?><label>Figure A1</label><caption><p id="d1e6179">The bias of CCOA and BBOA <inline-formula><mml:math id="M245" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> component in the five-component megacity datasets.</p></caption>
        <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f11.png"/>

      </fig>

      <?xmltex \floatpos{h!}?><fig id="App1.Ch1.S1.F12"><?xmltex \currentcnt{A2}?><?xmltex \def\figurename{Figure}?><label>Figure A2</label><caption><p id="d1e6197">The bias of CCOA time series and composition in the five-component megacity datasets.</p></caption>
        <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f12.png"/>

      </fig>

<?xmltex \hack{\clearpage}?>
</app>

<?pagebreak page1270?><app id="App1.Ch1.S2">
  <?xmltex \currentcnt{B}?><label>Appendix B</label><title>Correlation between true components in the datasets</title>
      <p id="d1e6216">Comparing the true components of the datasets shows that the European dataset components are much more correlated both in <inline-formula><mml:math id="M246" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> Tables <xref ref-type="table" rid="App1.Ch1.S2.T3"/> and <xref ref-type="table" rid="App1.Ch1.S2.T5"/>, as well as <inline-formula><mml:math id="M247" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> Tables <xref ref-type="table" rid="App1.Ch1.S2.T4"/> and <xref ref-type="table" rid="App1.Ch1.S2.T6"/>. Overall <inline-formula><mml:math id="M248" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> components have similar, very high, correlations with each other, while the <inline-formula><mml:math id="M249" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> components are markedly more correlated with each other in the European dataset than in the megacity datasets, with biological SOA being the exception.</p>

<?xmltex \floatpos{h!}?><table-wrap id="App1.Ch1.S2.T3"><?xmltex \hack{\hsize\textwidth}?><?xmltex \currentcnt{B1}?><label>Table B1</label><caption><p id="d1e6260">Pearson correlation and standard deviation of the <inline-formula><mml:math id="M250" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> components used to construct the 10 five-component megacity datasets.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="6">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:colspec colnum="6" colname="col6" align="right"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Component</oasis:entry>
         <oasis:entry colname="col2">OOA</oasis:entry>
         <oasis:entry colname="col3">HOA</oasis:entry>
         <oasis:entry colname="col4">COA</oasis:entry>
         <oasis:entry colname="col5">BBOA</oasis:entry>
         <oasis:entry colname="col6">CCOA</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">OOA</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M251" display="inline"><mml:mn mathvariant="normal">1</mml:mn></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M252" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.079</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.067</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M253" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.039</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.173</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M254" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.044</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.158</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col6"><inline-formula><mml:math id="M255" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.033</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.091</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">HOA</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M256" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.079</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.067</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M257" display="inline"><mml:mn mathvariant="normal">1</mml:mn></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M258" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.076</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.069</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M259" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.133</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.104</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col6"><inline-formula><mml:math id="M260" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.171</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.061</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">COA</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M261" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.039</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.173</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M262" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.076</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.069</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M263" display="inline"><mml:mn mathvariant="normal">1</mml:mn></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M264" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.330</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.093</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col6"><inline-formula><mml:math id="M265" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.346</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.047</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">BBOA</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M266" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.044</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.158</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M267" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.133</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.104</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M268" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.330</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.093</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M269" display="inline"><mml:mn mathvariant="normal">1</mml:mn></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col6"><inline-formula><mml:math id="M270" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.503</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.096</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">CCOA</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M271" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.033</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.091</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M272" display="inline"><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.171</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.061</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M273" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.346</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.047</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M274" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.503</mml:mn><mml:mo>±</mml:mo><mml:mn mathvariant="normal">0.096</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col6"><inline-formula><mml:math id="M275" display="inline"><mml:mn mathvariant="normal">1</mml:mn></mml:math></inline-formula></oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table><?xmltex \gdef\@currentlabel{B1}?></table-wrap>

<?xmltex \floatpos{h!}?><table-wrap id="App1.Ch1.S2.T4"><?xmltex \currentcnt{B2}?><label>Table B2</label><caption><p id="d1e6670">Spearman correlation of the <inline-formula><mml:math id="M276" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> components used to construct the five-component megacity datasets. Note that <inline-formula><mml:math id="M277" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> does not change between datasets, and thus standard deviation is 0.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="6">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:colspec colnum="6" colname="col6" align="right"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Component</oasis:entry>
         <oasis:entry colname="col2">OOA</oasis:entry>
         <oasis:entry colname="col3">HOA</oasis:entry>
         <oasis:entry colname="col4">COA</oasis:entry>
         <oasis:entry colname="col5">BBOA</oasis:entry>
         <oasis:entry colname="col6">CCOA</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">OOA</oasis:entry>
         <oasis:entry colname="col2">1</oasis:entry>
         <oasis:entry colname="col3">0.761</oasis:entry>
         <oasis:entry colname="col4">0.659</oasis:entry>
         <oasis:entry colname="col5">0.804</oasis:entry>
         <oasis:entry colname="col6">0.816</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">HOA</oasis:entry>
         <oasis:entry colname="col2">0.761</oasis:entry>
         <oasis:entry colname="col3">1</oasis:entry>
         <oasis:entry colname="col4">0.839</oasis:entry>
         <oasis:entry colname="col5">0.864</oasis:entry>
         <oasis:entry colname="col6">0.845</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">COA</oasis:entry>
         <oasis:entry colname="col2">0.659</oasis:entry>
         <oasis:entry colname="col3">0.839</oasis:entry>
         <oasis:entry colname="col4">1</oasis:entry>
         <oasis:entry colname="col5">0.850</oasis:entry>
         <oasis:entry colname="col6">0.754</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">BBOA</oasis:entry>
         <oasis:entry colname="col2">0.804</oasis:entry>
         <oasis:entry colname="col3">0.864</oasis:entry>
         <oasis:entry colname="col4">0.850</oasis:entry>
         <oasis:entry colname="col5">1</oasis:entry>
         <oasis:entry colname="col6">0.871</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">CCOA</oasis:entry>
         <oasis:entry colname="col2">0.816</oasis:entry>
         <oasis:entry colname="col3">0.845</oasis:entry>
         <oasis:entry colname="col4">0.754</oasis:entry>
         <oasis:entry colname="col5">0.871</oasis:entry>
         <oasis:entry colname="col6">1</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table><?xmltex \gdef\@currentlabel{B2}?></table-wrap>

<?xmltex \floatpos{h!}?><table-wrap id="App1.Ch1.S2.T5"><?xmltex \currentcnt{B3}?><label>Table B3</label><caption><p id="d1e6847">Pearson correlation of the <inline-formula><mml:math id="M278" display="inline"><mml:mi mathvariant="bold">G</mml:mi></mml:math></inline-formula> components used to construct the European dataset.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="5">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Component</oasis:entry>
         <oasis:entry colname="col2">HOA</oasis:entry>
         <oasis:entry colname="col3">BBOA</oasis:entry>
         <oasis:entry colname="col4">SOA traffic</oasis:entry>
         <oasis:entry colname="col5">SOA bio</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">HOA</oasis:entry>
         <oasis:entry colname="col2">1</oasis:entry>
         <oasis:entry colname="col3">0.773</oasis:entry>
         <oasis:entry colname="col4">0.718</oasis:entry>
         <oasis:entry colname="col5">0.005</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">BBOA</oasis:entry>
         <oasis:entry colname="col2">0.773</oasis:entry>
         <oasis:entry colname="col3">1</oasis:entry>
         <oasis:entry colname="col4">0.629</oasis:entry>
         <oasis:entry colname="col5">0.073</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SOA traffic</oasis:entry>
         <oasis:entry colname="col2">0.718</oasis:entry>
         <oasis:entry colname="col3">0.629</oasis:entry>
         <oasis:entry colname="col4">1</oasis:entry>
         <oasis:entry colname="col5">0.018</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SOA bio</oasis:entry>
         <oasis:entry colname="col2">0.005</oasis:entry>
         <oasis:entry colname="col3">0.073</oasis:entry>
         <oasis:entry colname="col4">0.018</oasis:entry>
         <oasis:entry colname="col5">1</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table><?xmltex \gdef\@currentlabel{B3}?></table-wrap>

<?xmltex \floatpos{h!}?><table-wrap id="App1.Ch1.S2.T6"><?xmltex \currentcnt{B4}?><label>Table B4</label><caption><p id="d1e6972">Spearman correlation of the <inline-formula><mml:math id="M279" display="inline"><mml:mi mathvariant="bold">F</mml:mi></mml:math></inline-formula> components used to construct the European dataset.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="5">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Component</oasis:entry>
         <oasis:entry colname="col2">HOA</oasis:entry>
         <oasis:entry colname="col3">BBOA</oasis:entry>
         <oasis:entry colname="col4">SOA traffic</oasis:entry>
         <oasis:entry colname="col5">SOA bio</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">HOA</oasis:entry>
         <oasis:entry colname="col2">1</oasis:entry>
         <oasis:entry colname="col3">0.852</oasis:entry>
         <oasis:entry colname="col4">0.844</oasis:entry>
         <oasis:entry colname="col5">0.801</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">BBOA</oasis:entry>
         <oasis:entry colname="col2">0.852</oasis:entry>
         <oasis:entry colname="col3">1</oasis:entry>
         <oasis:entry colname="col4">0.875</oasis:entry>
         <oasis:entry colname="col5">0.847</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SOA traffic</oasis:entry>
         <oasis:entry colname="col2">0.844</oasis:entry>
         <oasis:entry colname="col3">0.875</oasis:entry>
         <oasis:entry colname="col4">1</oasis:entry>
         <oasis:entry colname="col5">0.845</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SOA bio</oasis:entry>
         <oasis:entry colname="col2">0.801</oasis:entry>
         <oasis:entry colname="col3">0.847</oasis:entry>
         <oasis:entry colname="col4">0.845</oasis:entry>
         <oasis:entry colname="col5">1</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table><?xmltex \gdef\@currentlabel{B4}?></table-wrap>

<?xmltex \hack{\newpage}?><?xmltex \hack{\vspace*{81mm}}?>
</app>

<app id="App1.Ch1.S3">
  <?xmltex \currentcnt{C}?><label>Appendix C</label><?xmltex \opttitle{Concentration and uncertainty at selected $m/$zs for synthetic megacity ToF-ACSM OA data}?><title>Concentration and uncertainty at selected <inline-formula><mml:math id="M280" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo></mml:mrow></mml:math></inline-formula>zs for synthetic megacity ToF-ACSM OA data</title>
      <p id="d1e7116">Figure <xref ref-type="fig" rid="App1.Ch1.S3.F13"/> shows example time series for selected <inline-formula><mml:math id="M281" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula>s from the synthetic megacity data.</p>

      <?xmltex \floatpos{h!}?><fig id="App1.Ch1.S3.F13"><?xmltex \currentcnt{C1}?><?xmltex \def\figurename{Figure}?><label>Figure C1</label><caption><p id="d1e7135">Concentration and uncertainty time series at selected <inline-formula><mml:math id="M282" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula>s for synthetic megacity ToF-ACSM OA data. The shaded area contains 95 % of the probability mass of the Gaussian distribution of the error.</p></caption>
        <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f13.png"/>

      </fig>

<?xmltex \hack{\clearpage}?>
</app>

<?pagebreak page1271?><app id="App1.Ch1.S4">
  <?xmltex \currentcnt{D}?><label>Appendix D</label><title>Overspecified BAMF-C results</title>
      <p id="d1e7166">Figure <xref ref-type="fig" rid="App1.Ch1.S4.F14"/> shows results from BAMF-C with too many components.</p>

      <?xmltex \floatpos{h!}?><fig id="App1.Ch1.S4.F14"><?xmltex \currentcnt{D1}?><?xmltex \def\figurename{Figure}?><label>Figure D1</label><caption><p id="d1e7173">Results from overspecified BAMF-C model for the synthetic European city ToF-ACSM OA dataset with five modeled components instead of four and HOA and BBOA fully constrained.</p></caption>
        <?xmltex \hack{\hsize\textwidth}?>
        <?xmltex \igopts{width=341.433071pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f14.png"/>

      </fig>

<?xmltex \hack{\clearpage}?>
</app>

<?pagebreak page1272?><app id="App1.Ch1.S5">
  <?xmltex \currentcnt{E}?><label>Appendix E</label><title>Workflow in this study</title>
      <p id="d1e7194">Figure <xref ref-type="fig" rid="App1.Ch1.S5.F15"/> summarizes the steps used to run the BAMF model in this study and their order. In this study the profiles were from the literature (see Appendix <xref ref-type="sec" rid="App1.Ch1.S6"/>). In general use, the data generation step is not needed.</p>

      <?xmltex \floatpos{h!}?><fig id="App1.Ch1.S5.F15"><?xmltex \currentcnt{E1}?><?xmltex \def\figurename{Figure}?><label>Figure E1</label><caption><p id="d1e7203">Workflow of running the BAMF model in this study. Pre- and post-processing steps are technically optional but help in the convergence and interpretation of the results. With PMF the pre-processing and denormalization are skipped and the modeling box is just PMF, but we still sort the data similarly.</p></caption>
        <?xmltex \hack{\hsize\textwidth}?>
        <?xmltex \igopts{width=398.338583pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f15.png"/>

      </fig>

</app>

<app id="App1.Ch1.S6">
  <?xmltex \currentcnt{F}?><label>Appendix F</label><title>Profiles and constraints used</title>
      <p id="d1e7223">Table <xref ref-type="table" rid="App1.Ch1.S6.T7"/> summarizes the source profiles and constraints used. Constraints were only applied in the European dataset.</p>

<?xmltex \floatpos{h!}?><table-wrap id="App1.Ch1.S6.T7"><?xmltex \hack{\hsize\textwidth}?><?xmltex \currentcnt{F1}?><label>Table F1</label><caption><p id="d1e7232">The source profiles used to construct the datasets. The profiles were restricted to <inline-formula><mml:math id="M283" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>/</mml:mo><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula> 12–100, summed to unit mass intervals, and normalized to sum to 1.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="3">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:colspec colnum="3" colname="col3" align="left"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Component</oasis:entry>
         <oasis:entry colname="col2">Megacity dataset</oasis:entry>
         <oasis:entry colname="col3">European dataset</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">HOA<inline-formula><mml:math id="M285" display="inline"><mml:msup><mml:mi/><mml:mo>*</mml:mo></mml:msup></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col2">
                  <xref ref-type="bibr" rid="bib1.bibx14" id="text.54"/>
                </oasis:entry>
         <oasis:entry colname="col3">
                  <xref ref-type="bibr" rid="bib1.bibx14" id="text.55"/>
                </oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">BBOA<inline-formula><mml:math id="M286" display="inline"><mml:msup><mml:mi/><mml:mo>*</mml:mo></mml:msup></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col2">
                  <xref ref-type="bibr" rid="bib1.bibx14" id="text.56"/>
                </oasis:entry>
         <oasis:entry colname="col3">
                  <xref ref-type="bibr" rid="bib1.bibx14" id="text.57"/>
                </oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">CCOA</oasis:entry>
         <oasis:entry colname="col2">
                  <xref ref-type="bibr" rid="bib1.bibx14" id="text.58"/>
                </oasis:entry>
         <oasis:entry colname="col3">Not applicable</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">COA</oasis:entry>
         <oasis:entry colname="col2">
                  <xref ref-type="bibr" rid="bib1.bibx14" id="text.59"/>
                </oasis:entry>
         <oasis:entry colname="col3">Not applicable</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">OOA</oasis:entry>
         <oasis:entry colname="col2">
                  <xref ref-type="bibr" rid="bib1.bibx14" id="text.60"/>
                </oasis:entry>
         <oasis:entry colname="col3">Not applicable</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SOA bio</oasis:entry>
         <oasis:entry colname="col2">Not applicable</oasis:entry>
         <oasis:entry colname="col3">
                  <xref ref-type="bibr" rid="bib1.bibx12" id="text.61"/>
                </oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SOA anthro</oasis:entry>
         <oasis:entry colname="col2">Not applicable</oasis:entry>
         <oasis:entry colname="col3">
                  <xref ref-type="bibr" rid="bib1.bibx34" id="text.62"/>
                </oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table><table-wrap-foot><p id="d1e7247"><inline-formula><mml:math id="M284" display="inline"><mml:msup><mml:mi/><mml:mo>*</mml:mo></mml:msup></mml:math></inline-formula> Indicates the values from this profile were also used as constraints.</p></table-wrap-foot><?xmltex \gdef\@currentlabel{F1}?></table-wrap>

<?xmltex \hack{\clearpage}?>
</app>

<?pagebreak page1273?><app id="App1.Ch1.S7">
  <?xmltex \currentcnt{G}?><label>Appendix G</label><title>Comparison of constrained PMF and BAMF</title>
      <p id="d1e7425">Figure <xref ref-type="fig" rid="App1.Ch1.S7.F16"/> compares PMF with constraints to BAMF without them. This represents the absolute best-case scenario for PMF where you know the exact HOA and BBOA profiles beforehand. As mentioned in the main text, this does not fix the inability to resolve SOA bio.</p>

      <?xmltex \floatpos{h!}?><fig id="App1.Ch1.S7.F16"><?xmltex \currentcnt{G1}?><?xmltex \def\figurename{Figure}?><label>Figure G1</label><caption><p id="d1e7432">Unconstrained BAMF and constrained PMF on the European dataset with four components.</p></caption>
        <?xmltex \hack{\hsize\textwidth}?>
        <?xmltex \igopts{width=398.338583pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f16.png"/>

      </fig>

<?xmltex \hack{\clearpage}?>
</app>

<?pagebreak page1274?><app id="App1.Ch1.S8">
  <?xmltex \currentcnt{H}?><label>Appendix H</label><title>Error estimates and interquartile range</title>
      <p id="d1e7453">The error bars and the shaded areas in the time series are based on the interquartile ranges (IQR) in the empirical distribution given by the MCMC sampler. This gives us an idea of how accurately we can fix the modeled concentrations and compositions. In these results the error estimation is a bit optimistic, since it does not always cover the true solution. The underestimation is possibly due to the strictness of IQR and it not considering the model choice error. Figure <xref ref-type="fig" rid="App1.Ch1.S8.F17"/> shows how the IQR compares to the median answer, with small concentrations having the most relative uncertainty. It also shows that BAMF-0 and PMF are often more uncertain than BAMF. In the case of PMF this is probably due to the values not being samples but solutions with different random seeds.</p>

      <?xmltex \floatpos{h!}?><fig id="App1.Ch1.S8.F17"><?xmltex \currentcnt{H1}?><?xmltex \def\figurename{Figure}?><label>Figure H1</label><caption><p id="d1e7460">IQR compared to the median on the base case of the European dataset</p></caption>
        <?xmltex \hack{\hsize\textwidth}?>
        <?xmltex \igopts{width=426.791339pt}?><graphic xlink:href="https://amt.copernicus.org/articles/17/1251/2024/amt-17-1251-2024-f17.png"/>

      </fig>

</app>
  </app-group><notes notes-type="codedataavailability"><title>Code and data availability</title>

      <p id="d1e7475">The datasets are available at <ext-link xlink:href="https://doi.org/10.5281/zenodo.10629577" ext-link-type="DOI">10.5281/zenodo.10629577</ext-link> <xref ref-type="bibr" rid="bib1.bibx32" id="paren.63"/> and the software at <ext-link xlink:href="https://doi.org/10.5281/zenodo.10629849" ext-link-type="DOI">10.5281/zenodo.10629849</ext-link> <xref ref-type="bibr" rid="bib1.bibx33" id="paren.64"/>.</p>
  </notes><notes notes-type="authorcontribution"><title>Author contributions</title>

      <p id="d1e7493">AR, AB, KRD, and KP participated in the model, dataset, and experiment development. AR ran and analyzed the experiments. KRD and MIM contributed to the PMF model solutions. JJ did the transport model runs. KRD, KP and MTK supervised the work. All authors contributed to the writing of the manuscript.</p>
  </notes><notes notes-type="competinginterests"><title>Competing interests</title>

      <p id="d1e7499">The contact author has declared that none of the authors has any competing interests.</p>
  </notes><?xmltex \hack{\newpage}?><?xmltex \hack{\vspace*{153mm}}?><notes notes-type="disclaimer"><title>Disclaimer</title>

      <p id="d1e7507">Publisher’s note: Copernicus Publications remains neutral with regard to jurisdictional claims made in the text, published maps, institutional affiliations, or any other geographical representation in this paper. While Copernicus Publications makes every effort to include appropriate place names, the final responsibility lies with the authors.</p>
  </notes><ack><title>Acknowledgements</title><p id="d1e7513">We thank the Research Council of Finland for its support. Kaspar R. Daellenbach acknowledges support by SNSF Ambizione. Jianhui Jiang acknowledges support by Science and Technology Commission of Shanghai Municipality, China.</p></ack><notes notes-type="financialsupport"><title>Financial support</title>

      <p id="d1e7518">This research has been supported by the Research Council of Finland (decision nos. 337549, 345704,<?pagebreak page1275?> and 346376). Kaspar R. Daellenbach has been supported by SNSF Ambizione (grant no. PZPGP2_201992). Jianhui Jiang has been supported by Science and Technology Commission of Shanghai Municipality, China (Shanghai Pujiang Program, grant no. 21PJ1402800). <?xmltex \hack{\newline}?><?xmltex \hack{\newline}?> Open-access funding was provided by the Helsinki<?xmltex \notforhtml{\newline}?> University Library.</p>
  </notes><notes notes-type="reviewstatement"><title>Review statement</title>

      <p id="d1e7530">This paper was edited by Eric C. Apel and reviewed by two anonymous referees.</p>
  </notes><ref-list>
    <title>References</title>

      <ref id="bib1.bibx1"><?xmltex \def\ref@label{{Allan et al.(2003)}}?><label>Allan et al.(2003)</label><?label Allan2003?><mixed-citation>Allan, J. D., Jimenez, J. L., Williams, P. I., Alfarra, M. R., Bower, K. N.,  Jayne, J. T., Coe, H., and Worsnop, D. R.: Quantitative sampling using an  Aerodyne aerosol mass spectrometer 1. Techniques of data interpretation and  error analysis, J. Geophys. Res.-Atmos., 108, 4090, <ext-link xlink:href="https://doi.org/10.1029/2002JD002358" ext-link-type="DOI">10.1029/2002JD002358</ext-link>, 2003.</mixed-citation></ref>
      <ref id="bib1.bibx2"><?xmltex \def\ref@label{{Bates et al.(2019)}}?><label>Bates et al.(2019)</label><?label Bates2019?><mixed-citation>Bates, J. T., Fang, T., Verma, V., Zeng, L., Weber, R. J., Tolbert, P. E.,  Abrams, J. Y., Sarnat, S. E., Klein, M., Mulholland, J. A., and Russell,  A. G.: Review of Acellular Assays of Ambient Particulate Matter Oxidative  Potential: Methods and Relationships with Composition, Sources, and Health  Effects, Environ. Sci. Technol., 53, 4003–4019,  <ext-link xlink:href="https://doi.org/10.1021/acs.est.8b03430" ext-link-type="DOI">10.1021/acs.est.8b03430</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bibx3"><?xmltex \def\ref@label{{Canagaratna et al.(2007)}}?><label>Canagaratna et al.(2007)</label><?label Canagarat2007?><mixed-citation> Canagaratna, M., Jayne, J., Jimenez, J., Allan, J., Alfarra, M., Zhang, Q.,  Onasch, T., Drewnick, F., Coe, H., Middlebrook, A., Delia, A., Williams, L.,  Trimborn, A., Northway, M., DeCarlo, P., Kolb, C., Davidovits, P., and  Worsnop, D.: Chemical and microphysical characterization of ambient aerosols  with the aerodyne aerosol mass spectrometer, Mass Spectrom. Rev., 26,  185–222, 2007.</mixed-citation></ref>
      <ref id="bib1.bibx4"><?xmltex \def\ref@label{{Canonaco et al.(2013)}}?><label>Canonaco et al.(2013)</label><?label Sofipaper?><mixed-citation>Canonaco, F., Crippa, M., Slowik, J. G., Baltensperger, U., and Prévôt, A. S. H.: SoFi, an IGOR-based interface for the efficient use of the generalized multilinear engine (ME-2) for the source apportionment: ME-2 application to aerosol mass spectrometer data, Atmos. Meas. Tech., 6, 3649–3661, <ext-link xlink:href="https://doi.org/10.5194/amt-6-3649-2013" ext-link-type="DOI">10.5194/amt-6-3649-2013</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bibx5"><?xmltex \def\ref@label{{Canonaco et al.(2021)}}?><label>Canonaco et al.(2021)</label><?label Canonaco2021?><mixed-citation>Canonaco, F., Tobler, A., Chen, G., Sosedova, Y., Slowik, J. G., Bozzetti, C., Daellenbach, K. R., El Haddad, I., Crippa, M., Huang, R.-J., Furger, M., Baltensperger, U., and Prévôt, A. S. H.: A new method for long-term source apportionment with time-dependent factor profiles and uncertainty assessment using SoFi Pro: application to 1 year of organic aerosol data, Atmos. Meas. Tech., 14, 923–943, <ext-link xlink:href="https://doi.org/10.5194/amt-14-923-2021" ext-link-type="DOI">10.5194/amt-14-923-2021</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bibx6"><?xmltex \def\ref@label{{Carpenter et al.(2017)}}?><label>Carpenter et al.(2017)</label><?label stan?><mixed-citation>Carpenter, B., Gelman, A., Hoffman, M., Lee, D., Goodrich, B., Betancourt, M., Brubaker, M., Guo, J., Li, P., and Riddell, A.: Stan: A Probabilistic  Programming Language, J. Stat. Softw., 76, 1–32,  <ext-link xlink:href="https://doi.org/10.18637/jss.v076.i01" ext-link-type="DOI">10.18637/jss.v076.i01</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bibx7"><?xmltex \def\ref@label{{Chazeau et al.(2022)}}?><label>Chazeau et al.(2022)</label><?label Chazeau2022?><mixed-citation>Chazeau, B., El Haddad, I., Canonaco, F., Temime-Roussel, B., D'Anna, B.,  Gille, G., Mesbah, B., Prévôt, A. S., Wortham, H., and Marchand, N.: Organic aerosol source apportionment by using rolling positive matrix factorization: Application to a Mediterranean coastal city, Atmospheric Environment: X, 14, 100176, <ext-link xlink:href="https://doi.org/10.1016/j.aeaoa.2022.100176" ext-link-type="DOI">10.1016/j.aeaoa.2022.100176</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bibx8"><?xmltex \def\ref@label{{Chen(2022)}}?><label>Chen(2022)</label><?label fig1dataset?><mixed-citation>Chen, G.: European Aerosol Phenomenology - 8: Harmonised Source Apportionment of Organic Aerosol using 22 Year-long ACSM/AMS Datasets, Version 2nd, Zenodo [data set], <ext-link xlink:href="https://doi.org/10.5281/zenodo.6672710" ext-link-type="DOI">10.5281/zenodo.6672710</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bibx9"><?xmltex \def\ref@label{{Chen et al.(2022)}}?><label>Chen et al.(2022)</label><?label Chen2022?><mixed-citation>Chen, G., Canonaco, F., Tobler, A., Aas, W., Alastuey, A., Allan, J.,  Atabakhsh, S., Aurela, M., Baltensperger, U., Bougiatioti, A., De Brito,  J. F., Ceburnis, D., Chazeau, B., Chebaicheb, H., Daellenbach, K. R., Ehn,  M., El Haddad, I., Eleftheriadis, K., Favez, O., Flentje, H., Font, A.,  Fossum, K., Freney, E., Gini, M., Green, D. C., Heikkinen, L., Herrmann, H.,  Kalogridis, A.-C., Keernik, H., Lhotka, R., Lin, C., Lunder, C., Maasikmets,  M., Manousakas, M. I., Marchand, N., Marin, C., Marmureanu, L., Mihalopoulos,  N., Močnik, G., Nęcki, J., O'Dowd, C., Ovadnevaite, J., Peter, T., Petit,  J.-E., Pikridas, M., Matthew Platt, S., Pokorná, P., Poulain, L.,  Priestman, M., Riffault, V., Rinaldi, M., Różański, K., Schwarz, J., Sciare, J., Simon, L., Skiba, A., Slowik, J. G., Sosedova, Y., Stavroulas, I., Styszko, K., Teinemaa, E., Timonen, H., Tremper, A., Vasilescu, J., Via, M., Vodička, P., Wiedensohler, A., Zografou, O., Cruz Minguillón, M., and Prévôt, A. S.: European aerosol phenomenology - 8: Harmonised source apportionment of organic aerosol using 22 Year-long ACSM/AMS datasets,  Environ. Int., 166, 107325, <ext-link xlink:href="https://doi.org/10.1016/j.envint.2022.107325" ext-link-type="DOI">10.1016/j.envint.2022.107325</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bibx10"><?xmltex \def\ref@label{{Crippa et al.(2013)}}?><label>Crippa et al.(2013)</label><?label crippa?><mixed-citation>Crippa, M., DeCarlo, P. F., Slowik, J. G., Mohr, C., Heringa, M. F., Chirico, R., Poulain, L., Freutel, F., Sciare, J., Cozic, J., Di Marco, C. F., Elsasser, M., Nicolas, J. B., Marchand, N., Abidi, E., Wiedensohler, A., Drewnick, F., Schneider, J., Borrmann, S., Nemitz, E., Zimmermann, R., Jaffrezo, J.-L., Prévôt, A. S. H., and Baltensperger, U.: Wintertime aerosol chemical composition and source apportionment of the organic fraction in the metropolitan area of Paris, Atmos. Chem. Phys., 13, 961–981, <ext-link xlink:href="https://doi.org/10.5194/acp-13-961-2013" ext-link-type="DOI">10.5194/acp-13-961-2013</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bibx11"><?xmltex \def\ref@label{{Crippa et al.(2014)}}?><label>Crippa et al.(2014)</label><?label Crippa2014?><mixed-citation>Crippa, M., Canonaco, F., Lanz, V. A., Äijälä, M., Allan, J. D., Carbone, S., Capes, G., Ceburnis, D., Dall'Osto, M., Day, D. A., DeCarlo, P. F., Ehn, M., Eriksson, A., Freney, E., Hildebrandt Ruiz, L., Hillamo, R., Jimenez, J. L., Junninen, H., Kiendler-Scharr, A., Kortelainen, A.-M., Kulmala, M., Laaksonen, A., Mensah, A. A., Mohr, C., Nemitz, E., O'Dowd, C., Ovadnevaite, J., Pandis, S. N., Petäjä, T., Poulain, L., Saarikoski, S., Sellegri, K., Swietlicki, E., Tiitta, P., Worsnop, D. R., Baltensperger, U., and Prévôt, A. S. H.: Organic aerosol components derived from 25 AMS data sets across Europe using a consistent ME-2 based source apportionment approach, Atmos. Chem. Phys., 14, 6159–6176, <ext-link xlink:href="https://doi.org/10.5194/acp-14-6159-2014" ext-link-type="DOI">10.5194/acp-14-6159-2014</ext-link>, 2014.</mixed-citation></ref>
      <ref id="bib1.bibx12"><?xmltex \def\ref@label{{Daellenbach et al.(2017)}}?><label>Daellenbach et al.(2017)</label><?label Daellenbach2017?><mixed-citation>Daellenbach, K. R., Stefenelli, G., Bozzetti, C., Vlachou, A., Fermo, P., Gonzalez, R., Piazzalunga, A., Colombi, C., Canonaco, F., Hueglin, C., Kasper-Giebl, A., Jaffrezo, J.-L., Bianchi, F., Slowik, J. G., Baltensperger, U., El-Haddad, I., and Prévôt, A. S. H.: Long-term chemical analysis and organic aerosol source apportionment at nine sites in central Europe: source identification and uncertainty assessment, Atmos. Chem. Phys., 17, 13265–13282, <ext-link xlink:href="https://doi.org/10.5194/acp-17-13265-2017" ext-link-type="DOI">10.5194/acp-17-13265-2017</ext-link>, 2017.</mixed-citation></ref>
      <?pagebreak page1276?><ref id="bib1.bibx13"><?xmltex \def\ref@label{{Daellenbach et al.(2020)}}?><label>Daellenbach et al.(2020)</label><?label Daellenbach2020?><mixed-citation> Daellenbach, K. R., Uzu, G., Jiang, J., Cassagnes, L.-E., Leni, Z., Vlachou,  A., Stefenelli, G., Canonaco, F., Weber, S., Segers, A., Kuenen, J. J. P.,  Schaap, M., Favez, O., Albinet, A., Aksoyoglu, S., Dommen, J., Baltensperger,  U., Geiser, M., El Haddad, I., Jaffrezo, J.-L., and Prévôt, A. S. H.:  Sources of particulate-matter air pollution and its oxidative potential in  Europe, Nature, 587, 414–419, 2020.</mixed-citation></ref>
      <ref id="bib1.bibx14"><?xmltex \def\ref@label{{Elser et al.(2016)}}?><label>Elser et al.(2016)</label><?label Elser2016?><mixed-citation>Elser, M., Huang, R.-J., Wolf, R., Slowik, J. G., Wang, Q., Canonaco, F., Li, G., Bozzetti, C., Daellenbach, K. R., Huang, Y., Zhang, R., Li, Z., Cao, J., Baltensperger, U., El-Haddad, I., and Prévôt, A. S. H.: New insights into PM<inline-formula><mml:math id="M287" display="inline"><mml:msub><mml:mi/><mml:mn mathvariant="normal">2.5</mml:mn></mml:msub></mml:math></inline-formula> chemical composition and sources in two major cities in China during extreme haze events using aerosol mass spectrometry, Atmos. Chem. Phys., 16, 3207–3225, <ext-link xlink:href="https://doi.org/10.5194/acp-16-3207-2016" ext-link-type="DOI">10.5194/acp-16-3207-2016</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx15"><?xmltex \def\ref@label{{Fröhlich et al.(2013)}}?><label>Fröhlich et al.(2013)</label><?label Frohlich2013?><mixed-citation>Fröhlich, R., Cubison, M. J., Slowik, J. G., Bukowiecki, N., Prévôt, A. S. H., Baltensperger, U., Schneider, J., Kimmel, J. R., Gonin, M., Rohner, U., Worsnop, D. R., and Jayne, J. T.: The ToF-ACSM: a portable aerosol chemical speciation monitor with TOFMS detection, Atmos. Meas. Tech., 6, 3225–3241, <ext-link xlink:href="https://doi.org/10.5194/amt-6-3225-2013" ext-link-type="DOI">10.5194/amt-6-3225-2013</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bibx16"><?xmltex \def\ref@label{{Gelman et al.(2014)}}?><label>Gelman et al.(2014)</label><?label BDA?><mixed-citation> Gelman, A., Carlin, J. B., Stern, H. S., Dunson, D. B., Vehtari, A., and  Rubin, D. B.: Bayesian data analysis, 3rd edn., CRC Press, ISBN 9781439898208, 2014.</mixed-citation></ref>
      <ref id="bib1.bibx17"><?xmltex \def\ref@label{{Heikkinen et al.(2021)}}?><label>Heikkinen et al.(2021)</label><?label Heikkinen2021?><mixed-citation>Heikkinen, L., Äijälä, M., Daellenbach, K. R., Chen, G., Garmash, O., Aliaga, D., Graeffe, F., Räty, M., Luoma, K., Aalto, P., Kulmala, M., Petäjä, T., Worsnop, D., and Ehn, M.: Eight years of sub-micrometre organic aerosol composition data from the boreal forest characterized using a machine-learning approach, Atmos. Chem. Phys., 21, 10081–10109, <ext-link xlink:href="https://doi.org/10.5194/acp-21-10081-2021" ext-link-type="DOI">10.5194/acp-21-10081-2021</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bibx18"><?xmltex \def\ref@label{{Hirtzel et al.(1982)}}?><label>Hirtzel et al.(1982)</label><?label hirtzel1982estimating?><mixed-citation> Hirtzel, C., Corotis, R., and Quon, J.: Estimating the maximum value of  autocorrelated air quality measurements, Atmospheric Environment (1967), 16,  2603–2608, 1982.</mixed-citation></ref>
      <ref id="bib1.bibx19"><?xmltex \def\ref@label{{Hoffman and Gelman(2014)}}?><label>Hoffman and Gelman(2014)</label><?label NUTS?><mixed-citation> Hoffman, M. D. and Gelman, A.: The No-U-Turn sampler: adaptively setting path  lengths in Hamiltonian Monte Carlo, J. Mach. Learn. Res., 15, 1593–1623, 2014.</mixed-citation></ref>
      <ref id="bib1.bibx20"><?xmltex \def\ref@label{{Hopke(2016)}}?><label>Hopke(2016)</label><?label Hopke_pmf_amount?><mixed-citation>Hopke, P. K.: Review of receptor modeling methods for source apportionment,  J. Air Waste Manage., 66, 237–259, <ext-link xlink:href="https://doi.org/10.1080/10962247.2016.1140693" ext-link-type="DOI">10.1080/10962247.2016.1140693</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx21"><?xmltex \def\ref@label{{Huang et al.(2019)}}?><label>Huang et al.(2019)</label><?label Huang2019?><mixed-citation>Huang, R.-J., Wang, Y., Cao, J., Lin, C., Duan, J., Chen, Q., Li, Y., Gu, Y., Yan, J., Xu, W., Fröhlich, R., Canonaco, F., Bozzetti, C., Ovadnevaite, J., Ceburnis, D., Canagaratna, M. R., Jayne, J., Worsnop, D. R., El-Haddad, I., Prévôt, A. S. H., and O'Dowd, C. D.: Primary emissions versus secondary formation of fine particulate matter in the most polluted city (Shijiazhuang) in North China, Atmos. Chem. Phys., 19, 2283–2298, <ext-link xlink:href="https://doi.org/10.5194/acp-19-2283-2019" ext-link-type="DOI">10.5194/acp-19-2283-2019</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bibx22"><?xmltex \def\ref@label{{IPCC(2023)}}?><label>IPCC(2023)</label><?label IPCC2021?><mixed-citation>IPCC (Intergovernmental Panel on Climate Change): Climate Change 2021 – The Physical Science Basis: Working Group I Contribution to the Sixth Assessment Report of the Intergovernmental Panel on Climate Change, Cambridge University Press, Cambridge, <ext-link xlink:href="https://doi.org/10.1017/9781009157896" ext-link-type="DOI">10.1017/9781009157896</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bibx23"><?xmltex \def\ref@label{{Isokääntä et al.(2020)}}?><label>Isokääntä et al.(2020)</label><?label isokaanta?><mixed-citation>Isokääntä, S., Kari, E., Buchholz, A., Hao, L., Schobesberger, S., Virtanen, A., and Mikkonen, S.: Comparison of dimension reduction techniques in the analysis of mass spectrometry data, Atmos. Meas. Tech., 13, 2995–3022, <ext-link xlink:href="https://doi.org/10.5194/amt-13-2995-2020" ext-link-type="DOI">10.5194/amt-13-2995-2020</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bibx24"><?xmltex \def\ref@label{{Jiang et al.(2019)}}?><label>Jiang et al.(2019)</label><?label CAMx?><mixed-citation>Jiang, J., Aksoyoglu, S., El-Haddad, I., Ciarelli, G., Denier van der Gon, H. A. C., Canonaco, F., Gilardoni, S., Paglione, M., Minguillón, M. C., Favez, O., Zhang, Y., Marchand, N., Hao, L., Virtanen, A., Florou, K., O'Dowd, C., Ovadnevaite, J., Baltensperger, U., and Prévôt, A. S. H.: Sources of organic aerosols in Europe: a modeling study using CAMx with modified volatility basis set scheme, Atmos. Chem. Phys., 19, 15247–15270, <ext-link xlink:href="https://doi.org/10.5194/acp-19-15247-2019" ext-link-type="DOI">10.5194/acp-19-15247-2019</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bibx25"><?xmltex \def\ref@label{{Kuhn(1955)}}?><label>Kuhn(1955)</label><?label kuhn?><mixed-citation> Kuhn, H. W.: The Hungarian method for the assignment problem, Naval Res.  Logist. Q., 2, 83–97, 1955.</mixed-citation></ref>
      <ref id="bib1.bibx26"><?xmltex \def\ref@label{{Kulmala et al.(2021)}}?><label>Kulmala et al.(2021)</label><?label KulmalaFaraday?><mixed-citation> Kulmala, M., Dada, L., Daellenbach, K. R., Yan, C., Stolzenburg, D., Kontkanen, J., Ezhova, E., Hakala, S., Tuovinen, S., Kokkonen, T. V., Kurppa, M., Cai, R., Zhou, Y., Yin, R., Baalbaki, R., Chan, T., Chu, B., Deng, C., Fu, Y., Ge, M., He, H., Heikkinen, L., Junninen, H., Liu, Y., Lu, Y., Nie, W., Rusanen, A., Vakkari, V., Wang, Y., Yang, G., Yao, L., Zheng, J., Kujansuu, J., Kangasluoma, J., Petäjä, T., Paasonen, P., Järvi, L., Worsnop, D., Ding, A., Liu, Y., Wang, L., Jiang, J., Bianchi, F., and Kerminen, V.-M.: Is reducing new particle formation a plausible solution to mitigate particulate air pollution in Beijing and other Chinese megacities?, Faraday Discuss., 226, 334–347, 2021.</mixed-citation></ref>
      <ref id="bib1.bibx27"><?xmltex \def\ref@label{{Lelieveld et al.(2015)}}?><label>Lelieveld et al.(2015)</label><?label Lelieveld2015?><mixed-citation> Lelieveld, J., Evans, J. S., Fnais, M., Giannadaki, D., and Pozzer, A.: The  contribution of outdoor air pollution sources to premature mortality on a  global scale, Nature, 525, 367–371, 2015.</mixed-citation></ref>
      <ref id="bib1.bibx28"><?xmltex \def\ref@label{{Liu and Nocedal(1989)}}?><label>Liu and Nocedal(1989)</label><?label liu1989limited?><mixed-citation>Liu, D. C. and Nocedal, J.: On the Limited Memory BFGS Method for Large  Scale Optimization, Math. Program., 45, 503–528,  <ext-link xlink:href="https://doi.org/10.1007/BF01589116" ext-link-type="DOI">10.1007/BF01589116</ext-link>, 1989.</mixed-citation></ref>
      <ref id="bib1.bibx29"><?xmltex \def\ref@label{{Ng et al.(2011)}}?><label>Ng et al.(2011)</label><?label Ng2011?><mixed-citation> Ng, N. L., Herndon, S. C., Trimborn, A., Canagaratna, M. R., Croteau, P. L.,  Onasch, T. B., Sueper, D., Worsnop, D. R., Zhang, Q., Sun, Y. L., and Jayne,  J. T.: An Aerosol Chemical Speciation Monitor (ACSM) for Routine Monitoring  of the Composition and Mass Concentrations of Ambient Aerosol, Aerosol Sci. Tech., 45, 780–794, 2011.</mixed-citation></ref>
      <ref id="bib1.bibx30"><?xmltex \def\ref@label{{Paatero and Tapper(1994)}}?><label>Paatero and Tapper(1994)</label><?label paatero1994?><mixed-citation> Paatero, P. and Tapper, U.: Positive matrix factorization: A non-negative  factor model with optimal utilization of error estimates of data values,  Environmetrics, 5, 111–126, 1994.</mixed-citation></ref>
      <ref id="bib1.bibx31"><?xmltex \def\ref@label{{Reyes-Villegas et al.(2016)}}?><label>Reyes-Villegas et al.(2016)</label><?label ReyesVillegas2016?><mixed-citation>Reyes-Villegas, E., Green, D. C., Priestman, M., Canonaco, F., Coe, H., Reyes-Villegas, E., Green, D. C., Priestman, M., Canonaco, F., Coe, H., Prévôt, A. S. H., and Allan, J. D.: Organic aerosol source apportionment in London 2013 with ME-2: exploring the solution space with annual and seasonal analysis, Atmos. Chem. Phys., 16, 15545–15559, <ext-link xlink:href="https://doi.org/10.5194/acp-16-15545-2016" ext-link-type="DOI">10.5194/acp-16-15545-2016</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx32"><?xmltex \def\ref@label{Rusanen et al.(2024a)}?><label>Rusanen et al.(2024a)</label><?label Rusanenetal2024a?><mixed-citation>Rusanen, A., Björklund, A., Manousakas, M. I., Jiang, J., Kulmala, M. T., Puolamäki, K., and Daellenbach, K. R.: Bayesian auto-correlated matrix factorization, datasets, Version 1.0.0, Zenodo [data set], <ext-link xlink:href="https://doi.org/10.5281/zenodo.10629577" ext-link-type="DOI">10.5281/zenodo.10629577</ext-link>, 2024a.</mixed-citation></ref>
      <ref id="bib1.bibx33"><?xmltex \def\ref@label{Rusanen et al.(2024b)}?><label>Rusanen et al.(2024b)</label><?label Rusanenetal2024b?><mixed-citation>Rusanen, A., Björklund, A., Manousakas, M. I., Jiang, J., Kulmala, M. T., Puolamäki, K., and Daellenbach, K. R.: Bayesian auto-correlated matrix factorization, software, Version 1.0.0, Zenodo [code], <ext-link xlink:href="https://doi.org/10.5281/zenodo.10629849" ext-link-type="DOI">10.5281/zenodo.10629849</ext-link>, 2024b.</mixed-citation></ref>
      <ref id="bib1.bibx34"><?xmltex \def\ref@label{{Sage et al.(2008)}}?><label>Sage et al.(2008)</label><?label Sage2008?><mixed-citation>Sage, A. M., Weitkamp, E. A., Robinson, A. L., and Donahue, N. M.: Evolving mass spectra of the oxidized component of organic aerosol: results from aerosol mass spectrometer analyses of aged diesel emissions, Atmos. Chem. Phys., 8, 1139–1152, <ext-link xlink:href="https://doi.org/10.5194/acp-8-1139-2008" ext-link-type="DOI">10.5194/acp-8-1139-2008</ext-link>, 2008.</mixed-citation></ref>
      <ref id="bib1.bibx35"><?xmltex \def\ref@label{{Schlag et al.(2017)}}?><label>Schlag et al.(2017)</label><?label Schlag2017?><mixed-citation> Schlag, P., Rubach, F., Mentel, T. F., Reimer, D., Canonaco, F., Henzing,  J. S., Moerman, M., Otjes, R., Prévôt, A. S. H., Rohrer, F., Rosati, B.,  Tillmann, R., Weingartner, E., and Kiendler-Scharr, A.: Ambient and  laboratory observations of organic ammonium salts in PM1, Faraday Discuss., 200, 331–351, 2017.</mixed-citation></ref>
      <ref id="bib1.bibx36"><?xmltex \def\ref@label{{Ulbrich et al.(2022)}}?><label>Ulbrich et al.(2022)</label><?label Ulbrich_web?><mixed-citation>Ulbrich, I., Handschy, A., Lechner, M., and Jimenez, J.: AMS Spectral Database, <uri>http://cires.colorado.edu/jimenez-group/AMSsd/</uri> (last access: 25 April 2022), 2022.</mixed-citation></ref>
      <?pagebreak page1277?><ref id="bib1.bibx37"><?xmltex \def\ref@label{{Ulbrich et al.(2009)}}?><label>Ulbrich et al.(2009)</label><?label Ulbrich2009?><mixed-citation>Ulbrich, I. M., Canagaratna, M. R., Zhang, Q., Worsnop, D. R., and Jimenez, J. L.: Interpretation of organic components from Positive Matrix Factorization of aerosol mass spectrometric data, Atmos. Chem. Phys., 9, 2891–2918, <ext-link xlink:href="https://doi.org/10.5194/acp-9-2891-2009" ext-link-type="DOI">10.5194/acp-9-2891-2009</ext-link>, 2009.</mixed-citation></ref>
      <ref id="bib1.bibx38"><?xmltex \def\ref@label{{Wang and Zhang(2012)}}?><label>Wang and Zhang(2012)</label><?label wangreview?><mixed-citation> Wang, Y.-X. and Zhang, Y.-J.: Nonnegative matrix factorization: A comprehensive review, IEEE T. Knowl. Data En., 25, 1336–1353, 2012.</mixed-citation></ref>
      <ref id="bib1.bibx39"><?xmltex \def\ref@label{{Watson et al.(2001)}}?><label>Watson et al.(2001)</label><?label CMBreview?><mixed-citation>Watson, J. G., Chow, J. C., and Fujita, E. M.: Review of volatile organic  compound source apportionment by chemical mass balance, Atmos. Environ., 35, 1567–1584, <ext-link xlink:href="https://doi.org/10.1016/S1352-2310(00)00461-1" ext-link-type="DOI">10.1016/S1352-2310(00)00461-1</ext-link>, 2001.</mixed-citation></ref>
      <ref id="bib1.bibx40"><?xmltex \def\ref@label{{Zhang et al.(2019)}}?><label>Zhang et al.(2019)</label><?label Zhang2019?><mixed-citation> Zhang, J., Li, R., Zhang, X., Bai, Y., Cao, P., and Hua, P.: Vehicular  contribution of PAHs in size dependent road dust: A source apportionment by  PCA-MLR, PMF, and Unmix receptor models, Science Total Environ., 649, 1314–1322, 2019.</mixed-citation></ref>
      <ref id="bib1.bibx41"><?xmltex \def\ref@label{{Zhang et al.(2007)}}?><label>Zhang et al.(2007)</label><?label Zhang2007?><mixed-citation>Zhang, Q., Jimenez, J. L., Canagaratna, M. R., Allan, J. D., Coe, H., Ulbrich, I., Alfarra, M. R., Takami, A., Middlebrook, A. M., Sun, Y. L., Dzepina, K., Dunlea, E., Docherty, K., DeCarlo, P. F., Salcedo, D., Onasch, T., Jayne, J. T., Miyoshi, T., Shimono, A., Hatakeyama, S., Takegawa, N., Kondo, Y., Schneider, J., Drewnick, F., Borrmann, S., Weimer, S., Demerjian, K., Williams, P., Bower, K., Bahreini, R., Cottrell, L., Griffin, R. J.,  Rautiainen, J., Sun, J. Y., Zhang, Y. M., and Worsnop, D. R.: Ubiquity and  dominance of oxygenated species in organic aerosols in  anthropogenically-influenced Northern Hemisphere midlatitudes, Geophys. Res. Lett., 34, L13801, <ext-link xlink:href="https://doi.org/10.1029/2007GL029979" ext-link-type="DOI">10.1029/2007GL029979</ext-link>, 2007. </mixed-citation></ref><?xmltex \hack{\newpage}?>
      <ref id="bib1.bibx42"><?xmltex \def\ref@label{{Zhang et al.(2011)}}?><label>Zhang et al.(2011)</label><?label Zhang2011?><mixed-citation> Zhang, Q., Jimenez, J. L., Canagaratna, M. R., Ulbrich, I. M., Ng, N. L.,  Worsnop, D. R., and Sun, Y.: Understanding atmospheric organic aerosols via  factor analysis of aerosol mass spectrometry: a review, Anal. Bioanal. Chem., 401, 3045–3067, 2011.</mixed-citation></ref>
      <ref id="bib1.bibx43"><?xmltex \def\ref@label{{Zhang et al.(2018)}}?><label>Zhang et al.(2018)</label><?label Zhang2018?><mixed-citation>Zhang, Y., Favez, O., Canonaco, F., Liu, D., Močnik, G., Amodeo, T., Sciare,  J., Prévôt, A. S. H., Gros, V., and Albinet, A.: Evidence of major secondary organic aerosol contribution to lensing effect black carbon absorption enhancement, npj Climate and Atmospheric Science, 1, 47, <ext-link xlink:href="https://doi.org/10.1038/s41612-018-0056-2" ext-link-type="DOI">10.1038/s41612-018-0056-2</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bibx44"><?xmltex \def\ref@label{{Zhu et al.(2018)}}?><label>Zhu et al.(2018)</label><?label Zhu2018?><mixed-citation>Zhu, Q., Huang, X.-F., Cao, L.-M., Wei, L.-T., Zhang, B., He, L.-Y., Elser, M., Canonaco, F., Slowik, J. G., Bozzetti, C., El-Haddad, I., and Prévôt, A. S. H.: Improved source apportionment of organic aerosols in complex urban air pollution using the multilinear engine (ME-2), Atmos. Meas. Tech., 11, 1049–1060, <ext-link xlink:href="https://doi.org/10.5194/amt-11-1049-2018" ext-link-type="DOI">10.5194/amt-11-1049-2018</ext-link>, 2018.</mixed-citation></ref>

  </ref-list></back>
    <!--<article-title-html>A novel probabilistic source apportionment approach: Bayesian auto-correlated matrix factorization</article-title-html>
<abstract-html/>
<ref-html id="bib1.bib1"><label>Allan et al.(2003)</label><mixed-citation>
      
Allan, J. D., Jimenez, J. L., Williams, P. I., Alfarra, M. R., Bower, K. N.,  Jayne, J. T., Coe, H., and Worsnop, D. R.: Quantitative sampling using an  Aerodyne aerosol mass spectrometer 1. Techniques of data interpretation and  error analysis, J. Geophys. Res.-Atmos., 108, 4090, <a href="https://doi.org/10.1029/2002JD002358" target="_blank">https://doi.org/10.1029/2002JD002358</a>, 2003.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib2"><label>Bates et al.(2019)</label><mixed-citation>
      
Bates, J. T., Fang, T., Verma, V., Zeng, L., Weber, R. J., Tolbert, P. E.,  Abrams, J. Y., Sarnat, S. E., Klein, M., Mulholland, J. A., and Russell,  A. G.: Review of Acellular Assays of Ambient Particulate Matter Oxidative  Potential: Methods and Relationships with Composition, Sources, and Health  Effects, Environ. Sci. Technol., 53, 4003–4019,  <a href="https://doi.org/10.1021/acs.est.8b03430" target="_blank">https://doi.org/10.1021/acs.est.8b03430</a>, 2019.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib3"><label>Canagaratna et al.(2007)</label><mixed-citation>
      
Canagaratna, M., Jayne, J., Jimenez, J., Allan, J., Alfarra, M., Zhang, Q.,  Onasch, T., Drewnick, F., Coe, H., Middlebrook, A., Delia, A., Williams, L.,  Trimborn, A., Northway, M., DeCarlo, P., Kolb, C., Davidovits, P., and  Worsnop, D.: Chemical and microphysical characterization of ambient aerosols  with the aerodyne aerosol mass spectrometer, Mass Spectrom. Rev., 26,  185–222, 2007.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib4"><label>Canonaco et al.(2013)</label><mixed-citation>
      
Canonaco, F., Crippa, M., Slowik, J. G., Baltensperger, U., and Prévôt, A. S. H.: SoFi, an IGOR-based interface for the efficient use of the generalized multilinear engine (ME-2) for the source apportionment: ME-2 application to aerosol mass spectrometer data, Atmos. Meas. Tech., 6, 3649–3661, <a href="https://doi.org/10.5194/amt-6-3649-2013" target="_blank">https://doi.org/10.5194/amt-6-3649-2013</a>, 2013.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib5"><label>Canonaco et al.(2021)</label><mixed-citation>
      
Canonaco, F., Tobler, A., Chen, G., Sosedova, Y., Slowik, J. G., Bozzetti, C., Daellenbach, K. R., El Haddad, I., Crippa, M., Huang, R.-J., Furger, M., Baltensperger, U., and Prévôt, A. S. H.: A new method for long-term source apportionment with time-dependent factor profiles and uncertainty assessment using SoFi Pro: application to 1 year of organic aerosol data, Atmos. Meas. Tech., 14, 923–943, <a href="https://doi.org/10.5194/amt-14-923-2021" target="_blank">https://doi.org/10.5194/amt-14-923-2021</a>, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib6"><label>Carpenter et al.(2017)</label><mixed-citation>
      
Carpenter, B., Gelman, A., Hoffman, M., Lee, D., Goodrich, B., Betancourt, M., Brubaker, M., Guo, J., Li, P., and Riddell, A.: Stan: A Probabilistic  Programming Language, J. Stat. Softw., 76, 1–32,  <a href="https://doi.org/10.18637/jss.v076.i01" target="_blank">https://doi.org/10.18637/jss.v076.i01</a>, 2017.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib7"><label>Chazeau et al.(2022)</label><mixed-citation>
      
Chazeau, B., El Haddad, I., Canonaco, F., Temime-Roussel, B., D'Anna, B.,  Gille, G., Mesbah, B., Prévôt, A. S., Wortham, H., and Marchand, N.: Organic aerosol source apportionment by using rolling positive matrix factorization: Application to a Mediterranean coastal city, Atmospheric Environment: X, 14, 100176, <a href="https://doi.org/10.1016/j.aeaoa.2022.100176" target="_blank">https://doi.org/10.1016/j.aeaoa.2022.100176</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib8"><label>Chen(2022)</label><mixed-citation>
      
Chen, G.: European Aerosol Phenomenology - 8: Harmonised Source Apportionment of Organic Aerosol using 22 Year-long ACSM/AMS Datasets, Version 2nd, Zenodo [data set], <a href="https://doi.org/10.5281/zenodo.6672710" target="_blank">https://doi.org/10.5281/zenodo.6672710</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib9"><label>Chen et al.(2022)</label><mixed-citation>
      
Chen, G., Canonaco, F., Tobler, A., Aas, W., Alastuey, A., Allan, J.,  Atabakhsh, S., Aurela, M., Baltensperger, U., Bougiatioti, A., De Brito,  J. F., Ceburnis, D., Chazeau, B., Chebaicheb, H., Daellenbach, K. R., Ehn,  M., El Haddad, I., Eleftheriadis, K., Favez, O., Flentje, H., Font, A.,  Fossum, K., Freney, E., Gini, M., Green, D. C., Heikkinen, L., Herrmann, H.,  Kalogridis, A.-C., Keernik, H., Lhotka, R., Lin, C., Lunder, C., Maasikmets,  M., Manousakas, M. I., Marchand, N., Marin, C., Marmureanu, L., Mihalopoulos,  N., Močnik, G., Nęcki, J., O'Dowd, C., Ovadnevaite, J., Peter, T., Petit,  J.-E., Pikridas, M., Matthew Platt, S., Pokorná, P., Poulain, L.,  Priestman, M., Riffault, V., Rinaldi, M., Różański, K., Schwarz, J., Sciare, J., Simon, L., Skiba, A., Slowik, J. G., Sosedova, Y., Stavroulas, I., Styszko, K., Teinemaa, E., Timonen, H., Tremper, A., Vasilescu, J., Via, M., Vodička, P., Wiedensohler, A., Zografou, O., Cruz Minguillón, M., and Prévôt, A. S.: European aerosol phenomenology - 8: Harmonised source apportionment of organic aerosol using 22 Year-long ACSM/AMS datasets,  Environ. Int., 166, 107325, <a href="https://doi.org/10.1016/j.envint.2022.107325" target="_blank">https://doi.org/10.1016/j.envint.2022.107325</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib10"><label>Crippa et al.(2013)</label><mixed-citation>
      
Crippa, M., DeCarlo, P. F., Slowik, J. G., Mohr, C., Heringa, M. F., Chirico, R., Poulain, L., Freutel, F., Sciare, J., Cozic, J., Di Marco, C. F., Elsasser, M., Nicolas, J. B., Marchand, N., Abidi, E., Wiedensohler, A., Drewnick, F., Schneider, J., Borrmann, S., Nemitz, E., Zimmermann, R., Jaffrezo, J.-L., Prévôt, A. S. H., and Baltensperger, U.: Wintertime aerosol chemical composition and source apportionment of the organic fraction in the metropolitan area of Paris, Atmos. Chem. Phys., 13, 961–981, <a href="https://doi.org/10.5194/acp-13-961-2013" target="_blank">https://doi.org/10.5194/acp-13-961-2013</a>, 2013.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib11"><label>Crippa et al.(2014)</label><mixed-citation>
      
Crippa, M., Canonaco, F., Lanz, V. A., Äijälä, M., Allan, J. D., Carbone, S., Capes, G., Ceburnis, D., Dall'Osto, M., Day, D. A., DeCarlo, P. F., Ehn, M., Eriksson, A., Freney, E., Hildebrandt Ruiz, L., Hillamo, R., Jimenez, J. L., Junninen, H., Kiendler-Scharr, A., Kortelainen, A.-M., Kulmala, M., Laaksonen, A., Mensah, A. A., Mohr, C., Nemitz, E., O'Dowd, C., Ovadnevaite, J., Pandis, S. N., Petäjä, T., Poulain, L., Saarikoski, S., Sellegri, K., Swietlicki, E., Tiitta, P., Worsnop, D. R., Baltensperger, U., and Prévôt, A. S. H.: Organic aerosol components derived from 25 AMS data sets across Europe using a consistent ME-2 based source apportionment approach, Atmos. Chem. Phys., 14, 6159–6176, <a href="https://doi.org/10.5194/acp-14-6159-2014" target="_blank">https://doi.org/10.5194/acp-14-6159-2014</a>, 2014.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib12"><label>Daellenbach et al.(2017)</label><mixed-citation>
      
Daellenbach, K. R., Stefenelli, G., Bozzetti, C., Vlachou, A., Fermo, P., Gonzalez, R., Piazzalunga, A., Colombi, C., Canonaco, F., Hueglin, C., Kasper-Giebl, A., Jaffrezo, J.-L., Bianchi, F., Slowik, J. G., Baltensperger, U., El-Haddad, I., and Prévôt, A. S. H.: Long-term chemical analysis and organic aerosol source apportionment at nine sites in central Europe: source identification and uncertainty assessment, Atmos. Chem. Phys., 17, 13265–13282, <a href="https://doi.org/10.5194/acp-17-13265-2017" target="_blank">https://doi.org/10.5194/acp-17-13265-2017</a>, 2017.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib13"><label>Daellenbach et al.(2020)</label><mixed-citation>
      
Daellenbach, K. R., Uzu, G., Jiang, J., Cassagnes, L.-E., Leni, Z., Vlachou,  A., Stefenelli, G., Canonaco, F., Weber, S., Segers, A., Kuenen, J. J. P.,  Schaap, M., Favez, O., Albinet, A., Aksoyoglu, S., Dommen, J., Baltensperger,  U., Geiser, M., El Haddad, I., Jaffrezo, J.-L., and Prévôt, A. S. H.:  Sources of particulate-matter air pollution and its oxidative potential in  Europe, Nature, 587, 414–419, 2020.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib14"><label>Elser et al.(2016)</label><mixed-citation>
      
Elser, M., Huang, R.-J., Wolf, R., Slowik, J. G., Wang, Q., Canonaco, F., Li, G., Bozzetti, C., Daellenbach, K. R., Huang, Y., Zhang, R., Li, Z., Cao, J., Baltensperger, U., El-Haddad, I., and Prévôt, A. S. H.: New insights into PM<sub>2.5</sub> chemical composition and sources in two major cities in China during extreme haze events using aerosol mass spectrometry, Atmos. Chem. Phys., 16, 3207–3225, <a href="https://doi.org/10.5194/acp-16-3207-2016" target="_blank">https://doi.org/10.5194/acp-16-3207-2016</a>, 2016.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib15"><label>Fröhlich et al.(2013)</label><mixed-citation>
      
Fröhlich, R., Cubison, M. J., Slowik, J. G., Bukowiecki, N., Prévôt, A. S. H., Baltensperger, U., Schneider, J., Kimmel, J. R., Gonin, M., Rohner, U., Worsnop, D. R., and Jayne, J. T.: The ToF-ACSM: a portable aerosol chemical speciation monitor with TOFMS detection, Atmos. Meas. Tech., 6, 3225–3241, <a href="https://doi.org/10.5194/amt-6-3225-2013" target="_blank">https://doi.org/10.5194/amt-6-3225-2013</a>, 2013.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib16"><label>Gelman et al.(2014)</label><mixed-citation>
      
Gelman, A., Carlin, J. B., Stern, H. S., Dunson, D. B., Vehtari, A., and  Rubin, D. B.: Bayesian data analysis, 3rd edn., CRC Press, ISBN&thinsp;9781439898208, 2014.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib17"><label>Heikkinen et al.(2021)</label><mixed-citation>
      
Heikkinen, L., Äijälä, M., Daellenbach, K. R., Chen, G., Garmash, O., Aliaga, D., Graeffe, F., Räty, M., Luoma, K., Aalto, P., Kulmala, M., Petäjä, T., Worsnop, D., and Ehn, M.: Eight years of sub-micrometre organic aerosol composition data from the boreal forest characterized using a machine-learning approach, Atmos. Chem. Phys., 21, 10081–10109, <a href="https://doi.org/10.5194/acp-21-10081-2021" target="_blank">https://doi.org/10.5194/acp-21-10081-2021</a>, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib18"><label>Hirtzel et al.(1982)</label><mixed-citation>
      
Hirtzel, C., Corotis, R., and Quon, J.: Estimating the maximum value of  autocorrelated air quality measurements, Atmospheric Environment (1967), 16,  2603–2608, 1982.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib19"><label>Hoffman and Gelman(2014)</label><mixed-citation>
      
Hoffman, M. D. and Gelman, A.: The No-U-Turn sampler: adaptively setting path  lengths in Hamiltonian Monte Carlo, J. Mach. Learn. Res., 15, 1593–1623, 2014.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib20"><label>Hopke(2016)</label><mixed-citation>
      
Hopke, P. K.: Review of receptor modeling methods for source apportionment,  J. Air Waste Manage., 66, 237–259, <a href="https://doi.org/10.1080/10962247.2016.1140693" target="_blank">https://doi.org/10.1080/10962247.2016.1140693</a>, 2016.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib21"><label>Huang et al.(2019)</label><mixed-citation>
      
Huang, R.-J., Wang, Y., Cao, J., Lin, C., Duan, J., Chen, Q., Li, Y., Gu, Y., Yan, J., Xu, W., Fröhlich, R., Canonaco, F., Bozzetti, C., Ovadnevaite, J., Ceburnis, D., Canagaratna, M. R., Jayne, J., Worsnop, D. R., El-Haddad, I., Prévôt, A. S. H., and O'Dowd, C. D.: Primary emissions versus secondary formation of fine particulate matter in the most polluted city (Shijiazhuang) in North China, Atmos. Chem. Phys., 19, 2283–2298, <a href="https://doi.org/10.5194/acp-19-2283-2019" target="_blank">https://doi.org/10.5194/acp-19-2283-2019</a>, 2019.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib22"><label>IPCC(2023)</label><mixed-citation>
      
IPCC (Intergovernmental Panel on Climate Change): Climate Change 2021 – The Physical Science Basis: Working Group I Contribution to the Sixth Assessment Report of the Intergovernmental Panel on Climate Change, Cambridge University Press, Cambridge, <a href="https://doi.org/10.1017/9781009157896" target="_blank">https://doi.org/10.1017/9781009157896</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib23"><label>Isokääntä et al.(2020)</label><mixed-citation>
      
Isokääntä, S., Kari, E., Buchholz, A., Hao, L., Schobesberger, S., Virtanen, A., and Mikkonen, S.: Comparison of dimension reduction techniques in the analysis of mass spectrometry data, Atmos. Meas. Tech., 13, 2995–3022, <a href="https://doi.org/10.5194/amt-13-2995-2020" target="_blank">https://doi.org/10.5194/amt-13-2995-2020</a>, 2020.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib24"><label>Jiang et al.(2019)</label><mixed-citation>
      
Jiang, J., Aksoyoglu, S., El-Haddad, I., Ciarelli, G., Denier van der Gon, H. A. C., Canonaco, F., Gilardoni, S., Paglione, M., Minguillón, M. C., Favez, O., Zhang, Y., Marchand, N., Hao, L., Virtanen, A., Florou, K., O'Dowd, C., Ovadnevaite, J., Baltensperger, U., and Prévôt, A. S. H.: Sources of organic aerosols in Europe: a modeling study using CAMx with modified volatility basis set scheme, Atmos. Chem. Phys., 19, 15247–15270, <a href="https://doi.org/10.5194/acp-19-15247-2019" target="_blank">https://doi.org/10.5194/acp-19-15247-2019</a>, 2019.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib25"><label>Kuhn(1955)</label><mixed-citation>
      
Kuhn, H. W.: The Hungarian method for the assignment problem, Naval Res.  Logist. Q., 2, 83–97, 1955.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib26"><label>Kulmala et al.(2021)</label><mixed-citation>
      
Kulmala, M., Dada, L., Daellenbach, K. R., Yan, C., Stolzenburg, D., Kontkanen, J., Ezhova, E., Hakala, S., Tuovinen, S., Kokkonen, T. V., Kurppa, M., Cai, R., Zhou, Y., Yin, R., Baalbaki, R., Chan, T., Chu, B., Deng, C., Fu, Y., Ge, M., He, H., Heikkinen, L., Junninen, H., Liu, Y., Lu, Y., Nie, W., Rusanen, A., Vakkari, V., Wang, Y., Yang, G., Yao, L., Zheng, J., Kujansuu, J., Kangasluoma, J., Petäjä, T., Paasonen, P., Järvi, L., Worsnop, D., Ding, A., Liu, Y., Wang, L., Jiang, J., Bianchi, F., and Kerminen, V.-M.: Is reducing new particle formation a plausible solution to mitigate particulate air pollution in Beijing and other Chinese megacities?, Faraday Discuss., 226, 334–347, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib27"><label>Lelieveld et al.(2015)</label><mixed-citation>
      
Lelieveld, J., Evans, J. S., Fnais, M., Giannadaki, D., and Pozzer, A.: The  contribution of outdoor air pollution sources to premature mortality on a  global scale, Nature, 525, 367–371, 2015.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib28"><label>Liu and Nocedal(1989)</label><mixed-citation>
      
Liu, D. C. and Nocedal, J.: On the Limited Memory BFGS Method for Large  Scale Optimization, Math. Program., 45, 503–528,  <a href="https://doi.org/10.1007/BF01589116" target="_blank">https://doi.org/10.1007/BF01589116</a>, 1989.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib29"><label>Ng et al.(2011)</label><mixed-citation>
      
Ng, N. L., Herndon, S. C., Trimborn, A., Canagaratna, M. R., Croteau, P. L.,  Onasch, T. B., Sueper, D., Worsnop, D. R., Zhang, Q., Sun, Y. L., and Jayne,  J. T.: An Aerosol Chemical Speciation Monitor (ACSM) for Routine Monitoring  of the Composition and Mass Concentrations of Ambient Aerosol, Aerosol Sci. Tech., 45, 780–794, 2011.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib30"><label>Paatero and Tapper(1994)</label><mixed-citation>
      
Paatero, P. and Tapper, U.: Positive matrix factorization: A non-negative  factor model with optimal utilization of error estimates of data values,  Environmetrics, 5, 111–126, 1994.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib31"><label>Reyes-Villegas et al.(2016)</label><mixed-citation>
      
Reyes-Villegas, E., Green, D. C., Priestman, M., Canonaco, F., Coe, H.,
Reyes-Villegas, E., Green, D. C., Priestman, M., Canonaco, F., Coe, H., Prévôt, A. S. H., and Allan, J. D.: Organic aerosol source apportionment in London 2013 with ME-2: exploring the solution space with annual and seasonal analysis, Atmos. Chem. Phys., 16, 15545–15559, <a href="https://doi.org/10.5194/acp-16-15545-2016" target="_blank">https://doi.org/10.5194/acp-16-15545-2016</a>, 2016.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib32"><label>Rusanen et al.(2024a)</label><mixed-citation>
      
Rusanen, A., Björklund, A., Manousakas, M. I., Jiang, J., Kulmala, M. T., Puolamäki, K., and Daellenbach, K. R.: Bayesian auto-correlated matrix factorization, datasets, Version 1.0.0, Zenodo [data set], <a href="https://doi.org/10.5281/zenodo.10629577" target="_blank">https://doi.org/10.5281/zenodo.10629577</a>, 2024a.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib33"><label>Rusanen et al.(2024b)</label><mixed-citation>
      
Rusanen, A., Björklund, A., Manousakas, M. I., Jiang, J., Kulmala, M. T., Puolamäki, K., and Daellenbach, K. R.: Bayesian auto-correlated matrix factorization, software, Version 1.0.0, Zenodo [code], <a href="https://doi.org/10.5281/zenodo.10629849" target="_blank">https://doi.org/10.5281/zenodo.10629849</a>, 2024b.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib34"><label>Sage et al.(2008)</label><mixed-citation>
      
Sage, A. M., Weitkamp, E. A., Robinson, A. L., and Donahue, N. M.: Evolving mass spectra of the oxidized component of organic aerosol: results from aerosol mass spectrometer analyses of aged diesel emissions, Atmos. Chem. Phys., 8, 1139–1152, <a href="https://doi.org/10.5194/acp-8-1139-2008" target="_blank">https://doi.org/10.5194/acp-8-1139-2008</a>, 2008.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib35"><label>Schlag et al.(2017)</label><mixed-citation>
      
Schlag, P., Rubach, F., Mentel, T. F., Reimer, D., Canonaco, F., Henzing,  J. S., Moerman, M., Otjes, R., Prévôt, A. S. H., Rohrer, F., Rosati, B.,  Tillmann, R., Weingartner, E., and Kiendler-Scharr, A.: Ambient and  laboratory observations of organic ammonium salts in PM1, Faraday Discuss., 200, 331–351, 2017.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib36"><label>Ulbrich et al.(2022)</label><mixed-citation>
      
Ulbrich, I., Handschy, A., Lechner, M., and Jimenez, J.: AMS Spectral Database, <a href="http://cires.colorado.edu/jimenez-group/AMSsd/" target="_blank"/> (last access:
25 April 2022), 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib37"><label>Ulbrich et al.(2009)</label><mixed-citation>
      
Ulbrich, I. M., Canagaratna, M. R., Zhang, Q., Worsnop, D. R., and Jimenez, J. L.: Interpretation of organic components from Positive Matrix Factorization of aerosol mass spectrometric data, Atmos. Chem. Phys., 9, 2891–2918, <a href="https://doi.org/10.5194/acp-9-2891-2009" target="_blank">https://doi.org/10.5194/acp-9-2891-2009</a>, 2009.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib38"><label>Wang and Zhang(2012)</label><mixed-citation>
      
Wang, Y.-X. and Zhang, Y.-J.: Nonnegative matrix factorization: A comprehensive review, IEEE T. Knowl. Data En., 25, 1336–1353, 2012.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib39"><label>Watson et al.(2001)</label><mixed-citation>
      
Watson, J. G., Chow, J. C., and Fujita, E. M.: Review of volatile organic  compound source apportionment by chemical mass balance, Atmos. Environ., 35, 1567–1584, <a href="https://doi.org/10.1016/S1352-2310(00)00461-1" target="_blank">https://doi.org/10.1016/S1352-2310(00)00461-1</a>, 2001.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib40"><label>Zhang et al.(2019)</label><mixed-citation>
      
Zhang, J., Li, R., Zhang, X., Bai, Y., Cao, P., and Hua, P.: Vehicular  contribution of PAHs in size dependent road dust: A source apportionment by  PCA-MLR, PMF, and Unmix receptor models, Science Total Environ., 649, 1314–1322, 2019.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib41"><label>Zhang et al.(2007)</label><mixed-citation>
      
Zhang, Q., Jimenez, J. L., Canagaratna, M. R., Allan, J. D., Coe, H., Ulbrich, I., Alfarra, M. R., Takami, A., Middlebrook, A. M., Sun, Y. L., Dzepina, K., Dunlea, E., Docherty, K., DeCarlo, P. F., Salcedo, D., Onasch, T., Jayne, J. T., Miyoshi, T., Shimono, A., Hatakeyama, S., Takegawa, N., Kondo, Y., Schneider, J., Drewnick, F., Borrmann, S., Weimer, S., Demerjian, K., Williams, P., Bower, K., Bahreini, R., Cottrell, L., Griffin, R. J.,  Rautiainen, J., Sun, J. Y., Zhang, Y. M., and Worsnop, D. R.: Ubiquity and  dominance of oxygenated species in organic aerosols in  anthropogenically-influenced Northern Hemisphere midlatitudes, Geophys. Res. Lett., 34, L13801, <a href="https://doi.org/10.1029/2007GL029979" target="_blank">https://doi.org/10.1029/2007GL029979</a>, 2007.


    </mixed-citation></ref-html>
<ref-html id="bib1.bib42"><label>Zhang et al.(2011)</label><mixed-citation>
      
Zhang, Q., Jimenez, J. L., Canagaratna, M. R., Ulbrich, I. M., Ng, N. L.,  Worsnop, D. R., and Sun, Y.: Understanding atmospheric organic aerosols via  factor analysis of aerosol mass spectrometry: a review, Anal. Bioanal. Chem., 401, 3045–3067, 2011.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib43"><label>Zhang et al.(2018)</label><mixed-citation>
      
Zhang, Y., Favez, O., Canonaco, F., Liu, D., Močnik, G., Amodeo, T., Sciare,  J., Prévôt, A. S. H., Gros, V., and Albinet, A.: Evidence of major secondary organic aerosol contribution to lensing effect black carbon absorption enhancement, npj Climate and Atmospheric Science, 1, 47, <a href="https://doi.org/10.1038/s41612-018-0056-2" target="_blank">https://doi.org/10.1038/s41612-018-0056-2</a>, 2018.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib44"><label>Zhu et al.(2018)</label><mixed-citation>
      
Zhu, Q., Huang, X.-F., Cao, L.-M., Wei, L.-T., Zhang, B., He, L.-Y., Elser, M., Canonaco, F., Slowik, J. G., Bozzetti, C., El-Haddad, I., and Prévôt, A. S. H.: Improved source apportionment of organic aerosols in complex urban air pollution using the multilinear engine (ME-2), Atmos. Meas. Tech., 11, 1049–1060, <a href="https://doi.org/10.5194/amt-11-1049-2018" target="_blank">https://doi.org/10.5194/amt-11-1049-2018</a>, 2018.

    </mixed-citation></ref-html>--></article>
