<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing with OASIS Tables v3.0 20080202//EN" "journalpub-oasis3.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:oasis="http://docs.oasis-open.org/ns/oasis-exchange/table" xml:lang="en" dtd-version="3.0" article-type="data-paper">
  <front>
    <journal-meta><journal-id journal-id-type="publisher">ESSD</journal-id><journal-title-group>
    <journal-title>Earth System Science Data</journal-title>
    <abbrev-journal-title abbrev-type="publisher">ESSD</abbrev-journal-title><abbrev-journal-title abbrev-type="nlm-ta">Earth Syst. Sci. Data</abbrev-journal-title>
  </journal-title-group><issn pub-type="epub">1866-3516</issn><publisher>
    <publisher-name>Copernicus Publications</publisher-name>
    <publisher-loc>Göttingen, Germany</publisher-loc>
  </publisher></journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.5194/essd-13-4241-2021</article-id><title-group><article-title>An all-sky 1 km daily land surface air temperature product over mainland
China for 2003–2019 from <?xmltex \hack{\break}?> MODIS and ancillary data</article-title><alt-title>An all-sky 1 km daily land surface air temperature product</alt-title>
      </title-group><?xmltex \runningtitle{An all-sky 1\,km daily land surface air temperature product}?><?xmltex \runningauthor{Y.~Chen et al.}?>
      <contrib-group>
        <contrib contrib-type="author" corresp="no" rid="aff1">
          <name><surname>Chen</surname><given-names>Yan</given-names></name>
          
        <ext-link>https://orcid.org/0000-0002-4937-4818</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff2">
          <name><surname>Liang</surname><given-names>Shunlin</given-names></name>
          
        <ext-link>https://orcid.org/0000-0003-2708-9183</ext-link></contrib>
        <contrib contrib-type="author" corresp="yes" rid="aff1">
          <name><surname>Ma</surname><given-names>Han</given-names></name>
          <email>mahanrs@whu.edu.cn</email>
        <ext-link>https://orcid.org/0000-0002-1123-7447</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff1">
          <name><surname>Li</surname><given-names>Bing</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff1">
          <name><surname>He</surname><given-names>Tao</given-names></name>
          
        <ext-link>https://orcid.org/0000-0003-2079-7988</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff3">
          <name><surname>Wang</surname><given-names>Qian</given-names></name>
          
        <ext-link>https://orcid.org/0000-0002-7697-5168</ext-link></contrib>
        <aff id="aff1"><label>1</label><institution>School of Remote Sensing and Information Engineering, Wuhan
University, Wuhan 430079, China</institution>
        </aff>
        <aff id="aff2"><label>2</label><institution>Department of Geographical Sciences, University of Maryland, College
Park, MD 20742, USA</institution>
        </aff>
        <aff id="aff3"><label>3</label><institution>State Key Laboratory of Remote Sensing Science, Beijing Normal
University, Beijing 100875, China</institution>
        </aff>
      </contrib-group>
      <author-notes><corresp id="corr1">Han Ma (mahanrs@whu.edu.cn)</corresp></author-notes><pub-date><day>30</day><month>August</month><year>2021</year></pub-date>
      
      <volume>13</volume>
      <issue>8</issue>
      <fpage>4241</fpage><lpage>4261</lpage>
      <history>
        <date date-type="received"><day>27</day><month>January</month><year>2021</year></date>
           <date date-type="rev-request"><day>12</day><month>March</month><year>2021</year></date>
           <date date-type="rev-recd"><day>22</day><month>July</month><year>2021</year></date>
           <date date-type="accepted"><day>27</day><month>July</month><year>2021</year></date>
      </history>
      <permissions>
        <copyright-statement>Copyright: © 2021 </copyright-statement>
        <copyright-year>2021</copyright-year>
      <license license-type="open-access"><license-p>This work is licensed under the Creative Commons Attribution 4.0 International License. To view a copy of this licence, visit <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link></license-p></license></permissions><self-uri xlink:href="https://essd.copernicus.org/articles/.html">This article is available from https://essd.copernicus.org/articles/.html</self-uri><self-uri xlink:href="https://essd.copernicus.org/articles/.pdf">The full text article is available as a PDF file from https://essd.copernicus.org/articles/.pdf</self-uri>
      <abstract><title>Abstract</title>
    <p id="d1e142">Surface air temperature (<inline-formula><mml:math id="M1" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>), as an important
climate variable, has been used in a wide range of fields such as ecology,
hydrology, climatology, epidemiology, and environmental science. However,
ground measurements are limited by poor spatial representation and
inconsistency, and reanalysis and meteorological forcing datasets suffer
from coarse spatial resolution and inaccuracy. Previous studies using
satellite data have mainly estimated <inline-formula><mml:math id="M2" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under clear-sky conditions or
with limited temporal and spatial coverage. In this study, an all-sky daily
mean land <inline-formula><mml:math id="M3" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> product at a 1 km spatial resolution over mainland China for
2003–2019 has been generated mainly from the Moderate Resolution Imaging
Spectroradiometer (MODIS) products and the Global Land Data Assimilation
System (GLDAS) dataset. Three <inline-formula><mml:math id="M4" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation models based on random
forest were trained using ground measurements from 2384 stations for three
different clear-sky and cloudy-sky conditions. The random sample validation
results showed that the <inline-formula><mml:math id="M5" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> and root-mean-square error (RMSE) values of the
three models ranged from 0.984 to 0.986 and from 1.342 to 1.440 K,
respectively. We examined the spatiotemporal patterns and land cover type
dependences of model accuracy. Two cross-validation (CV) strategies of
leave-time-out (LTO) CV and leave-location-out (LLO) CV were also used to
evaluate the models. Finally, we developed the all-sky <inline-formula><mml:math id="M6" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> dataset from
2003 to 2009 and compared it with the China Land Data Assimilation System
(CLDAS) dataset at a 0.0625<inline-formula><mml:math id="M7" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> spatial resolution, the China
Meteorological Forcing Data (CMFD) dataset at a 0.1<inline-formula><mml:math id="M8" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> spatial
resolution, and the GLDAS dataset at a 0.25<inline-formula><mml:math id="M9" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> spatial resolution.
Validation accuracy of our product in 2010 was significantly better than
other datasets, with <inline-formula><mml:math id="M10" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> and RMSE values of 0.992 and 1.010 K,
respectively. In summary, the developed all-sky daily mean land <inline-formula><mml:math id="M11" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
dataset has achieved satisfactory accuracy and high spatial resolution
simultaneously, which fills the current dataset gap in this field and plays
an important role in the studies of climate change and the hydrological cycle.
This dataset is currently freely available at <ext-link xlink:href="https://doi.org/10.5281/zenodo.4399453" ext-link-type="DOI">10.5281/zenodo.4399453</ext-link>
(Chen et al., 2021b) and the University of Maryland
(<uri>http://glass.umd.edu/Ta_China/</uri>, last access: 24 August 2021). A sub-dataset
that covers Beijing generated from this dataset is also publicly available
at <ext-link xlink:href="https://doi.org/10.5281/zenodo.4405123" ext-link-type="DOI">10.5281/zenodo.4405123</ext-link> (Chen et al., 2021a).</p>
  </abstract>
    </article-meta>
  </front>
<body>
      

<?pagebreak page4242?><sec id="Ch1.S1" sec-type="intro">
  <label>1</label><title>Introduction</title>
      <p id="d1e280">Surface air temperature (<inline-formula><mml:math id="M12" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>) is one of the most important variables in
a wide range of fields including ecology, hydrology, climatology,
epidemiology, and environmental science (Goetz et al., 2000; Stisen et al.,
2007; Vancutsem et al., 2010; Zhang et al., 2018). <inline-formula><mml:math id="M13" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> refers to the
atmospheric temperature 1.5–2 m above the surface, which represents the
thermal state information of the surface and the lower atmosphere. It
influences the carbon cycle through the biophysical effects of vegetation
and regulates many surface processes such as photosynthesis, respiration,
and evaporation (Khesali and Mobasheri, 2020). Reliable estimates of <inline-formula><mml:math id="M14" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
at fine spatiotemporal resolution are important to better understand and
simulate complex surface processes and reveal changes due to climate change
or local disturbances (Guan et al., 2013). Moreover, in the context of
continuous global warming, meteorological disasters caused by frequent
extreme weather events and consequential social and economic losses are
gradually increasing. A deep understanding of the spatiotemporal patterns of
<inline-formula><mml:math id="M15" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is also of great guiding significance for disaster prevention and
reduction.</p>
      <p id="d1e327">However, because of its proximity to the interface between the land/ocean and
atmosphere, the near-surface air is influenced by various exchange processes
between these three Earth system compartments (Schwingshackl et al., 2018).
The spatiotemporal patterns of <inline-formula><mml:math id="M16" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> can vary and be complicated due to
the heterogeneity of various environmental factors (such as solar radiation,
latitude, underlying surface, cloud cover, and season) that impact the
energy balance of the land–atmosphere system (Benali et al., 2012; Chen et
al., 2015; Prihodko and Goward, 1997).</p>
      <p id="d1e341"><inline-formula><mml:math id="M17" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> data are one of the most frequent forms of observation data recorded by
meteorological stations. In situ <inline-formula><mml:math id="M18" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> usually has reliable accuracy and
high temporal resolution; however, it has some flaws, such as limited
spatial representation, measurement inconsistency, and uneven spatial
distribution of ground stations (Jang et al., 2014; Prihodko and Goward,
1997). Geographical interpolation methods such as inverse distance weighting
(IDW), kriging, and spline function have been widely used to estimate the
spatial distribution of <inline-formula><mml:math id="M19" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> (Benavides et al., 2007; Ishida and
Kawashima, 1993; Kurtzman and Kadmon, 1999). However, these methods usually
consider only the autocorrelation of <inline-formula><mml:math id="M20" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, ignoring the complex factors
that lead to its heterogeneity. The accuracy of interpolated <inline-formula><mml:math id="M21" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is
greatly affected by station network density, which leads to relatively poor
accuracy being obtained in areas with sparse station density (Stisen et al., 2007;
Vogt et al., 1997). Therefore, the accuracy of interpolated <inline-formula><mml:math id="M22" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> may have
significant errors associated with unrepresentative spatial patterns, and
there can be great uncertainty in describing the spatial patterns of <inline-formula><mml:math id="M23" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
over large areas in this way (Benali et al., 2012; Rao et al., 2018).</p>
      <p id="d1e421">Remotely sensed data have provided unprecedented spatial coverage at
regional and global spatial scales (Liang, 2004). Over the past few decades,
many schemes have been developed to estimate <inline-formula><mml:math id="M24" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> from remotely sensed
data. The strong physical relationship between the land surface temperature
(LST) and <inline-formula><mml:math id="M25" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> has become the research basis of many <inline-formula><mml:math id="M26" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation
methods. Generally speaking, the LST-based <inline-formula><mml:math id="M27" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation methods can be
divided into three distinct categories. The first type is the traditional
statistical method, including the univariate regression method to establish a
linear relationship between <inline-formula><mml:math id="M28" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and LST, and multiple regression methods
considering various variables (such as solar zenith angle, elevation, and Julian
day) in addition to LST (Lin et al., 2012; Zeng et al., 2015). The
second is the temperature–vegetation index (TVX) method, based on the
negative correlation between the normalized difference vegetation index (NDVI)
and LST in the study area (Stisen et al., 2007; Vancutsem et al., 2010; Zhu
et al., 2013). The third is the land surface energy-balance physical method,
which uses the crop water stress index (CWSI) and the aerodynamic resistance to
estimate <inline-formula><mml:math id="M29" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. This method has a good physical basis but usually relies
on numerous input parameters (such as roughness and soil physical properties),
which are always difficult to obtain (Sun et al., 2004). In principle, the
atmospheric profile products from satellite observations include the temperature
profile of the entire atmosphere but usually require additional processes
to obtain <inline-formula><mml:math id="M30" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. The Moderate Resolution Imaging Spectroradiometer (MODIS)
atmospheric profile product has been used for this purpose (Bisht and Bras,
2010; Borbas and Menzel, 2017; Famiglietti et al., 2018; Zhu et al., 2017).
Generally, traditional statistical methods have been commonly used but have
reported low accuracy. In recent years, machine learning methods,
particularly deep learning methods, such as support vector machine (Zhang et
al., 2016), artificial neural network (Jang et al., 2010; Zhang et al.,
2016), M5 model trees (Emamifar et al., 2013), random forest (RF) models
(Noi et al., 2017; Xu et al., 2014; Zhang et al., 2016), cubist models
(Meyer et al., 2016; Noi et al., 2017; Rao et al., 2019), and advanced deep
learning methods (Shen et al., 2020), have been gradually applied to <inline-formula><mml:math id="M31" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
estimation from satellite data because of their stronger learning ability to
capture the complex nonlinear relationship between various factors.</p>
      <p id="d1e514">Most LST-based <inline-formula><mml:math id="M32" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation methods mentioned above are suitable only
for clear-sky conditions as the current LST datasets are mainly derived from
satellite thermal infrared radiances (TIR) that are susceptible to cloud
contamination (Liang et al., 2019; Ma et al., 2020). Currently, there are
two main strategies for estimating all-sky <inline-formula><mml:math id="M33" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> based on LST: one is to
first derive <inline-formula><mml:math id="M34" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> from the available LST and then fill the <inline-formula><mml:math id="M35" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> gaps
(Rosenfeld et al., 2017; Zhang, 2017); the other is to first fill the LST
gaps to develop a seamless product and then estimate the all-sky <inline-formula><mml:math id="M36" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
(Kilibarda et al., 2014; Li et al., 2018; Rao et al., 2019). For example,
Zhang (2017)  estimated <inline-formula><mml:math id="M37" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under clear-sky conditions based on
MODIS LST, and the Atmospheric Infrared Sounder (AIRS) standard <inline-formula><mml:math id="M38" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
products were used to fill the cloudy-sky pixels after a<?pagebreak page4243?> downscaling
process, with a mean absolute error (MAE) of 1.2 K and a root-mean-square
error (RMSE) of 1.6 K overall. According to the research conducted by
Kilibarda et al. (2014), the 8 d composite LST was interpolated into a
daily dataset and then combined with topographic layers and a geometric
temperature trend to interpolate the all-sky daily <inline-formula><mml:math id="M39" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, and the results
reported that the RMSE values were between 2 and 4 <inline-formula><mml:math id="M40" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>C for daily mean, maximum, and minimum <inline-formula><mml:math id="M41" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. In addition, Zhu et al. (2017)
developed a parameterization scheme to estimate all-sky instantaneous
daytime <inline-formula><mml:math id="M42" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> only relying on the MODIS atmospheric profile product. They
first established the relationship between LST and <inline-formula><mml:math id="M43" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under clear-sky
conditions and then estimated <inline-formula><mml:math id="M44" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under cloudy-sky conditions based on
the established relationship, with RMSE values ranging from 2.50
to 2.56 <inline-formula><mml:math id="M45" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>C.</p>
      <p id="d1e669">Currently, several studies have been conducted to develop all-sky <inline-formula><mml:math id="M46" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
datasets based on remotely sensed data. For instance, Li et al. (2018) used
a three-step hybrid gap-filling method to attain seamless LST; they then
developed daily geographically weighted regression (GWR) models to
interpolate <inline-formula><mml:math id="M47" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> using gap-filled LST and elevation, and finally
developed a 1 km daily minimum/maximum <inline-formula><mml:math id="M48" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> dataset in urban and
surrounding areas in the conterminous US for 2003–2016. The
cross-validation results reported that the RMSE values were 2.1 and 1.9 <inline-formula><mml:math id="M49" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>C for daily minimum and maximum <inline-formula><mml:math id="M50" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, respectively.
In the recent work conducted by Yao et al. (2020), the MODIS 8 d composite
LST was averaged to obtain monthly mean LST and was then combined with the enhanced
vegetation index (EVI), solar radiation, topographic index, and other
features to establish a cubist model for generating 1 km monthly
maximum/mean/minimum <inline-formula><mml:math id="M51" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> products in China: the RMSE of the
estimated monthly mean <inline-formula><mml:math id="M52" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> was 0.629 <inline-formula><mml:math id="M53" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>C. Rao et al. (2019)
first filled the gaps of LSTs and then used the gap-filled LSTs and some
radiation products to build cubist models for estimating all-sky daily mean
<inline-formula><mml:math id="M54" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, with an RMSE of 1.87 <inline-formula><mml:math id="M55" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>C. Finally, a 0.05<inline-formula><mml:math id="M56" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> <inline-formula><mml:math id="M57" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 0.05<inline-formula><mml:math id="M58" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> daily mean <inline-formula><mml:math id="M59" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> product over the Tibetan
Plateau for 2002–2016 was developed. In addition, multiple
reanalysis and meteorological forcing datasets covering large areas or
global areas exist, which are usually generated by data assimilation or data
interpolation, such as the Global Land Data Assimilation System (GLDAS; Rodell et al., 2004); Modern-Era Retrospective Analysis and Research and
Application, version 2 (MERRA-2; Gelaro et al., 2017); China Meteorological
Forcing Data (CMFD; Yang and He, 2019); and China Land Data Assimilation
System (CLDAS; Shi et al., 2011). However, these datasets have coarse
spatial resolution (generally <inline-formula><mml:math id="M60" display="inline"><mml:mo>≥</mml:mo></mml:math></inline-formula> 0.1<inline-formula><mml:math id="M61" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> except for CLDAS, which has a
spatial resolution of 0.0625<inline-formula><mml:math id="M62" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>) and regional inaccuracy, which may
limit their potential to accurately capture the spatial heterogeneity of
<inline-formula><mml:math id="M63" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in the urban and mountainous areas and lead to uncertainties for
applications at local to regional scales (Jang et al., 2014; Li et al.,
2018; Zhu et al., 2017). To our knowledge, there is currently a lack of long-time-series all-sky <inline-formula><mml:math id="M64" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> products covering vast areas with both high spatial
and temporal resolution.</p>
      <p id="d1e862">The main objective of this study is to develop an all-sky 1 km daily mean
land <inline-formula><mml:math id="M65" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> over mainland China for 2003–2019 by integrating satellite
data products, model simulations, and ground measurements. For the first
time, assimilated <inline-formula><mml:math id="M66" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> was applied to supplement and substitute MODIS
LSTs and provide the initial values of model prediction. In order to solve
the issue of missing LST, a simple temporal gap-filling method was used to
fill the gaps of MODIS LSTs first. Considering the differences in the
relationship between <inline-formula><mml:math id="M67" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and other features under different weather
conditions, we divided all data pairs into three types of weather
conditions – (1) clear-sky conditions, (2) cloudy-sky conditions case I, and (3) cloudy-sky conditions case II – and then established three machine learning
models to estimate daily mean <inline-formula><mml:math id="M68" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under different weather conditions.
The structure of this paper is organized as follows: Sect. 2 describes the
study area and the data used; Sect. 3 summarizes the overall research method;
Sect. 4 reports the validation results and discusses the model
performance; Sect. 5 compares the developed dataset with the existing
datasets; and Sect. 6 presents the overall conclusions.</p>
</sec>
<sec id="Ch1.S2">
  <label>2</label><title>Data</title>
<sec id="Ch1.S2.SS1">
  <label>2.1</label><title>Meteorological station data</title>
      <p id="d1e924">This study was conducted in mainland China. The station-observed daily mean
<inline-formula><mml:math id="M69" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> from 2003 to 2019 was collected from 2384 standard meteorological
stations in mainland China for model training and validation. During the
production process of this dataset, it experienced strict quality control.
Figure 1 shows the study area and the geographical location of the
meteorological stations used in this study. Each dot represents a station,
and different colors correspond to different land cover types. The land
cover data used in the study are Finer Resolution Observation and Monitoring
of Global Land Cover (FROM-GLC), version2 (2015_v1), which are
30 m resolution global land cover maps (Gong et al., 2013).</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F1" specific-use="star"><?xmltex \currentcnt{1}?><?xmltex \def\figurename{Figure}?><label>Figure 1</label><caption><p id="d1e940">Study area and the location of meteorological stations used in this
study. Each dot represents a station, and different colors correspond to
different land cover types as shown in this figure legend.</p></caption>
          <?xmltex \igopts{width=355.659449pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f01.png"/>

        </fig>

</sec>
<sec id="Ch1.S2.SS2">
  <label>2.2</label><title>Remotely sensed data</title>
      <p id="d1e957">The satellite datasets used in this study are listed in Table 1.</p>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T1" specific-use="star"><?xmltex \currentcnt{1}?><label>Table 1</label><caption><p id="d1e963">The satellite datasets used in this study.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="4">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:colspec colnum="3" colname="col3" align="left"/>
     <oasis:colspec colnum="4" colname="col4" align="left"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Product</oasis:entry>
         <oasis:entry colname="col2">Dataset(s)</oasis:entry>
         <oasis:entry colname="col3">Spatial resolution</oasis:entry>
         <oasis:entry colname="col4">Temporal resolution</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">Land surface temperature (LST)</oasis:entry>
         <oasis:entry colname="col2">MOD11A1, MYD11A1</oasis:entry>
         <oasis:entry colname="col3">1 km</oasis:entry>
         <oasis:entry colname="col4">Daily</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Downward shortwave radiation (DSR)</oasis:entry>
         <oasis:entry colname="col2">GLASS05B01</oasis:entry>
         <oasis:entry colname="col3">0.05<inline-formula><mml:math id="M70" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col4">Daily</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Surface albedo (ALB)</oasis:entry>
         <oasis:entry colname="col2">GLASS02A06</oasis:entry>
         <oasis:entry colname="col3">1 km</oasis:entry>
         <oasis:entry colname="col4">8 d</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Leaf area index (LAI)</oasis:entry>
         <oasis:entry colname="col2">GLASS01A01</oasis:entry>
         <oasis:entry colname="col3">1 km</oasis:entry>
         <oasis:entry colname="col4">8 d</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Elevation</oasis:entry>
         <oasis:entry colname="col2">GMTED2010</oasis:entry>
         <oasis:entry colname="col3">15 arcsec</oasis:entry>
         <oasis:entry colname="col4">–</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <p id="d1e1084">Terra and Aqua MODIS daily 1 km LST products (MOD11A1/MYD11A1, C6) both
provide daytime and nighttime LSTs with a spatial resolution of 1 km (Wan
et al., 2015).</p>
      <p id="d1e1088">Three all-sky products from the Global LAnd Surface Satellite (GLASS)
products suite (Liang et al., 2013, 2021) were used, including
the GLASS 1 km 8 d surface broadband albedo (ALB) product GLASS02A06 (Liu
et al., 2013), the GLASS 0.05<inline-formula><mml:math id="M71" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> daily downward shortwave radiation
(DSR) product GLASS05B01 (Zhang et al., 2019), and the GLASS<?pagebreak page4244?> 1 km 8 d leaf
area index (LAI) product GLASS01A01 (Xiao et al., 2014). For the ALB
product, we used the black-sky albedo of shortwave (BSA_sw),
visible (BSA_vis), and near-infrared (BSA_nir)
bands. As radiation products, DSR and ALB determine the shortwave solar
radiation received at the surface and the fraction of total radiation
reflected and absorbed by the surface, respectively.</p>
      <p id="d1e1100">The Global Multi-resolution Terrain Elevation Data 2010 (GMTED2010)
elevation dataset, downloaded from the United States Geological Survey
(USGS, <uri>https://topotools.cr.usgs.gov/GMTED_viewer/viewer.htm</uri>, last access: 24 August 2021), was also
chosen to estimate <inline-formula><mml:math id="M72" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>.</p>
</sec>
</sec>
<sec id="Ch1.S3">
  <label>3</label><title>Methods</title>
      <p id="d1e1126">The overall framework of this study is shown in Fig. 2. First, all
datasets from 2003 to 2019 were preprocessed into identical spatial and
temporal resolutions. Second, we filled the gaps of MODIS LSTs and then
divided all data pairs into three weather conditions according to the
gap-filling results. Next, the values of all datasets were extracted using the
nearest-neighbor method according to the geographical location of stations
and then matched with the in situ <inline-formula><mml:math id="M73" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> to obtain data pairs. Data pairs
under different weather conditions from 2003 to 2016 were randomly divided
into training, validation, and test sets (ratio of <inline-formula><mml:math id="M74" display="inline"><mml:mrow><mml:mn mathvariant="normal">3</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula>). Three RF models for
different weather conditions were established and trained using the training
set. Three model validation strategies of random sample validation,
leave-time-out (LTO) cross-validation (CV), and leave-location-out (LLO) CV
were then used to evaluate the models. Finally, we used the models to develop the
all-sky <inline-formula><mml:math id="M75" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> dataset from 2003 to 2009 and compared it with the existing
datasets.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F2" specific-use="star"><?xmltex \currentcnt{2}?><?xmltex \def\figurename{Figure}?><label>Figure 2</label><caption><p id="d1e1169">The overall framework of this study.</p></caption>
        <?xmltex \igopts{width=384.112205pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f02.png"/>

      </fig>

<sec id="Ch1.S3.SS1">
  <label>3.1</label><title>Data preprocessing</title>
      <p id="d1e1185">Because the spatial and temporal resolutions of all datasets were not
completely consistent, we preprocessed all remotely sensed datasets and
reanalysis datasets from 2003<?pagebreak page4245?> to 2019 into identical 1 km and daily spatial
and temporal resolutions, respectively. DSR, elevation, and assimilated
<inline-formula><mml:math id="M76" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> were resampled to a 1 km spatial resolution using the nearest-neighbor method. As LAI and ALB datasets both have an 8 d temporal
resolution, we first combined them into a time series and then interpolated
the time series using the linear interpolation method to obtain the daily datasets.
For GLDAS assimilation data with a 3 h temporal resolution, we averaged
all assimilated instantaneous <inline-formula><mml:math id="M77" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in a day to acquire the assimilated
daily mean <inline-formula><mml:math id="M78" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> for all days.</p>
      <p id="d1e1221">The values of all datasets were then extracted using the nearest-neighbor
method according to the geographical locations of stations and matched
with the in situ <inline-formula><mml:math id="M79" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> to obtain data pairs. Next, we used a temporal
gap-filling method to fill the MODIS LST gaps and divided all data pairs
into three weather conditions according to the gap-filling results. The
detailed gap-filling method and strategy for the division into weather condition categories is
described in the Sect. 3.2. The data pairs with different weather
conditions from 2003 to 2016 were then randomly divided into training,
validation, and test sets (ratio of <inline-formula><mml:math id="M80" display="inline"><mml:mrow><mml:mn mathvariant="normal">3</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula>). Among them, the training set was used
for model training, the validation set was used to determine the best model
parameters, and the test set was used to evaluate the final model performance.</p>
</sec>
<sec id="Ch1.S3.SS2">
  <label>3.2</label><title>Strategies for LST gap-filling and division into weather condition categories</title>
      <p id="d1e1260">MODIS LSTs were produced under strict quality control, with each pixel
marked as either a clear-sky or cloudy-sky observation. Pixels under
cloudy-sky conditions had missing LST values, meaning that the LST-based
method could not be applied to estimate <inline-formula><mml:math id="M81" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. In this study, a simple
multitemporal method was used to fill the MODIS LST gaps. First, we set a
time threshold (<inline-formula><mml:math id="M82" display="inline"><mml:mo lspace="0mm">±</mml:mo></mml:math></inline-formula>2 d), and the missing pixel value was replaced by
the clear-sky value of the nearest date within the set time threshold. If no
clear-sky pixel was found within the time threshold, the missing pixel was
not filled to avoid introducing high uncertainty caused by a huge
temperature change between dates with large differences. This
multitemporal method was used to fill the gaps of all four MODIS LSTs each
day.</p>
      <p id="d1e1281">Considering the differences in the relationship between <inline-formula><mml:math id="M83" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and other
features under different weather conditions, we divided data pairs into
clear-sky conditions and cloudy-sky conditions according to the LSTs
gap-filling results. When all four LSTs in a day were under clear-sky
conditions, the data pair was identified as being under clear-sky
conditions; otherwise, it was identified as being under cloudy-sky
conditions. To control the uncertainty introduced by LST gap-filling,
cloudy-sky conditions were divided into two cases: case I and case II. In
particular, a data pair was identified as being under cloudy-sky conditions
case I when there were<?pagebreak page4246?> LST gaps in the data pair and the gaps could be
filled using the method mentioned above. If the LST gaps could not all be
filled, the data pair was identified as being under cloudy-sky conditions
case II. Therefore, we finally divided all data pairs into three
weather condition categories: (1) clear-sky conditions, (2) cloudy-sky conditions case I, and (3) cloudy-sky conditions case II. The detailed criteria for dividing
weather conditions are shown in Fig. 3.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F3" specific-use="star"><?xmltex \currentcnt{3}?><?xmltex \def\figurename{Figure}?><label>Figure 3</label><caption><p id="d1e1297">The criteria for the weather-condition-based division of a data pair.</p></caption>
          <?xmltex \igopts{width=341.433071pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f03.png"/>

        </fig>

      <p id="d1e1307">Next, we established three machine learning models (clear-sky model,
cloudy-sky model I, and cloudy-sky model II) and trained them separately for
different weather conditions. Daily LSTs were used in models for clear-sky
conditions (clear-sky model) and cloudy-sky conditions case I (cloudy-sky
model I), but not for cloudy-sky conditions case II (cloudy-sky model II).
GLDAS-assimilated <inline-formula><mml:math id="M84" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, GLASS DSR, GLASS ALB, GLASS LAI, elevation, and
temporal and locational information were also used in all three models as
input features. For the clear-sky model, the utilized features included four
clear-sky LSTs in a day. The qualification for a pixel of a given day to be
judged as clear-sky may be harsh, but this ensured the use of completely
clear-sky LSTs. The features of cloudy-sky model I included gap-filled
LST(s), which increased the availability of LST, but the simple gap-filling
strategy also introduced errors to the models. To avoid instilling high
uncertainty caused by a large temperature change between dates with large
differences, cloudy-sky model II did not use LST to estimate <inline-formula><mml:math id="M85" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>.</p>
</sec>
<sec id="Ch1.S3.SS3">
  <label>3.3</label><title>Random forest</title>
      <p id="d1e1340">The RF method (Breiman, 2001) is an ensemble learning method based on classification and
regression tree (CART) proposed by Breiman et al. (1984). Since it was
proposed, it has attracted the attention of quite a few fields of study and has specifically had various applications in remote sensing in recent years (Gislason
et al., 2006; Ham et al., 2005; Li and Zha, 2019; Xu et al., 2014).</p>
      <p id="d1e1343">A decision tree is a tree-like prediction model composed of nodes and
directed edges. In each internal node of the decision tree, the sample set
is segmented by selecting the optimal splitting feature until the
segmentation termination condition is reached. Each path from the root node
to the leaf nodes of a decision tree forms a classification. There are many
algorithms for decision tree, such as ID3 (Quinlan, 1986), C4.5 (Quinlan,
1992), and CART. These algorithms all adopt the top-down greedy algorithm,
and each internal node chooses the feature with the best classification
effect to split, in order to achieve the goal of dividing samples into subsets that
are as homogenous as possible, with the fastest speed. In the generation
algorithms of ID3 and C4.5 decision tree, information gain or the information
gain rate is used as the criterion to judge the optimal segmentation.
Another type of optimal segmentation criterion is Gini impurity, which is
utilized in the CART decision tree. In the RF model, multiple CART decision
trees are included. The bagging method (Breiman, 1996) is used to generate
independent identically distributed training sample sets for each tree and
train on them.</p>
      <p id="d1e1346">Although the application of RF at present is mainly focused on
classification, it can be also used in regression analysis effectively,
which can usually achieve higher accuracy than traditional regression
analysis methods. The training and prediction process of the RF regression
model is shown in Fig. 4. First, the bootstrapping method is used to acquire
<inline-formula><mml:math id="M86" display="inline"><mml:mi>k</mml:mi></mml:math></inline-formula> datasets <inline-formula><mml:math id="M87" display="inline"><mml:mrow><mml:mo mathvariant="italic">{</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mi mathvariant="normal">…</mml:mi><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula> and then <inline-formula><mml:math id="M88" display="inline"><mml:mi>k</mml:mi></mml:math></inline-formula>
decision trees <inline-formula><mml:math id="M89" display="inline"><mml:mrow><mml:mo mathvariant="italic">{</mml:mo><mml:mi>h</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mi mathvariant="normal">…</mml:mi><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula> are established, respectively, where <inline-formula><mml:math id="M90" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> is the input
vector, and <inline-formula><mml:math id="M91" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> (<inline-formula><mml:math id="M92" display="inline"><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mi mathvariant="normal">…</mml:mi></mml:mrow></mml:math></inline-formula>) is the random vector determining
the sampling of bootstrap datasets and candidate splitting features of each
tree. The construction of a decision tree is realized by iteratively
dividing the datasets into two subsets. Different from the RF classification
model, the mean square error (MSE) is used as the optimal segmentation
criterion in the RF regression model to split the nodes. Each decision tree
in the RF regression model takes values rather than types as output targets,
and the average of the predicted values of all the trees <inline-formula><mml:math id="M93" display="inline"><mml:mrow><mml:mo mathvariant="italic">{</mml:mo><mml:mi>h</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mi mathvariant="normal">…</mml:mi><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula> is used as the final
prediction.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F4" specific-use="star"><?xmltex \currentcnt{4}?><?xmltex \def\figurename{Figure}?><label>Figure 4</label><caption><p id="d1e1512">The training and prediction process of the RF regression model.</p></caption>
          <?xmltex \igopts{width=441.017717pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f04.png"/>

        </fig>

</sec>
<sec id="Ch1.S3.SS4">
  <label>3.4</label><title>Model training and validation</title>
      <p id="d1e1529">During the model training process, the training set was used for model training, and the
validation set was used to determine the models with the optimal
hyperparameters.</p>
      <p id="d1e1532">Compared with artificial neural network, the RF regression model does not need
to carry out complicated parameter tuning work, and changing some
insignificant parameters of the RF model may not cause substantial
fluctuations in model performance. The two most critical hyperparameters,
ntree and mtry, need to be determined during training. Among them, ntree
refers to the number of decision trees in the RF model. Increasing ntree is
conducive to improving the model performance and stability but also affects
the computational efficiency of the program. Mtry refers to the maximum
number of features used in a single decision tree. When mtry is less than
the total number of features, the segmentation of a node is determined based
on partial features that are randomly selected rather than all features.
Increasing mtry allows nodes to consider more features when splitting but
also reduces the diversity of individual trees, thereby increasing the risk of
overfitting. Therefore, both parameters need to be properly balanced and
selected, and we used the validation set to evaluate the model performance
with different combinations of parameters to obtain the optimal
hyperparameters.</p>
      <p id="d1e1535">Assuming the total number of features of a sample is <inline-formula><mml:math id="M94" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>, the values of mtry
include log<inline-formula><mml:math id="M95" display="inline"><mml:mrow><mml:msub><mml:mi/><mml:mn mathvariant="normal">2</mml:mn></mml:msub><mml:mi>m</mml:mi></mml:mrow></mml:math></inline-formula>, sqrt(<inline-formula><mml:math id="M96" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>), and <inline-formula><mml:math id="M97" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>, and ntree is set to 5–200. To analyze
the RF model performance sensitivity to hyperparameters, the RMSE values of
the three models for different weather conditions were calculated when
setting different parameters, and the result is shown in Fig. 5. It can<?pagebreak page4247?> be
seen from the results that with the change in model parameters, the three
models showed similar variation patterns. With the increase of ntree, the
RMSE value decreased gradually until it became almost constant (when ntree <inline-formula><mml:math id="M98" display="inline"><mml:mo>≥</mml:mo></mml:math></inline-formula> 100). The continued increase of ntree made very little contribution to
improving the model performance but affected the computing efficiency. For
mtry, we can see that using partial features (mtry of log<inline-formula><mml:math id="M99" display="inline"><mml:mrow><mml:msub><mml:mi/><mml:mn mathvariant="normal">2</mml:mn></mml:msub><mml:mi>m</mml:mi></mml:mrow></mml:math></inline-formula> or
sqrt(<inline-formula><mml:math id="M100" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>)) resulted in significantly better performance than using all features (mtry of
<inline-formula><mml:math id="M101" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>). Overall, setting mtry to log<inline-formula><mml:math id="M102" display="inline"><mml:mrow><mml:msub><mml:mi/><mml:mn mathvariant="normal">2</mml:mn></mml:msub><mml:mi>m</mml:mi></mml:mrow></mml:math></inline-formula> and sqrt(<inline-formula><mml:math id="M103" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>) presented similar
performance, and the setting of sqrt(<inline-formula><mml:math id="M104" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>) performed slightly better than
log<inline-formula><mml:math id="M105" display="inline"><mml:mrow><mml:msub><mml:mi/><mml:mn mathvariant="normal">2</mml:mn></mml:msub><mml:mi>m</mml:mi></mml:mrow></mml:math></inline-formula> when ntree was larger than 175. Therefore, we set ntree to 200
and mtry to sqrt(<inline-formula><mml:math id="M106" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>) in all models.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F5" specific-use="star"><?xmltex \currentcnt{5}?><?xmltex \def\figurename{Figure}?><label>Figure 5</label><caption><p id="d1e1654">RF model performance sensitivity to hyperparameters.</p></caption>
          <?xmltex \igopts{width=412.564961pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f05.png"/>

        </fig>

      <?pagebreak page4248?><p id="d1e1663">To quantitatively evaluate the effect of each feature on the models, we
calculated the feature importance (FI) of every feature using the permutation
method for each model. The permutation method breaks the statistical
relationship between feature <inline-formula><mml:math id="M107" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula> and the target variable and then measures the
degree of deterioration in the model performance to evaluate the importance
of feature <inline-formula><mml:math id="M108" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula> to the model (Mcgovern et al., 2019). Specifically, the
model is first trained with the training set, and the RMSE of the validation set
(RMSE<inline-formula><mml:math id="M109" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">true</mml:mi></mml:msub></mml:math></inline-formula>) is then calculated using Eq. (1). For the calculation of the FI
of feature <inline-formula><mml:math id="M110" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula>, RMSE<inline-formula><mml:math id="M111" display="inline"><mml:msub><mml:mi/><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is calculated again after all the features <inline-formula><mml:math id="M112" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula> of the
validation set are shuffled. The difference between RMSE<inline-formula><mml:math id="M113" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">true</mml:mi></mml:msub></mml:math></inline-formula> and
RMSE<inline-formula><mml:math id="M114" display="inline"><mml:msub><mml:mi/><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is calculated and then divided by RMSE<inline-formula><mml:math id="M115" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">true</mml:mi></mml:msub></mml:math></inline-formula>, and the result
is used as FI, as shown in the Eq. (2). A large FI value means that the
model performance decreases significantly after shuffling this feature,
which indicates that this feature has a great impact on the accuracy of
prediction results. On the contrary, if the model performance does not
deteriorate significantly, it is obvious that this feature has less
influence on the prediction process, or that other linearly dependent
features are included in the model to make this feature
redundant.

                <disp-formula specific-use="align" content-type="numbered"><mml:math id="M116" display="block"><mml:mtable displaystyle="true"><mml:mlabeledtr id="Ch1.E1"><mml:mtd><mml:mtext>1</mml:mtext></mml:mtd><mml:mtd><mml:mstyle class="stylechange" displaystyle="true"/></mml:mtd><mml:mtd><mml:mrow><mml:mstyle displaystyle="true" class="stylechange"/><mml:mtext>RMSE</mml:mtext><mml:mo>=</mml:mo><mml:msqrt><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:munderover><mml:msup><mml:mfenced close=")" open="("><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi mathvariant="normal">pre</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi mathvariant="normal">obs</mml:mi></mml:msub></mml:mrow></mml:mfenced><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow><mml:mi>n</mml:mi></mml:mfrac></mml:mstyle></mml:msqrt><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mlabeledtr><mml:mlabeledtr id="Ch1.E2"><mml:mtd><mml:mtext>2</mml:mtext></mml:mtd><mml:mtd><mml:mstyle displaystyle="true" class="stylechange"/></mml:mtd><mml:mtd><mml:mrow><mml:mstyle class="stylechange" displaystyle="true"/><mml:msub><mml:mi mathvariant="normal">FI</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:msub><mml:mi mathvariant="normal">RMSE</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mi mathvariant="normal">RMSE</mml:mi><mml:mi mathvariant="normal">true</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi mathvariant="normal">RMSE</mml:mi><mml:mi mathvariant="normal">true</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mstyle><mml:mspace linebreak="nobreak" width="0.125em"/><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mlabeledtr></mml:mtable></mml:math></disp-formula>

            where <inline-formula><mml:math id="M117" display="inline"><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi mathvariant="normal">pre</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> refers to model prediction result, and <inline-formula><mml:math id="M118" display="inline"><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi mathvariant="normal">obs</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> refers to
the corresponding station observation. RMSE<inline-formula><mml:math id="M119" display="inline"><mml:msub><mml:mi/><mml:mi mathvariant="normal">true</mml:mi></mml:msub></mml:math></inline-formula> is the RMSE of the
validation set, and RMSE<inline-formula><mml:math id="M120" display="inline"><mml:msub><mml:mi/><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> refers to the RMSE of the validation set
after feature <inline-formula><mml:math id="M121" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula> has been shuffled.</p>
      <p id="d1e1884">The <inline-formula><mml:math id="M122" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> predicted by the models was compared to the corresponding
station observations. The RMSE, MAE, and <inline-formula><mml:math id="M123" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> were selected as criteria for
model evaluation. In order to comprehensively evaluate the performance of
the models, we adopted three model validation strategies: random sample
validation, LTO CV, and LLO CV. For random sample validation, the test set (one-fifth
of the total data from 2003 to 2016 selected randomly) was used to evaluate
the performance of the final <inline-formula><mml:math id="M124" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation models. The results were
grouped by elevation range, land cover type, and month to evaluate the model
performance under different situations. For LTO CV and LLO CV, we divided
all data pairs into 14 groups according to calendar year and 7 groups
according to geographical location. In each iteration, one group of data was
used for validation, and the other groups of data were used as the training
set for model training. The modeling and validation process were repeated 14
and 7 times until each year's data and each cluster of data were validated, respectively.
These two CV strategies have been used in some studies to evaluate the
performance of spatiotemporal models in unknown time or unknown space (Liu
et al., 2020; Ploton et al., 2020; Xiao et al., 2018).</p>
</sec>
</sec>
<?pagebreak page4249?><sec id="Ch1.S4">
  <label>4</label><title>Analysis of the results</title>
<sec id="Ch1.S4.SS1">
  <label>4.1</label><title>Overall accuracy and model comparison</title>
      <p id="d1e1936">Approximately three-fifths and one-fifth of the data pairs from 2003 to 2016 were randomly
selected for training and tuning the models, respectively, and the remaining
one-fifth of the total data pairs were used to evaluate the performance of the
final <inline-formula><mml:math id="M125" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation models. Validation statistics of models for
different weather conditions and the overall accuracy of all estimated daily
mean <inline-formula><mml:math id="M126" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> are shown in Table 2. The three models presented similar
validation statistics, with <inline-formula><mml:math id="M127" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, MAE, RMSE, and bias values ranging from 0.984
to 0.986, 1.033 K to 1.100 K, 1.342 K to 1.440 K, and 0.012 K to 0.051 K,
respectively. The overall <inline-formula><mml:math id="M128" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, MAE, RMSE, and bias values of the estimated
all-sky <inline-formula><mml:math id="M129" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> were 0.985, 1.068 K, 1.409 K, and 0.03 K, respectively.
Compared with the in situ <inline-formula><mml:math id="M130" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, the estimated <inline-formula><mml:math id="M131" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> of all models
showed a high correlation with little difference, confirming the great
potential of the RF method to estimate the all-sky daily mean <inline-formula><mml:math id="M132" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> over a wide
spatial and temporal range.</p>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T2" specific-use="star"><?xmltex \currentcnt{2}?><label>Table 2</label><caption><p id="d1e2031">Model validation statistics.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="5">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Model</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M133" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">MAE (K)</oasis:entry>
         <oasis:entry colname="col4">RMSE (K)</oasis:entry>
         <oasis:entry colname="col5">Bias (K)</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">Clear-sky model</oasis:entry>
         <oasis:entry colname="col2">0.986</oasis:entry>
         <oasis:entry colname="col3">1.033</oasis:entry>
         <oasis:entry colname="col4">1.342</oasis:entry>
         <oasis:entry colname="col5">0.021</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Cloudy-sky model I</oasis:entry>
         <oasis:entry colname="col2">0.984</oasis:entry>
         <oasis:entry colname="col3">1.100</oasis:entry>
         <oasis:entry colname="col4">1.440</oasis:entry>
         <oasis:entry colname="col5">0.012</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Cloudy-sky model II</oasis:entry>
         <oasis:entry colname="col2">0.984</oasis:entry>
         <oasis:entry colname="col3">1.046</oasis:entry>
         <oasis:entry colname="col4">1.396</oasis:entry>
         <oasis:entry colname="col5">0.051</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">All</oasis:entry>
         <oasis:entry colname="col2">0.985</oasis:entry>
         <oasis:entry colname="col3">1.068</oasis:entry>
         <oasis:entry colname="col4">1.409</oasis:entry>
         <oasis:entry colname="col5">0.030</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <p id="d1e2155">In addition, to further investigate the distribution of the prediction
results and the differences between the three models, density scatterplots
of the estimated <inline-formula><mml:math id="M134" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> against the in situ <inline-formula><mml:math id="M135" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> for the three models
are shown in Fig. 6. In the three density scatterplots, most points were
very concentrated near the <inline-formula><mml:math id="M136" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line, which also confirmed that these three
models have achieved satisfactory accuracy in estimating daily mean <inline-formula><mml:math id="M137" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
under different weather conditions. Among all the models, the clear-sky
model had the highest stability and overall accuracy statistically, with the
highest <inline-formula><mml:math id="M138" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> and the lowest MAE and RMSE. It could predict <inline-formula><mml:math id="M139" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under
clear-sky conditions from less than 250 K to more than 300 K accurately and
steadily. Compared with the clear-sky model, cloudy-sky model I had a
relatively large error, which demonstrated that the LST gap-filling strategy
adopted in this study introduced errors into the model to some extent,
thereby increasing the uncertainty in estimating <inline-formula><mml:math id="M140" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under cloudy-sky
conditions case I. The accuracy of the cloudy-sky model II was statistically
similar to that of the clear-sky model, and it could predict a moderate
temperature range close to 275 K with satisfactory performance. However, it
can be seen from the density scatterplot for cloudy-sky model II that some
discrete points deviated from the <inline-formula><mml:math id="M141" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line in the low-temperature range,
which indicated that there may be much uncertainty in predicting the
low-temperature range, especially at temperatures less than 260 K.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F6" specific-use="star"><?xmltex \currentcnt{6}?><?xmltex \def\figurename{Figure}?><label>Figure 6</label><caption><p id="d1e2252">Density scatterplots of the estimated <inline-formula><mml:math id="M142" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> against the in situ
<inline-formula><mml:math id="M143" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> for three models.</p></caption>
          <?xmltex \igopts{width=412.564961pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f06.png"/>

        </fig>

      <p id="d1e2283">Many studies have proved that land cover type and elevation have a
significant impact on the heterogeneity of <inline-formula><mml:math id="M144" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> (Benali et al., 2012;
Good et al., 2017; Lin et al., 2012; Marzban et al., 2017). Therefore, to
comprehensively analyze the performance of the <inline-formula><mml:math id="M145" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation models, we
grouped the results by land cover type and elevation range, and then
compared the model performance for different groups. The model performance
for different land cover types is listed in Table 3. All models showed
relatively good performance (RMSE <inline-formula><mml:math id="M146" display="inline"><mml:mo>&lt;</mml:mo></mml:math></inline-formula> 1.5 K) for cropland, shrubland,
water, and impervious surface, whereas RMSE values were higher for grassland
and bare land, which was consistent with the findings of Shen et al. (2020).
The model performance for different elevation ranges is also listed in Table 4. With the increase in elevation, RMSE values of all models had a certain
upward trend. However, as shown in the Fig. 7, the elevations of the stations
used in this study are mainly distributed in the range from 0 to 2000 m, so the
quantity of training samples in this elevation range have an absolute
superiority, whereas the samples from higher elevations (elevation <inline-formula><mml:math id="M147" display="inline"><mml:mo>&gt;</mml:mo></mml:math></inline-formula> 2000 m) only occupy a small part. The problem of class imbalance may
contribute to the relatively large errors when predicting <inline-formula><mml:math id="M148" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> at high
elevation. In addition, factors such as complex and varied topography,
vertical variation in <inline-formula><mml:math id="M149" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, and scale differences between remotely sensed
image pixels and station observation data points will lead to high
difficulty and uncertainty in <inline-formula><mml:math id="M150" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation at higher elevations (Rao
et al., 2019).</p>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T3" specific-use="star"><?xmltex \currentcnt{3}?><label>Table 3</label><caption><p id="d1e2359">Model performance for different land cover types.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="7">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right" colsep="1"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right" colsep="1"/>
     <oasis:colspec colnum="6" colname="col6" align="right"/>
     <oasis:colspec colnum="7" colname="col7" align="right"/>
     <oasis:thead>
       <oasis:row>
         <oasis:entry colname="col1">Land cover type</oasis:entry>
         <oasis:entry rowsep="1" namest="col2" nameend="col3" align="center" colsep="1">Clear-sky model </oasis:entry>
         <oasis:entry rowsep="1" namest="col4" nameend="col5" align="center" colsep="1">Cloudy-sky model I </oasis:entry>
         <oasis:entry rowsep="1" namest="col6" nameend="col7" align="center">Cloudy-sky model II </oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">%</oasis:entry>
         <oasis:entry colname="col3">RMSE (K)</oasis:entry>
         <oasis:entry colname="col4">%</oasis:entry>
         <oasis:entry colname="col5">RMSE (K)</oasis:entry>
         <oasis:entry colname="col6">%</oasis:entry>
         <oasis:entry colname="col7">RMSE (K)</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">Cropland</oasis:entry>
         <oasis:entry colname="col2">20.1</oasis:entry>
         <oasis:entry colname="col3">1.295</oasis:entry>
         <oasis:entry colname="col4">22.8</oasis:entry>
         <oasis:entry colname="col5">1.379</oasis:entry>
         <oasis:entry colname="col6">24.4</oasis:entry>
         <oasis:entry colname="col7">1.327</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Forest</oasis:entry>
         <oasis:entry colname="col2">10.4</oasis:entry>
         <oasis:entry colname="col3">1.375</oasis:entry>
         <oasis:entry colname="col4">11.1</oasis:entry>
         <oasis:entry colname="col5">1.502</oasis:entry>
         <oasis:entry colname="col6">15.3</oasis:entry>
         <oasis:entry colname="col7">1.421</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Grassland</oasis:entry>
         <oasis:entry colname="col2">26.0</oasis:entry>
         <oasis:entry colname="col3">1.420</oasis:entry>
         <oasis:entry colname="col4">22.4</oasis:entry>
         <oasis:entry colname="col5">1.550</oasis:entry>
         <oasis:entry colname="col6">17.3</oasis:entry>
         <oasis:entry colname="col7">1.540</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Shrubland</oasis:entry>
         <oasis:entry colname="col2">1.2</oasis:entry>
         <oasis:entry colname="col3">1.392</oasis:entry>
         <oasis:entry colname="col4">1.2</oasis:entry>
         <oasis:entry colname="col5">1.473</oasis:entry>
         <oasis:entry colname="col6">1.3</oasis:entry>
         <oasis:entry colname="col7">1.338</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Wetland</oasis:entry>
         <oasis:entry colname="col2">0.1</oasis:entry>
         <oasis:entry colname="col3">1.286</oasis:entry>
         <oasis:entry colname="col4">0.1</oasis:entry>
         <oasis:entry colname="col5">1.445</oasis:entry>
         <oasis:entry colname="col6">0.1</oasis:entry>
         <oasis:entry colname="col7">2.063</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Water</oasis:entry>
         <oasis:entry colname="col2">3.3</oasis:entry>
         <oasis:entry colname="col3">1.366</oasis:entry>
         <oasis:entry colname="col4">3.2</oasis:entry>
         <oasis:entry colname="col5">1.451</oasis:entry>
         <oasis:entry colname="col6">3.8</oasis:entry>
         <oasis:entry colname="col7">1.383</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Impervious surface</oasis:entry>
         <oasis:entry colname="col2">29.2</oasis:entry>
         <oasis:entry colname="col3">1.241</oasis:entry>
         <oasis:entry colname="col4">32.8</oasis:entry>
         <oasis:entry colname="col5">1.341</oasis:entry>
         <oasis:entry colname="col6">35.5</oasis:entry>
         <oasis:entry colname="col7">1.327</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Bare land</oasis:entry>
         <oasis:entry colname="col2">9.6</oasis:entry>
         <oasis:entry colname="col3">1.462</oasis:entry>
         <oasis:entry colname="col4">6.4</oasis:entry>
         <oasis:entry colname="col5">1.613</oasis:entry>
         <oasis:entry colname="col6">2.3</oasis:entry>
         <oasis:entry colname="col7">1.793</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <?xmltex \floatpos{t}?><fig id="Ch1.F7"><?xmltex \currentcnt{7}?><?xmltex \def\figurename{Figure}?><label>Figure 7</label><caption><p id="d1e2630">Elevation histogram of stations used in this study.</p></caption>
          <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f07.png"/>

        </fig>

      <p id="d1e2639">We further evaluated the error distribution of the three models at the
stations. Due to the absence of in situ <inline-formula><mml:math id="M151" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> at some ground stations on
some days, only the stations that recorded more than 20 d for all three
weather conditions were taken into account. Thus, the results of 2320 valid
stations were finally obtained, as shown in Table 5. In general, the models
showed good performance at most stations, with a mean RMSE value of 1.383 K.
Moreover, 97 % of stations had RMSE values less than 2 K and
only 1 of the 2320 statistical stations had an RMSE value greater than 3 K.
The clear-sky model also had the best performance at the station scale, with
the lowest mean RMSE of 1.231 K. A total of 508 stations had RMSE values less than
1 K, 2286 stations had RMSE values less than 2 K, while only 2 stations had
RMSE values greater than 3 K. For cloudy-sky model I, the mean RMSE reached
1.432 K. The RMSE values of 2256 stations were less than 2 K, and only 1
station had an RMSE greater than 3 K. For cloudy-sky model II, the mean RMSE
was 1.440 K, which was close to cloudy-sky model I, and 121 stations had RMSE values
less than 1 K. However, 13 stations had RMSE values greater than 3 K for
cloudy-sky model II, and most of these stations had RMSE values less than 3 K for the other two models.</p>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T4" specific-use="star"><?xmltex \currentcnt{4}?><label>Table 4</label><caption><p id="d1e2657">Model performance for different elevation ranges.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="7">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right" colsep="1"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right" colsep="1"/>
     <oasis:colspec colnum="6" colname="col6" align="right"/>
     <oasis:colspec colnum="7" colname="col7" align="right"/>
     <oasis:thead>
       <oasis:row>
         <oasis:entry colname="col1">Elevation (m)</oasis:entry>
         <oasis:entry rowsep="1" namest="col2" nameend="col3" align="center" colsep="1">Clear-sky model </oasis:entry>
         <oasis:entry rowsep="1" namest="col4" nameend="col5" align="center" colsep="1">Cloudy-sky model I </oasis:entry>
         <oasis:entry rowsep="1" namest="col6" nameend="col7" align="center">Cloudy-sky model II </oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">%</oasis:entry>
         <oasis:entry colname="col3">RMSE (K)</oasis:entry>
         <oasis:entry colname="col4">%</oasis:entry>
         <oasis:entry colname="col5">RMSE (K)</oasis:entry>
         <oasis:entry colname="col6">%</oasis:entry>
         <oasis:entry colname="col7">RMSE (K)</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1"><inline-formula><mml:math id="M152" display="inline"><mml:mo>&lt;</mml:mo></mml:math></inline-formula> 1000</oasis:entry>
         <oasis:entry colname="col2">61.8</oasis:entry>
         <oasis:entry colname="col3">1.281</oasis:entry>
         <oasis:entry colname="col4">71.1</oasis:entry>
         <oasis:entry colname="col5">1.381</oasis:entry>
         <oasis:entry colname="col6">82.4</oasis:entry>
         <oasis:entry colname="col7">1.363</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">1000–2000</oasis:entry>
         <oasis:entry colname="col2">24.6</oasis:entry>
         <oasis:entry colname="col3">1.372</oasis:entry>
         <oasis:entry colname="col4">20.0</oasis:entry>
         <oasis:entry colname="col5">1.538</oasis:entry>
         <oasis:entry colname="col6">14.2</oasis:entry>
         <oasis:entry colname="col7">1.511</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">2000–3000</oasis:entry>
         <oasis:entry colname="col2">6.1</oasis:entry>
         <oasis:entry colname="col3">1.472</oasis:entry>
         <oasis:entry colname="col4">4.2</oasis:entry>
         <oasis:entry colname="col5">1.68</oasis:entry>
         <oasis:entry colname="col6">1.7</oasis:entry>
         <oasis:entry colname="col7">1.637</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">3000–4000</oasis:entry>
         <oasis:entry colname="col2">4.7</oasis:entry>
         <oasis:entry colname="col3">1.547</oasis:entry>
         <oasis:entry colname="col4">3.0</oasis:entry>
         <oasis:entry colname="col5">1.619</oasis:entry>
         <oasis:entry colname="col6">1.1</oasis:entry>
         <oasis:entry colname="col7">1.614</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"><inline-formula><mml:math id="M153" display="inline"><mml:mo>&gt;</mml:mo></mml:math></inline-formula> 4000</oasis:entry>
         <oasis:entry colname="col2">2.8</oasis:entry>
         <oasis:entry colname="col3">1.678</oasis:entry>
         <oasis:entry colname="col4">1.7</oasis:entry>
         <oasis:entry colname="col5">1.673</oasis:entry>
         <oasis:entry colname="col6">0.6</oasis:entry>
         <oasis:entry colname="col7">1.768</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T5" specific-use="star"><?xmltex \currentcnt{5}?><label>Table 5</label><caption><p id="d1e2865">Error distributions of three models at the stations.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="6">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:colspec colnum="6" colname="col6" align="right"/>
     <oasis:thead>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry rowsep="1" namest="col2" nameend="col6" align="center">RMSE </oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Model</oasis:entry>
         <oasis:entry colname="col2">Mean (K)</oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M154" display="inline"><mml:mo>&lt;</mml:mo></mml:math></inline-formula> 1 K</oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M155" display="inline"><mml:mo>&lt;</mml:mo></mml:math></inline-formula> 2 K</oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M156" display="inline"><mml:mo>&lt;</mml:mo></mml:math></inline-formula> 3 K</oasis:entry>
         <oasis:entry colname="col6"><inline-formula><mml:math id="M157" display="inline"><mml:mrow><mml:mo>≥</mml:mo><mml:mn mathvariant="normal">3</mml:mn></mml:mrow></mml:math></inline-formula> K</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">Clear-sky model</oasis:entry>
         <oasis:entry colname="col2">1.231</oasis:entry>
         <oasis:entry colname="col3">508</oasis:entry>
         <oasis:entry colname="col4">2286</oasis:entry>
         <oasis:entry colname="col5">2318</oasis:entry>
         <oasis:entry colname="col6">2</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Cloudy-sky model I</oasis:entry>
         <oasis:entry colname="col2">1.432</oasis:entry>
         <oasis:entry colname="col3">70</oasis:entry>
         <oasis:entry colname="col4">2256</oasis:entry>
         <oasis:entry colname="col5">2319</oasis:entry>
         <oasis:entry colname="col6">1</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Cloudy-sky model II</oasis:entry>
         <oasis:entry colname="col2">1.440</oasis:entry>
         <oasis:entry colname="col3">121</oasis:entry>
         <oasis:entry colname="col4">2099</oasis:entry>
         <oasis:entry colname="col5">2307</oasis:entry>
         <oasis:entry colname="col6">13</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">All</oasis:entry>
         <oasis:entry colname="col2">1.383</oasis:entry>
         <oasis:entry colname="col3">80</oasis:entry>
         <oasis:entry colname="col4">2249</oasis:entry>
         <oasis:entry colname="col5">2319</oasis:entry>
         <oasis:entry colname="col6">1</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <p id="d1e3037">For model comparison, as expected, the clear-sky model that used absolutely
clear-sky LSTs performed better than cloudy-sky model I and cloudy-sky model
II in almost every aspect and presented the highest stability. Cloudy-sky
model I, which contained gap-filled LSTs, did not perform as well as the
clear-sky model because, although the time threshold (<inline-formula><mml:math id="M158" display="inline"><mml:mo lspace="0mm">±</mml:mo></mml:math></inline-formula>2 d) of the
LST gap-filling method was relatively small, the LST value of a missing
pixel of a date may be replaced by a clear-sky value with a difference of up
to 2 d. However, the LST can vary considerably in just a few days, so the
LST gap-filling process can introduce large errors into the model, thereby
affecting the accuracy of <inline-formula><mml:math id="M159" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation. Surprisingly, cloudy-sky
model II, which did not use LST<?pagebreak page4250?> features, achieved a comparative accuracy with
the clear-sky model (respective RMSE values of 1.396 K vs. 1.342 K) statistically. However,
when we further analyzed the model performance in specific situations, we
detected differences in the performance of the three models. There may
be considerable uncertainty associated with the cloudy-sky model II with respect to predicting the low-temperature range, especially at less than 260 K. Notably, cloudy-sky
model II performed poorly for wetlands, with an RMSE of 2.063 K, whereas both the
clear-sky model and cloudy-sky model I performed well on this type of land
cover. This may be because wetlands are a mixture of water and land, with
diverse complex ecological environments. Using LST can significantly improve
the <inline-formula><mml:math id="M160" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation accuracy of this land cover type.</p>
      <p id="d1e3069">In summary, because of the strong correlation between <inline-formula><mml:math id="M161" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and LST,
adding daily LSTs as features to models can improve the model stability and
robustness. In the absence of LST, assimilated <inline-formula><mml:math id="M162" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> can be used as a
substitute for LST to provide an initial value or first guess for the model
to estimate <inline-formula><mml:math id="M163" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> with acceptable accuracy when combined with other
features. However, the resolution of the reanalysis product is relatively
coarse, and some local details were ignored when sampling from a larger
scale (0.25<inline-formula><mml:math id="M164" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> <inline-formula><mml:math id="M165" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 0.25<inline-formula><mml:math id="M166" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>) to a smaller scale (1 km <inline-formula><mml:math id="M167" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 1 km), thereby causing explicit uncertainties for cloudy-sky model II
with respect to predicting the low-temperature range or some regions, especially some specific
land cover types or regions with complex terrain. Overall, none of the three
models showed significant differences in the model performance, and the
model performance<?pagebreak page4251?> discrepancies for different land cover types and elevation
ranges were acceptable. The proposed models can perform well in different
situations and are suitable for <inline-formula><mml:math id="M168" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation under different weather
conditions.</p>
</sec>
<sec id="Ch1.S4.SS2">
  <label>4.2</label><title>Cross-validation</title>
      <p id="d1e3157">In addition to random sample validation, two CV methods were used to further
evaluate model performance. For the LTO CV, we divided the data pairs from 2003
to 2016 into 14 groups by calendar year. In each iteration, 13 groups of
data were used as a training set for model training, and the remaining 1
group of data was used for validation. The modeling and validation process
were repeated 14 times until the data for each year were validated. The results are
shown in Fig. 8. The RMSE values of validation results for different groups
of data ranged from 1.359 to 1.665 K. The minor difference between the LTO
CV results proved that these models have good extensibility in time.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F8" specific-use="star"><?xmltex \currentcnt{8}?><?xmltex \def\figurename{Figure}?><label>Figure 8</label><caption><p id="d1e3162">Density scatterplots of the LTO CV results for three models.</p></caption>
          <?xmltex \igopts{width=412.564961pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f08.png"/>

        </fig>

      <p id="d1e3171">For the LLO CV, we then divided seven clusters in the Chinese region using the
similar separation strategy of Xiao et al. (2018). Stations used in this
study were divided into different clusters according to their spatial
locations, and all data pairs were divided into seven groups according to the
cluster of station. In each iteration, six groups of data were used as
training sets and the remaining one group of data was used for validation.
The modeling and validation process were repeated seven times until the data of
each group were validated. The total validation results of the models under
the three weather conditions are shown in Fig. 9, with RMSE values ranging from
1.615 to 1.957 K. As expected, the error of the LLO CV increased relative to
random sample validation. This is because the relationship between <inline-formula><mml:math id="M169" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
and other features varies with geographical location. The prediction error
of the northwest and southwest clusters was larger than that of other
clusters. The RMSE values of these two clusters exceeded 2.5 K under cloudy-sky
conditions case II, whereas the RMSE values of the other clusters were about 1.5 K.
This is consistent with the analysis of the spatial distribution of model
accuracy in Sect. 4.4 of the paper. The<?pagebreak page4252?> meteorological stations in
northwestern and southwestern China are distributed discretely and at a distance from
other stations in China, leading to a large difference between the training
set and the test set and, ultimately, resulting in the relatively poor
performance in the LLO CV strategy in these two regions. Furthermore, the
LLO CV results of cloudy-sky model II were worse than those of the
clear-sky model and cloudy-sky model I, indicating that LSTs help to reduce
the spatial overfitting of the models.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F9" specific-use="star"><?xmltex \currentcnt{9}?><?xmltex \def\figurename{Figure}?><label>Figure 9</label><caption><p id="d1e3188">Density scatterplots of the LLO CV results for three models.</p></caption>
          <?xmltex \igopts{width=412.564961pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f09.png"/>

        </fig>

</sec>
<sec id="Ch1.S4.SS3">
  <label>4.3</label><title>Feature importance analysis</title>
      <p id="d1e3205">To quantitatively evaluate the contribution of each feature included in the
RF models, the FI of every feature for the three models was calculated using the
permutation method described in Sect. 3.4 and was then ranked. To reduce the
impact of contingency on the experimental results, we repeated the
experiment 30 times and took the average value of all experimental results
as the final FI of each feature for each model. The FI results are shown in
Fig. 10, with the importance decreasing from top to bottom. The gray line
indicates the FI range of each feature for multiple repeated experiments.
All features are divided into four types and represented by different
colors, among which the blue rectangles represent MODIS LSTs, the orange
rectangles represent GLDAS-assimilated <inline-formula><mml:math id="M170" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, the red rectangles represent
radiation products including DSR and ALB, and the green rectangles represent
other features.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F10" specific-use="star"><?xmltex \currentcnt{10}?><?xmltex \def\figurename{Figure}?><label>Figure 10</label><caption><p id="d1e3221">FI of each feature for three models.</p></caption>
          <?xmltex \igopts{width=497.923228pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f10.png"/>

        </fig>

      <?pagebreak page4253?><p id="d1e3230">For clear-sky model, Terra nighttime LST was of the highest importance (FI
of 2.92), followed by assimilated <inline-formula><mml:math id="M171" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> (FI of 2.48), indicating that
the prediction accuracy of the clear-sky model was significantly reduced
after permuting these two features. They were followed by Aqua nighttime LST
(FI of 1.3) and two daytime LSTs (FI of 0.49 and 0.21, respectively). For
cloudy-sky model I, assimilated <inline-formula><mml:math id="M172" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> ranked first (FI of 4.59), followed
by Terra nighttime LST (FI of 1.03). For cloudy-sky model II, which did not
include LST as features, assimilated <inline-formula><mml:math id="M173" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> played a more importance role
(FI of 6.65) than it did for cloudy-sky model I. The FI of radiation
products and other features were all less than one for all the models, showing
that they only slightly improved the model performance.</p>
      <p id="d1e3267">The energy exchange between the land surface and the near-surface atmosphere
takes the form of longwave radiation, evapotranspiration, and turbulent
exchange, or other phenomena. LST and land surface emissivity (LSE)
determine the longwave radiation in land surface radiation and energy
budgets (Liang and Wang, 2019). Thus, there is a strong and complicated
physical correlation between LST and <inline-formula><mml:math id="M174" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. It can be seen from Fig. 8
that all four daily LSTs, especially nighttime LSTs, had relatively high FI values
for both the clear-sky model and cloudy-sky model I. Among all the daily LSTs,
nighttime LSTs outweighed daytime LSTs, and the Terra nighttime LST was of
higher importance than the Aqua nighttime LST, which was consistent with the
findings of many studies (Benali et al., 2012; Li and Zha, 2019; Zhang et
al., 2011). In the study by Lin et al. (2012), the
MAE between LST and <inline-formula><mml:math id="M175" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> during the day and during the night were
calculated separately, and they found that there was better agreement between LST
and <inline-formula><mml:math id="M176" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> during the night. In addition, due to the
lack of solar radiation and its influence on the thermal infrared signal,
remotely sensed nighttime LST products usually have higher stability (Benali
et al., 2012; Vancutsem et al., 2010).</p>
      <?pagebreak page4254?><p id="d1e3303">Assimilated <inline-formula><mml:math id="M177" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> also mattered considerably for <inline-formula><mml:math id="M178" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation
models. Its FI was second only to Terra nighttime LST for the clear-sky model
and highest for cloudy-sky model I and cloudy-sky model II. For cloudy-sky
model I, originally missed LSTs were replaced by clear-sky values from a
nearby date, and the error introduced by this simple LST gap-filling strategy
resulted in a decrease in the overall LST accuracy, thereby causing the FI
of assimilated <inline-formula><mml:math id="M179" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> to exceed that of LSTs. Compared with cloudy-sky
model I, assimilated <inline-formula><mml:math id="M180" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> was of higher importance, with a FI of 6.65 for
cloudy-sky model II, indicating that it became the absolute dominant factor
in <inline-formula><mml:math id="M181" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation when LST was not included in the <inline-formula><mml:math id="M182" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation
model. Cloudy-sky model II also achieved satisfactory accuracy in the
validation results. This demonstrates that although the spatial resolution
of the assimilated <inline-formula><mml:math id="M183" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is relatively coarse, it can be a supplement
or substitute for MODIS LSTs and provide an initial value or first guess
for models to predict <inline-formula><mml:math id="M184" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> with a higher resolution.</p>
      <p id="d1e3395">Radiation products and other features helped to improve the accuracy of
<inline-formula><mml:math id="M185" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation models to a small extent. Among them, latitude,
longitude, elevation, and day of year had relatively high importance in all
three models. Latitude and longitude determine the relative position of the
sun, which influences day length and, thus, the distribution of total solar
radiation that the surface receives throughout the year; this, in turn, affects
the patterns of <inline-formula><mml:math id="M186" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> (Benali et al., 2012). Elevation affects how the
ground is heated and how much radiation energy is absorbed by the
atmosphere, resulting in vertical variations in <inline-formula><mml:math id="M187" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. In addition, the
relationship between <inline-formula><mml:math id="M188" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and LST has great heterogeneity in different
regions and at different times and is greatly affected by surface
characteristics and atmospheric conditions. The day of year helps to explain
the seasonal changes in atmospheric physical conditions, chemical
composition, and surface characteristics to distinguish the different
relationships between <inline-formula><mml:math id="M189" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and LST in different seasons and subsequently improve
the accuracy of <inline-formula><mml:math id="M190" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation (Yao et al., 2020; Zhang et al., 2011).
For LAI, DSR, and ALB, it is likely that other collinear features in the
models made the information provided by them redundant, so their FI was
relatively low in the <inline-formula><mml:math id="M191" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation models. However, in the analysis of
the results of some stations, it was found that adding radiation features to
the models helped improve the <inline-formula><mml:math id="M192" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation accuracy on some days. The
radiation features can play a supplementary role in the case of some other
features that do not perform well. Therefore, we finally decided to retain
the radiation features in the <inline-formula><mml:math id="M193" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation models.</p>
</sec>
<sec id="Ch1.S4.SS4">
  <label>4.4</label><title>Spatial distribution of accuracy</title>
      <?pagebreak page4255?><p id="d1e3506">The RMSE value was calculated for each meteorological station that recorded
more than 20 d for all three weather conditions. To obtain a deeper
understanding of the spatial distribution of model performance, the RMSE
spatial distribution of stations for the three models was mapped, as shown
in Fig. 11. It is evident that the model performance varied at different
geographical locations for all three models. The clear-sky model presented
the most stable results in different regions compared with cloudy-sky model
I and cloudy-sky model II, with RMSE values of all stations ranging from
0.566 to 3.453 K. The RMSE range of cloudy-sky model I was 0.823–4.370 K
and that of cloudy-sky model II was 0.809–4.198 K. The spatial patterns of
cloudy-sky model I and cloudy-sky model II were generally similar, but there were more stations with good performance (RMSE <inline-formula><mml:math id="M194" display="inline"><mml:mo>&lt;</mml:mo></mml:math></inline-formula> 1 K) and poor performance (RMSE <inline-formula><mml:math id="M195" display="inline"><mml:mo>&gt;</mml:mo></mml:math></inline-formula> 3 K) for
cloudy-sky model II, showing
relatively poor stability.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F11" specific-use="star"><?xmltex \currentcnt{11}?><?xmltex \def\figurename{Figure}?><label>Figure 11</label><caption><p id="d1e3525">RMSE spatial distribution of stations for three models.</p></caption>
          <?xmltex \igopts{width=497.923228pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f11.png"/>

        </fig>

      <p id="d1e3534">Overall, the stations in central, eastern, and southern China presented high
levels of accuracy for all three models, with RMSE values of most stations
in these places of less than 1.5 K. Most stations with large RMSE values were
located in southwest, northwest, and northern China, which was consistent
with the results of Shen et al. (2020), and the RMSE values of cloudy-sky
model II in these positions were larger than those of the clear-sky model and
cloudy-sky model I. On the one hand, the spatial heterogeneity of model
performance is largely due to the uneven distribution density of
meteorological stations. As can be seen from the geographical locations of
the meteorological stations used in this study (Fig. 1), it is obvious that
stations in central, eastern, and southern China are densely distributed,
whereas stations in northern and western China are relatively rare, which may
contribute to the uneven distribution of model performance. Additionally,
the terrain environment in central, eastern, and southern China is not
complex, whereas high elevation and some climate types will increase the
uncertainty of <inline-formula><mml:math id="M196" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation in northern and western China. The climate
types of stations with poor performance were mostly temperate continental
and plateau mountain climates, and the land cover types were mainly bare
land and grassland. It can be seen from Table 4 that cloudy-sky model II
showed relatively poor performance for these two land cover types.
Therefore, there was an explicit uncertainty when only assimilated <inline-formula><mml:math id="M197" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and
other features except LSTs were included to predict <inline-formula><mml:math id="M198" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in places with
these climate and land cover types. Overall, although the spatial
distribution of the model performance was relatively uneven, the <inline-formula><mml:math id="M199" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
estimation models for different weather conditions all showed satisfactory
performance.</p>
</sec>
<sec id="Ch1.S4.SS5">
  <label>4.5</label><title>Seasonal distribution of accuracy</title>
      <p id="d1e3590">The model performance at the monthly scale was also evaluated, and the RMSE
monthly distribution for the three models is shown in Fig. 12. The RMSE
range of the clear-sky model was 1.109–1.508 K, the cloudy-sky model I was
1.178–1.692 K, and the cloudy-sky model II was 1.056–1.777 K. It is obvious
that there was temporal heterogeneity in the model performance, and the
estimation accuracy presented similar seasonal variation patterns for all
three models. The RMSE values were lower in summer and autumn, and higher in
spring and winter, reaching a peak in February and reaching a bottom in July
or August. We can conclude that models performed better on warm days, with
RMSE values for all three models of below 1.22 K in July and August. This
finding was consistent with the validation results at the monthly scale of
Yao et al. (2020) and Li and Zha (2019). This phenomenon may be partly due
to the fact that China is vast in territory with a latitudinal difference
between the northernmost station and the southernmost stations of about
30<inline-formula><mml:math id="M200" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>; therefore, the range of <inline-formula><mml:math id="M201" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is wider on cold days than on hot
days.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F12"><?xmltex \currentcnt{12}?><?xmltex \def\figurename{Figure}?><label>Figure 12</label><caption><p id="d1e3615">RMSE monthly distribution for three models.</p></caption>
          <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f12.png"/>

        </fig>

      <p id="d1e3624">Monthly differences in model performance also indicated that the
relationship between <inline-formula><mml:math id="M202" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and other factors varied seasonally and may
have been more consistent in the same month. It was confirmed in the
research of Yao et al. (2020) that modeling data of the same month together
could achieve more accurate results. Therefore, although day of year was
used in the modeling in this study, this temporal difference was not
completely eliminated. Modeling the datasets of all seasons together in this
study may increase the temporal heterogeneity of accuracy. It is worthwhile
considering grouping the data of the same month to establish monthly models
in the future, which may be conducive to further improving the accuracy of
<inline-formula><mml:math id="M203" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation.</p>
</sec>
</sec>
<sec id="Ch1.S5">
  <label>5</label><title>Comparison with existing datasets</title>
      <p id="d1e3659">For a more comprehensive evaluation of the estimated daily mean <inline-formula><mml:math id="M204" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, we
compared it with three reanalysis and meteorological forcing datasets,
including CLDAS, CMFD, and GLDAS, in terms of validation statistics and
spatiotemporal patterns. The station observations in 2010 were used to
validate the accuracy of these four <inline-formula><mml:math id="M205" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> datasets. It should be noted
that we estimated daily mean <inline-formula><mml:math id="M206" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> for the period ending at local midnight
rather than 24:00 UTC. To ensure the time consistency, we calculated the
average value of all simulations on a local day as the daily mean <inline-formula><mml:math id="M207" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
for the reanalysis and meteorological forcing datasets. The statistical
results and the density scatterplots are shown in Table 6 and Fig. 13,
respectively. It can be seen that, compared with the reanalysis datasets, the
RF <inline-formula><mml:math id="M208" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> presented the highest consistency with the station observations,
with the best performance in all accuracy assessment criteria (<inline-formula><mml:math id="M209" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, MAE,
RMSE, and bias values were 0.992, 0.680 K, 1.010 K, and 0.063 K,
respectively). The points in the density scatterplot of the RF <inline-formula><mml:math id="M210" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> were
more concentrated near the <inline-formula><mml:math id="M211" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line. CLDAS <inline-formula><mml:math id="M212" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and CMFD <inline-formula><mml:math id="M213" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> both
showed almost zero bias with the station observations, but their RMSE values
were both close to 2 K. GLDAS <inline-formula><mml:math id="M214" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> reported slight underestimation (bias
of 0.900 K). In general, this comparison confirmed the applicability of the RF
method in <inline-formula><mml:math id="M215" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation and the higher accuracy of our estimated
<inline-formula><mml:math id="M216" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> compared with the reanalysis products.</p>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T6"><?xmltex \currentcnt{6}?><label>Table 6</label><caption><p id="d1e3811">Evaluation results of four datasets in 2010.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="5">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"><inline-formula><mml:math id="M217" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M218" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">MAE (K)</oasis:entry>
         <oasis:entry colname="col4">RMSE (K)</oasis:entry>
         <oasis:entry colname="col5">Bias (K)</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">RF</oasis:entry>
         <oasis:entry colname="col2">0.992</oasis:entry>
         <oasis:entry colname="col3">0.680</oasis:entry>
         <oasis:entry colname="col4">1.010</oasis:entry>
         <oasis:entry colname="col5">0.063</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">CLDAS</oasis:entry>
         <oasis:entry colname="col2">0.972</oasis:entry>
         <oasis:entry colname="col3">1.427</oasis:entry>
         <oasis:entry colname="col4">1.938</oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M219" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>0.078</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">CMFD</oasis:entry>
         <oasis:entry colname="col2">0.962</oasis:entry>
         <oasis:entry colname="col3">1.642</oasis:entry>
         <oasis:entry colname="col4">2.242</oasis:entry>
         <oasis:entry colname="col5">0.092</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">GLDAS</oasis:entry>
         <oasis:entry colname="col2">0.938</oasis:entry>
         <oasis:entry colname="col3">2.160</oasis:entry>
         <oasis:entry colname="col4">2.874</oasis:entry>
         <oasis:entry colname="col5">0.900</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <?xmltex \floatpos{t}?><fig id="Ch1.F13" specific-use="star"><?xmltex \currentcnt{13}?><?xmltex \def\figurename{Figure}?><label>Figure 13</label><caption><p id="d1e3952">Density scatterplots of the estimated <inline-formula><mml:math id="M220" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and reanalysis
<inline-formula><mml:math id="M221" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> against the in situ <inline-formula><mml:math id="M222" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in 2010.</p></caption>
        <?xmltex \igopts{width=384.112205pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f13.png"/>

      </fig>

      <p id="d1e3995">In addition, the spatiotemporal patterns of these four <inline-formula><mml:math id="M223" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> datasets were
compared. We calculated the monthly mean <inline-formula><mml:math id="M224" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in 2010 for all datasets.
The RF monthly mean land <inline-formula><mml:math id="M225" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> mappings over mainland China in February,
May,<?pagebreak page4256?> August, and November 2010 are shown in Fig. 14a–d. The CLDAS (Fig. 14e–h), CMFD (Fig. 14i–l), and GLDAS (Fig. 14m–p) monthly mean
<inline-formula><mml:math id="M226" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> mappings in the same months are also shown in Fig. 12. The spatial
resolutions of RF, CLDAS, CMFD, and GLDAS monthly mean <inline-formula><mml:math id="M227" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> are
approximately 0.01<inline-formula><mml:math id="M228" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> <inline-formula><mml:math id="M229" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 0.01<inline-formula><mml:math id="M230" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>, 0.0625<inline-formula><mml:math id="M231" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> <inline-formula><mml:math id="M232" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 0.0625<inline-formula><mml:math id="M233" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>, 0.1<inline-formula><mml:math id="M234" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> <inline-formula><mml:math id="M235" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 0.1<inline-formula><mml:math id="M236" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>,
and 0.25<inline-formula><mml:math id="M237" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> <inline-formula><mml:math id="M238" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 0.25<inline-formula><mml:math id="M239" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>, respectively. We used GLDAS-assimilated <inline-formula><mml:math id="M240" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and GLASS LAI in <inline-formula><mml:math id="M241" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation, which have no value
in most water bodies; thus, the <inline-formula><mml:math id="M242" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> of these areas was also not estimated.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F14" specific-use="star"><?xmltex \currentcnt{14}?><?xmltex \def\figurename{Figure}?><label>Figure 14</label><caption><p id="d1e4191">Maps of monthly mean <inline-formula><mml:math id="M243" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> over mainland China. Panels <bold>(a)</bold>–<bold>(d)</bold> show the
the RF <inline-formula><mml:math id="M244" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, panels <bold>(e)</bold>–<bold>(h)</bold> show the CLDAS <inline-formula><mml:math id="M245" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, panels <bold>(i)</bold>–<bold>(l)</bold> show the CMFD <inline-formula><mml:math id="M246" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, and panels <bold>(m)</bold>–<bold>(p)</bold> show the GLDAS <inline-formula><mml:math id="M247" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in February, May, August, and November 2010,
respectively. The white pixels in mainland China indicate no data value and are always water bodies.</p></caption>
        <?xmltex \igopts{width=497.923228pt}?><graphic xlink:href="https://essd.copernicus.org/articles/13/4241/2021/essd-13-4241-2021-f14.png"/>

      </fig>

      <?pagebreak page4257?><p id="d1e4281">As can be seen from Fig. 14, it is clear that these four datasets basically
showed a high degree of consistency in the spatiotemporal patterns over
mainland China. China has a vast territory, and its topography is
high in the west and low in the east. The spatial patterns of <inline-formula><mml:math id="M248" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> over
mainland China present great seasonal heterogeneity. In winter, the sun
shines directly in the Southern Hemisphere, and the Northern Hemisphere
consequentially receives less solar energy. The <inline-formula><mml:math id="M249" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in northern China
and the Tibetan Plateau are generally low, and the <inline-formula><mml:math id="M250" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> difference between
the north and the south exceeds 50 K. On the contrary, in summer, as the sun
shines directly in the Northern Hemisphere, <inline-formula><mml:math id="M251" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in most parts of China
are generally high except for the Tibetan Plateau, with little <inline-formula><mml:math id="M252" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
difference between the north and the south. As an expectable consequence of
higher spatial resolution, the RF <inline-formula><mml:math id="M253" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> mappings were capable of providing
more detail on the <inline-formula><mml:math id="M254" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> spatial patterns than the reanalysis and
meteorological forcing <inline-formula><mml:math id="M255" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, especially in mountainous areas with
complicated terrain. GLDAS <inline-formula><mml:math id="M256" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> presented an obvious pixel effect because
of the relatively coarse spatial resolution. In summary, the all-sky daily
mean land <inline-formula><mml:math id="M257" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> product developed in this study has achieved satisfactory
accuracy and high spatial resolution simultaneously, which can reveal the
seasonal variation trend and the spatial patterns of <inline-formula><mml:math id="M258" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> over China
well. This product can provide a long time series of daily mean <inline-formula><mml:math id="M259" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> with
the spatial resolution of 1 km over mainland China, which fills the current
dataset gap in this field. Moreover, this product is also conducive to
observing and analyzing the climate characteristic of China and plays an
important role in the studies of climate change and the hydrological cycle.</p>
</sec>
<sec id="Ch1.S6">
  <label>6</label><title>Data availability</title>
      <p id="d1e4426">The daily mean land <inline-formula><mml:math id="M260" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> product over mainland China is currently freely available
at <ext-link xlink:href="https://doi.org/10.5281/zenodo.4399453" ext-link-type="DOI">10.5281/zenodo.4399453</ext-link> for the period from 2003 to
2008 (Chen et al., 2021b) and  at the University of Maryland
(<uri>http://glass.umd.edu/Ta_China/</uri>, last access: 24 August 2021) for the period from 2003 to 2019.
In order to make this big dataset easier to understand and use, we created a
provincial sub-dataset with a smaller geographic coverage. An all-sky
0.01<inline-formula><mml:math id="M261" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> daily <inline-formula><mml:math id="M262" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> product over Beijing (2003–2019) was
generated from the developed dataset over mainland China after resampling
and clipping, and it is publicly available at
<ext-link xlink:href="https://doi.org/10.5281/zenodo.4405123" ext-link-type="DOI">10.5281/zenodo.4405123</ext-link> (Chen et al., 2021a).</p>
      <p id="d1e4470">The MODIS product and the GLDAS dataset were downloaded from
<uri>https://earthdata.nasa.gov/</uri> (last access: 24 August 2021). The GLASS products were downloaded from
<uri>http://www.glass.umd.edu</uri> (last access: 24 August 2021). The CLDAS and CMFD datasets were downloaded from
<uri>http://tipex.data.cma.cn</uri> (last access: 24 August 2021) and <uri>http://data.tpdc.ac.cn/</uri> (last access: 24 August 2021), respectively.</p>
</sec>
<sec id="Ch1.S7" sec-type="conclusions">
  <label>7</label><title>Conclusions</title>
      <?pagebreak page4258?><p id="d1e4493"><inline-formula><mml:math id="M263" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is a key variable in climate and global change research. In this
study, we developed an all-sky 1 km daily mean land <inline-formula><mml:math id="M264" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> product for
2003–2019 over mainland China mainly based on MODIS and GLDAS data using
the RF method. An efficient temporal gap-filling method was first used to
fill MODIS LST gaps under cloudy-sky conditions. We predicted <inline-formula><mml:math id="M265" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> under
three different weather conditions separately: clear-sky conditions (when
the daily LSTs are all clear-sky), cloudy-sky conditions case I (when the
daily LST gap(s) can be filled), and cloudy-sky conditions case II (when the
daily LST gap(s) cannot all be filled). The validation results using station
measurements (one-fifth of the total data from 2003 to 2016 selected randomly),
which were not used for model training, showed that the <inline-formula><mml:math id="M266" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> values were
0.986, 0.984, and 0.984 and the RMSE values were 1.342, 1.440, and 1.396 K for the
clear-sky model, cloudy-sky model I, and cloudy-sky model II, respectively.
In general, the models showed excellent performance at most stations, with a
mean RMSE of 1.383 K, and there were 97 % stations with RMSE values less
than 2 K and only 1 of 2320 stations with an RMSE value greater than 3 K. In
addition, we examined the spatiotemporal patterns and land cover type
dependences of model accuracy and concluded that model performance under all
conditions was acceptable overall, despite some heterogeneity under
different conditions. The relative contributions of different features to
models were also quantitatively analyzed, and it was found that LST and
assimilated <inline-formula><mml:math id="M267" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> were of great significance in <inline-formula><mml:math id="M268" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> estimation.
Finally, we compared the <inline-formula><mml:math id="M269" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> dataset in 2010 with the CLDAS, CMFD, and GLDAS
datasets, finding great consistency in the spatiotemporal patterns. The
estimated <inline-formula><mml:math id="M270" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> in 2010 reported significantly higher accuracy against the
station observations, with <inline-formula><mml:math id="M271" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, RMSE, and bias values of 0.992, 1.010 K,
and 0.063 K, respectively.</p>
      <p id="d1e4595">Overall, this study developed a robust scheme that used a machine learning method
to estimate all-sky daily mean <inline-formula><mml:math id="M272" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> over a large spatial and temporal
range. This approach can be applied globally. The generated all-sky <inline-formula><mml:math id="M273" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>
product has achieved a high degree of accuracy compared with the existing
datasets, which fills the current dataset gap in this field and plays an
important role in many scientific fields such as<?pagebreak page4259?> climate change,
the hydrological cycle, and the energy balance. Future work should focus on
developing better LST gap-filling methods and experimenting with more advanced
deep learning methods that take the spatial and temporal
dependence of <inline-formula><mml:math id="M274" display="inline"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">a</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> into account.</p>
</sec>

      
      </body>
    <back><notes notes-type="authorcontribution"><title>Author contributions</title>

      <p id="d1e4635">SL and YC contributed to the design of this study and developed the overall
methodology. HM, BL, and YC collected and preprocessed the data. YC carried
out the experiments. YC, BL, TH, and QW produced the product. YC wrote the
first draft of the paper. All authors revised the paper.</p>
  </notes><notes notes-type="competinginterests"><title>Competing interests</title>

      <p id="d1e4641">The authors declare that they have no conflict of interest.</p>
  </notes><notes notes-type="disclaimer"><title>Disclaimer</title>

      <p id="d1e4647">Publisher's note: Copernicus Publications remains neutral with regard to jurisdictional claims in published maps and institutional affiliations.</p>
  </notes><ack><title>Acknowledgements</title><p id="d1e4653">We gratefully acknowledge data support from the “National Earth System
Science Data Center, National Science &amp; Technology Infrastructure of
China” (<uri>http://www.geodata.cn</uri>, last access: 24 August 2021). We thank the GLASS team for providing the
data used in this study, which can be downloaded from <uri>https://www.glass.umd.edu</uri> (last access: 24 August 2021). We
are grateful to the National Aeronautics and Space Administration team for
providing the MODIS product and GLDAS data, which can be freely download from
<uri>https://earthdata.nasa.gov/</uri> (last access: 24 August 2021). We also thank the CLDAS and CMFD teams for
providing the respective CLDAS and CMFD datasets, which can be freely download from <uri>http://tipex.data.cma.cn</uri> (last access: 24 August 2021) and <uri>http://data.tpdc.ac.cn/</uri> (last access: 24 August 2021),
respectively. Additionally, the authors would like to acknowledge the Chinese
Meteorological Administration for providing in situ measurements.
We are also very grateful to the reviewers for their valuable comments
and suggestions.</p></ack><notes notes-type="financialsupport"><title>Financial support</title>

      <p id="d1e4673">This study was partially supported by the Chinese Grand Research Program on
Climate Change and Response under project no 2016YFA0600103.</p>
  </notes><notes notes-type="reviewstatement"><title>Review statement</title>

      <p id="d1e4680">This paper was edited by David Carlson and reviewed by two anonymous referees.</p>
  </notes><ref-list>
    <title>References</title>

      <ref id="bib1.bib1"><label>1</label><?label 1?><mixed-citation>Benali, A., Carvalho, A. C., Nunes, J. P., Carvalhais, N., and Santos, A.:
Estimating air surface temperature in Portugal using MODIS LST data, Remote
Sens. Environ., 124, 108–121, <ext-link xlink:href="https://doi.org/10.1016/j.rse.2012.04.024" ext-link-type="DOI">10.1016/j.rse.2012.04.024</ext-link>,
2012.</mixed-citation></ref>
      <ref id="bib1.bib2"><label>2</label><?label 1?><mixed-citation>Benavides, R., Montes, F., Rubio, A., and Osoro, K.: Geostatistical
modelling of air temperature in a mountainous region of Northern Spain, Agr.
Forest Meteorol., 146, 173–188,
<ext-link xlink:href="https://doi.org/10.1016/j.agrformet.2007.05.014" ext-link-type="DOI">10.1016/j.agrformet.2007.05.014</ext-link>, 2007.</mixed-citation></ref>
      <ref id="bib1.bib3"><label>3</label><?label 1?><mixed-citation>Bisht, G. and Bras, R. L.: Estimation of net radiation from the MODIS data
under all sky conditions: Southern Great Plains case study, Remote Sens.
Environ., 114, 1522–1534, <ext-link xlink:href="https://doi.org/10.1016/j.rse.2010.02.007" ext-link-type="DOI">10.1016/j.rse.2010.02.007</ext-link>, 2010.</mixed-citation></ref>
      <ref id="bib1.bib4"><label>4</label><?label 1?><mixed-citation>Borbas, E. and Menzel, P.: MODIS Atmosphere L2 Atmosphere Profile Product,
NASA MODIS Adaptive Processing System, Goddard Space Flight Center [data set], USA,
<ext-link xlink:href="https://doi.org/10.5067/MODIS/MOD07_L2.006" ext-link-type="DOI">10.5067/MODIS/MOD07_L2.006</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bib5"><label>5</label><?label 1?><mixed-citation>Breiman, L.: Bagging predictors, Mach. Learn., 24, 123–140, 1996.</mixed-citation></ref>
      <ref id="bib1.bib6"><label>6</label><?label 1?><mixed-citation>Breiman, L.: Random forests, Mach. Learn., 45, 5–32, 2001.</mixed-citation></ref>
      <ref id="bib1.bib7"><label>7</label><?label 1?><mixed-citation>Breiman, L., Friedman, J. H., Olshen, R. A., and Stone, C. J.:
Classification and Regression Trees, Wadsworth International Group, Belmont, California, USA, 1984.</mixed-citation></ref>
      <ref id="bib1.bib8"><label>8</label><?label 1?><mixed-citation>Chen, F., Liu, Y., Liu, Q., and Qin, F.: A statistical method based on
remote sensing for the estimation of air temperature in China, Int. J.
Climatol., 35, 2131-2143, <ext-link xlink:href="https://doi.org/10.1002/joc.4113" ext-link-type="DOI">10.1002/joc.4113</ext-link>, 2015.</mixed-citation></ref>
      <ref id="bib1.bib9"><label>9</label><?label 1?><mixed-citation>Chen, Y., Liang, S., Ma, H., Li, B., He, T., and Wang, Q.: An All-sky
0.01<inline-formula><mml:math id="M275" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> Daily Surface Air Temperature Product over Beijing
(2003–2019), Zenodo [data set], <ext-link xlink:href="https://doi.org/10.5281/zenodo.4405123" ext-link-type="DOI">10.5281/zenodo.4405123</ext-link>, 2021a.</mixed-citation></ref>
      <ref id="bib1.bib10"><label>10</label><?label 1?><mixed-citation>Chen, Y., Liang, S., Ma, H., Li, B., He, T., and Wang, Q.: An All-sky 1 km
Daily Surface Air Temperature Product over Mainland China, Zenodo [data set],
<ext-link xlink:href="https://doi.org/10.5281/zenodo.4399453" ext-link-type="DOI">10.5281/zenodo.4399453</ext-link>, 2021b.</mixed-citation></ref>
      <ref id="bib1.bib11"><label>11</label><?label 1?><mixed-citation>Emamifar, S., Rahimikhoob, A., and Noroozi, A. A.: Daily mean air
temperature estimation from MODIS land surface temperature products based on
M5 model tree, Int. J. Climatol., 33, 3174–3181,
<ext-link xlink:href="https://doi.org/10.1002/joc.3655" ext-link-type="DOI">10.1002/joc.3655</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bib12"><label>12</label><?label 1?><mixed-citation>Famiglietti, C. A., Fisher, J. B., Halverson, G., and Borbas, E. E.: Global
validation of MODIS near-surface air and dew point temperatures, Geophys.
Res. Lett., 45, 7772–7780, <ext-link xlink:href="https://doi.org/10.1029/2018GL077813" ext-link-type="DOI">10.1029/2018GL077813</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib13"><label>13</label><?label 1?><mixed-citation>Gelaro, R., McCarty, W., Suarez, M. J., Todling, R., Molod, A., Takacs, L.,
Randles, C. A., Darmenov, A., Bosilovich, M. G., Reichle, R., Wargan, K.,
Coy, L., Cullather, R., Draper, C., Akella, S., Buchard, V., Conaty, A., da
Silva, A. M., Gu, W., Kim, G. K., Koster, R., Lucchesi, R., Merkova, D.,
Nielsen, J. E., Partyka, G., Pawson, S., Putman, W., Rienecker, M.,
Schubert, S. D., Sienkiewicz, M., and Zhao, B.: The Modern-Era Retrospective
Analysis for Research and Applications, Version 2 (MERRA-2), J. Climate, 30,
5419–5454, <ext-link xlink:href="https://doi.org/10.1175/jcli-d-16-0758.1" ext-link-type="DOI">10.1175/jcli-d-16-0758.1</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bib14"><label>14</label><?label 1?><mixed-citation>Gislason, P. O., Benediktsson, J. A., and Sveinsson, J. R.: Random Forests
for land cover classification, Pattern Recogn. Lett., 27, 294–300,
<ext-link xlink:href="https://doi.org/10.1016/j.patrec.2005.08.011" ext-link-type="DOI">10.1016/j.patrec.2005.08.011</ext-link>, 2006.</mixed-citation></ref>
      <ref id="bib1.bib15"><label>15</label><?label 1?><mixed-citation>Goetz, S. J., Prince, S. D., and Small, J.: Advances in satellite remote
sensing of environmental variables for epidemiological applications, Adv.
Parasit., 47, 289–307, <ext-link xlink:href="https://doi.org/10.1016/S0065-308X(00)47012-0" ext-link-type="DOI">10.1016/S0065-308X(00)47012-0</ext-link>, 2000.</mixed-citation></ref>
      <ref id="bib1.bib16"><label>16</label><?label 1?><mixed-citation>Gong, P., Wang, J., Yu, L., Zhao, Y., Zhao, Y., Liang, L., Niu, Z., Huang,
X., Fu, H., and Liu, S.: Finer resolution observation and monitoring of
global land cover: First mapping results with Landsat TM and ETM<inline-formula><mml:math id="M276" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> data,
Int. J. Remote Sens., 34, 2607–2654,
<ext-link xlink:href="https://doi.org/10.1080/01431161.2012.748992" ext-link-type="DOI">10.1080/01431161.2012.748992</ext-link>, 2013.</mixed-citation></ref>
      <?pagebreak page4260?><ref id="bib1.bib17"><label>17</label><?label 1?><mixed-citation>Good, E. J., Ghent, D. J., Bulgin, C. E., and Remedios, J. J.: A
spatiotemporal analysis of the relationship between near-surface air
temperature and satellite land surface temperatures using 17 years of data
from the ATSR series, J. Geophys. Res.-Atmos., 122, 9185–9210,
<ext-link xlink:href="https://doi.org/10.1002/2017jd026880" ext-link-type="DOI">10.1002/2017jd026880</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bib18"><label>18</label><?label 1?><mixed-citation>Guan, H., Zhang, X., Makhnin, O., and Sun, Z.: Mapping Mean Monthly
Temperatures over a Coastal Hilly Area Incorporating Terrain Aspect Effects,
J. Hydrometeorol., 14, 233–250, <ext-link xlink:href="https://doi.org/10.1175/jhm-d-12-014.1" ext-link-type="DOI">10.1175/jhm-d-12-014.1</ext-link>,
2013.</mixed-citation></ref>
      <ref id="bib1.bib19"><label>19</label><?label 1?><mixed-citation>Ham, J., Yangchi, C., Crawford, M. M., and Ghosh, J.: Investigation of the
random forest framework for classification of hyperspectral data, IEEE T.
Geosci. Remote, 43, 492–501, <ext-link xlink:href="https://doi.org/10.1109/tgrs.2004.842481" ext-link-type="DOI">10.1109/tgrs.2004.842481</ext-link>, 2005.</mixed-citation></ref>
      <ref id="bib1.bib20"><label>20</label><?label 1?><mixed-citation>Ishida, T. and Kawashima, S.: Use of cokriging to estimate surface
air-temperature from elevation, Theor. Appl. Climatol., 47, 147–157,
<ext-link xlink:href="https://doi.org/10.1007/bf00867447" ext-link-type="DOI">10.1007/bf00867447</ext-link>, 1993.</mixed-citation></ref>
      <ref id="bib1.bib21"><label>21</label><?label 1?><mixed-citation>Jang, J. D., Viau, A. A., and Anctil, F.: Neural network estimation of air
temperatures from AVHRR data, Int. J. Remote Sens., 25, 4541–4554,
<ext-link xlink:href="https://doi.org/10.1080/01431160310001657533" ext-link-type="DOI">10.1080/01431160310001657533</ext-link>, 2010.</mixed-citation></ref>
      <ref id="bib1.bib22"><label>22</label><?label 1?><mixed-citation>Jang, K., Kang, S., Kimball, J., and Hong, S.: Retrievals of All-Weather
Daily Air Temperature Using MODIS and AMSR-E Data, Remote Sens., 6,
8387–8404, <ext-link xlink:href="https://doi.org/10.3390/rs6098387" ext-link-type="DOI">10.3390/rs6098387</ext-link>, 2014.</mixed-citation></ref>
      <ref id="bib1.bib23"><label>23</label><?label 1?><mixed-citation>Khesali, E. and Mobasheri, M.: A method in near-surface estimation of air
temperature (NEAT) in times following the satellite passing time using MODIS
images, Adv. Space Res., 65, 2339–2347,
<ext-link xlink:href="https://doi.org/10.1016/j.asr.2020.02.006" ext-link-type="DOI">10.1016/j.asr.2020.02.006</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bib24"><label>24</label><?label 1?><mixed-citation>Kilibarda, M., Hengl, T., Heuvelink, G. B. M., Gräler, B., Pebesma, E.,
Perčec Tadić, M., and Bajat, B.: Spatio-temporal interpolation of
daily temperatures for global land areas at 1 km resolution, J. Geophys.
Res.-Atmos., 119, 2294–2313, <ext-link xlink:href="https://doi.org/10.1002/2013jd020803" ext-link-type="DOI">10.1002/2013jd020803</ext-link>, 2014.</mixed-citation></ref>
      <ref id="bib1.bib25"><label>25</label><?label 1?><mixed-citation>Kurtzman, D. and Kadmon, R.: Mapping of temperature variables in Israel: a
comparison of different interpolation methods, Clim. Res., 13, 33–43,
<ext-link xlink:href="https://doi.org/10.3354/cr013033" ext-link-type="DOI">10.3354/cr013033</ext-link>, 1999.</mixed-citation></ref>
      <ref id="bib1.bib26"><label>26</label><?label 1?><mixed-citation>Li, L. and Zha, Y.: Estimating monthly average temperature by remote
sensing in China, Adv. Space Res., 63, 2345–2357,
<ext-link xlink:href="https://doi.org/10.1016/j.asr.2018.12.039" ext-link-type="DOI">10.1016/j.asr.2018.12.039</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bib27"><label>27</label><?label 1?><mixed-citation>Li, X., Zhou, Y., Asrar, G. R., and Zhu, Z.: Developing a 1 km resolution
daily air temperature dataset for urban and surrounding areas in the
conterminous United States, Remote Sens. Environ., 215, 74–84,
<ext-link xlink:href="https://doi.org/10.1016/j.rse.2018.05.034" ext-link-type="DOI">10.1016/j.rse.2018.05.034</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib28"><label>28</label><?label 1?><mixed-citation>Liang, S.: Quantitative remote sensing of land surfaces, John Wiley &amp;
Sons, Inc., Hoboken, NJ, USA, 2004.</mixed-citation></ref>
      <ref id="bib1.bib29"><label>29</label><?label 1?><mixed-citation>Liang, S., Zhao, X., Liu, S., Yuan, W., Cheng, X., Xiao, Z., Zhang, X., Liu,
Q., Cheng, J., Tang, H., Qu, Y., Bo, Y., Qu, Y., Ren, H., Yu, K., and
Townshend, J.: A long-term Global LAnd Surface Satellite (GLASS) data-set
for environmental studies, Int. J. Digit. Earth, 6, 5–33,
<ext-link xlink:href="https://doi.org/10.1080/17538947.2013.805262" ext-link-type="DOI">10.1080/17538947.2013.805262</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bib30"><label>30</label><?label 1?><mixed-citation>Liang, S., Wang, D., He, T., and Yu, Y.: Remote sensing of earth's energy
budget: synthesis and review, Int. J. Digit. Earth, 12, 737–780,
<ext-link xlink:href="https://doi.org/10.1080/17538947.2019.1597189" ext-link-type="DOI">10.1080/17538947.2019.1597189</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bib31"><label>31</label><?label 1?><mixed-citation>Liang, S. and Wang, J.: Advanced remote sensing: terrestrial information
extraction and applications, 2nd Edn., Academic Press, 2019.</mixed-citation></ref>
      <ref id="bib1.bib32"><label>32</label><?label 1?><mixed-citation>Liang, S., Cheng, J., Jia, K., Jiang, B., Liu, Q., Xiao, Z., Yao, Y., Yuan,
W., Zhang, X., and Zhao, X.: The Global LAnd Surface Satellite (GLASS)
product suite, B. Am. Meteorol. Soc., 102, E323–E337,
<ext-link xlink:href="https://doi.org/10.1175/BAMS-D-18-0341.1" ext-link-type="DOI">10.1175/BAMS-D-18-0341.1</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bib33"><label>33</label><?label 1?><mixed-citation>Lin, S., Moore, N. J., Messina, J. P., DeVisser, M. H., and Wu, J.:
Evaluation of estimating daily maximum and minimum air temperature with
MODIS data in east Africa, Int. J. Appl. Earth Obs., 18, 128–140,
<ext-link xlink:href="https://doi.org/10.1016/j.jag.2012.01.004" ext-link-type="DOI">10.1016/j.jag.2012.01.004</ext-link>, 2012.</mixed-citation></ref>
      <ref id="bib1.bib34"><label>34</label><?label 1?><mixed-citation>Liu, Q., Wang, L., Qu, Y., Liu, N., Liu, S., Tang, H., and Liang, S.:
Preliminary evaluation of the long-term GLASS albedo product, Int. J. Digit.
Earth, 6, 69–95, <ext-link xlink:href="https://doi.org/10.1175/BAMS-D-18-0341.1" ext-link-type="DOI">10.1175/BAMS-D-18-0341.1</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bib35"><label>35</label><?label 1?><mixed-citation>Liu, R., Ma, Z., Liu, Y., Shao, Y., Zhao, W., and Bi, J.: Spatiotemporal
distributions of surface ozone levels in China from 2005 to 2017: A machine
learning approach, Environ. Int., 142, 105823,
<ext-link xlink:href="https://doi.org/10.1016/j.envint.2020.105823" ext-link-type="DOI">10.1016/j.envint.2020.105823</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bib36"><label>36</label><?label 1?><mixed-citation>Ma, J., Zhou, J., Göttsche, F.-M., Liang, S., Wang, S., and Li, M.: A global long-term (1981–2000) land surface temperature product for NOAA AVHRR, Earth Syst. Sci. Data, 12, 3247–3268, <ext-link xlink:href="https://doi.org/10.5194/essd-12-3247-2020" ext-link-type="DOI">10.5194/essd-12-3247-2020</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bib37"><label>37</label><?label 1?><mixed-citation>Marzban, F., Sodoudi, S., and Preusker, R.: The influence of land-cover type
on the relationship between NDVI–LST and LST-Tair, Int. J. Remote Sens.,
39, 1377–1398, <ext-link xlink:href="https://doi.org/10.1080/01431161.2017.1402386" ext-link-type="DOI">10.1080/01431161.2017.1402386</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bib38"><label>38</label><?label 1?><mixed-citation>
McGovern, A., Lagerquist, R., Gagne, D. J., Jergensen, G. E., Elmore, K. L., Homeyer, C. R., and Smith, T.: Making the black box more transparent: Understanding the physical implications of machine learning, B. Am. Meteorol. Soc., 100, 2175–2199, 2019.</mixed-citation></ref>
      <ref id="bib1.bib39"><label>39</label><?label 1?><mixed-citation>Meyer, H., Katurji, M., Appelhans, T., Müller, M., Nauss, T., Roudier,
P., and Zawar-Reza, P.: Mapping Daily Air Temperature for Antarctica Based
on MODIS LST, Remote Sens., 8, 732, <ext-link xlink:href="https://doi.org/10.3390/rs8090732" ext-link-type="DOI">10.3390/rs8090732</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bib40"><label>40</label><?label 1?><mixed-citation>Noi, P., Degener, J., and Kappas, M.: Comparison of Multiple Linear
Regression, Cubist Regression, and Random Forest Algorithms to Estimate
Daily Air Surface Temperature from Dynamic Combinations of MODIS LST Data,
Remote Sens., 9, 398, <ext-link xlink:href="https://doi.org/10.3390/rs9050398" ext-link-type="DOI">10.3390/rs9050398</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bib41"><label>41</label><?label 1?><mixed-citation>Ploton, P., Mortier, F., Rejou-Mechain, M., Barbier, N., Picard, N., Rossi,
V., Dormann, C., Cornu, G., Viennois, G., Bayol, N., Lyapustin, A.,
Gourlet-Fleury, S., and Pelissier, R.: Spatial validation reveals poor
predictive performance of large-scale ecological mapping models, Nat.
Commun., 11, 4540, <ext-link xlink:href="https://doi.org/10.1038/s41467-020-18321-y" ext-link-type="DOI">10.1038/s41467-020-18321-y</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bib42"><label>42</label><?label 1?><mixed-citation>Prihodko, L. and Goward, S. N.: Estimation of air temperature from remotely
sensed surface observations, Remote Sens. Environ., 60, 335–346,
<ext-link xlink:href="https://doi.org/10.1016/S0034-4257(96)00216-7" ext-link-type="DOI">10.1016/S0034-4257(96)00216-7</ext-link>, 1997.</mixed-citation></ref>
      <ref id="bib1.bib43"><label>43</label><?label 1?><mixed-citation>Quinlan, J. R.:
Induction of decision trees, Mach. Learn., 1, 81–106, 1986.</mixed-citation></ref>
      <ref id="bib1.bib44"><label>44</label><?label 1?><mixed-citation>Quinlan, J. R.: C4.5 : programs for machine learning, Morgan Kaufmann
Publishers Inc., 1992.</mixed-citation></ref>
      <ref id="bib1.bib45"><label>45</label><?label 1?><mixed-citation>Rao, Y., Liang, S., and Yu, Y.: Land Surface Air Temperature Data Are
Considerably Different Among BEST-LAND, CRU-TEM4v, NASA-GISS, and NOAA-NCEI,
J. Geophys. Res.-Atmos., 123, 5881–5900,
<ext-link xlink:href="https://doi.org/10.1029/2018jd028355" ext-link-type="DOI">10.1029/2018jd028355</ext-link>, 2018.</mixed-citation></ref>
      <?pagebreak page4261?><ref id="bib1.bib46"><label>46</label><?label 1?><mixed-citation>Rao, Y., Liang, S., Wang, D., Yu, Y., Song, Z., Zhou, Y., Shen, M., and Xu,
B.: Estimating daily average surface air temperature using satellite land
surface temperature and top-of-atmosphere radiation products over the
Tibetan Plateau, Remote Sens. Environ., 234, 111462,
<ext-link xlink:href="https://doi.org/10.1016/j.rse.2019.111462" ext-link-type="DOI">10.1016/j.rse.2019.111462</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bib47"><label>47</label><?label 1?><mixed-citation>Rodell, M., Houser, P. R., Jambor, U., Gottschalck, J., Mitchell, K., Meng,
C. J., Arsenault, K., Cosgrove, B., Radakovich, J., Bosilovich, M., Entin,
J. K., Walker, J. P., Lohmann, D., and Toll, D.: The Global Land Data
Assimilation System, B. Am. Meteorol. Soc., 85, 381–394,
<ext-link xlink:href="https://doi.org/10.1175/bams-85-3-381" ext-link-type="DOI">10.1175/bams-85-3-381</ext-link>, 2004.</mixed-citation></ref>
      <ref id="bib1.bib48"><label>48</label><?label 1?><mixed-citation>Rosenfeld, A., Dorman, M., Schwartz, J., Novack, V., Just, A. C., and Kloog,
I.: Estimating daily minimum, maximum, and mean near surface air temperature
using hybrid satellite models across Israel, Environ. Res., 159, 297–312,
<ext-link xlink:href="https://doi.org/10.1016/j.envres.2017.08.017" ext-link-type="DOI">10.1016/j.envres.2017.08.017</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bib49"><label>49</label><?label 1?><mixed-citation>Schwingshackl, C., Hirschi, M., and Seneviratne, S. I.: Global Contributions
of Incoming Radiation and Land Surface Conditions to Maximum Near-Surface
Air Temperature Variability and Trend, Geophys. Res. Lett., 45, 5034–5044,
<ext-link xlink:href="https://doi.org/10.1029/2018GL077794" ext-link-type="DOI">10.1029/2018GL077794</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib50"><label>50</label><?label 1?><mixed-citation>Shen, H., Jiang, Y., Li, T., Cheng, Q., Zeng, C., and Zhang, L.: Deep
learning-based air temperature mapping by fusing remote sensing, station,
simulation and socioeconomic data, Remote Sens. Environ., 240, 111692,
<ext-link xlink:href="https://doi.org/10.1016/j.rse.2020.111692" ext-link-type="DOI">10.1016/j.rse.2020.111692</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bib51"><label>51</label><?label 1?><mixed-citation>Shi, C., Xie, Z., Qian, H., Liang, M., and Yang, X.: China land soil
moisture EnKF data assimilation based on satellite remote sensing data, Sci.
China Earth Sci., 54, 1430–1440, <ext-link xlink:href="https://doi.org/10.1007/s11430-010-4160-3" ext-link-type="DOI">10.1007/s11430-010-4160-3</ext-link>,
2011.</mixed-citation></ref>
      <ref id="bib1.bib52"><label>52</label><?label 1?><mixed-citation>Stisen, S., Sandholt, I., Nørgaard, A., Fensholt, R., and Eklundh, L.:
Estimation of diurnal air temperature using MSG SEVIRI data in West Africa,
Remote Sens. Environ., 110, 262–274,
<ext-link xlink:href="https://doi.org/10.1016/j.rse.2007.02.025" ext-link-type="DOI">10.1016/j.rse.2007.02.025</ext-link>, 2007.</mixed-citation></ref>
      <ref id="bib1.bib53"><label>53</label><?label 1?><mixed-citation>Sun, Y. J., Wang, J. F., Zhang, R. H., Gillies, R. R., Xue, Y., and Bo, Y.
C.: Air temperature retrieval from remote sensing data based on
thermodynamics, Theor. Appl. Climatol., 80, 37–48,
<ext-link xlink:href="https://doi.org/10.1007/s00704-004-0079-y" ext-link-type="DOI">10.1007/s00704-004-0079-y</ext-link>, 2004.</mixed-citation></ref>
      <ref id="bib1.bib54"><label>54</label><?label 1?><mixed-citation>Vancutsem, C., Ceccato, P., Dinku, T., and Connor, S. J.: Evaluation of
MODIS land surface temperature data to estimate air temperature in different
ecosystems over Africa, Remote Sens. Environ., 114, 449–465,
<ext-link xlink:href="https://doi.org/10.1016/j.rse.2009.10.002" ext-link-type="DOI">10.1016/j.rse.2009.10.002</ext-link>, 2010.</mixed-citation></ref>
      <ref id="bib1.bib55"><label>55</label><?label 1?><mixed-citation>Vogt, J. V., Viau, A. A., and Paquet, F.: Mapping regional air temperature
fields using satellite-derived surface skin temperatures, Int. J. Climatol.,
17, 1559–1579, 1997.</mixed-citation></ref>
      <ref id="bib1.bib56"><label>56</label><?label 1?><mixed-citation>Wan, Z., Hook, S., and Hulley, G.: MOD11A1 MODIS/Terra Land Surface
Temperature/Emissivity Daily L3 Global 1km SIN Grid, NASA LP DAAC [data set],
<ext-link xlink:href="https://doi.org/10.5067/MODIS/MOD11A1.006" ext-link-type="DOI">10.5067/MODIS/MOD11A1.006</ext-link>, 2015.</mixed-citation></ref>
      <ref id="bib1.bib57"><label>57</label><?label 1?><mixed-citation>Xiao, Q., Chang, H. H., Geng, G., and Liu, Y.: An Ensemble Machine-Learning
Model To Predict Historical PM<inline-formula><mml:math id="M277" display="inline"><mml:msub><mml:mi/><mml:mn mathvariant="normal">2.5</mml:mn></mml:msub></mml:math></inline-formula> Concentrations in China from Satellite
Data, Environ. Sci. Technol., 52, 13260–13269,
<ext-link xlink:href="https://doi.org/10.1021/acs.est.8b02917" ext-link-type="DOI">10.1021/acs.est.8b02917</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib58"><label>58</label><?label 1?><mixed-citation>Xiao, Z., Liang, S., Wang, J., Chen, P., Yin, X., Zhang, L., and Song, J.:
Use of General Regression Neural Networks for Generating the GLASS Leaf Area
Index Product From Time-Series MODIS Surface Reflectance, IEEE T. Geosci.
Remote, 52, 209–223, <ext-link xlink:href="https://doi.org/10.1109/tgrs.2013.2237780" ext-link-type="DOI">10.1109/tgrs.2013.2237780</ext-link>, 2014.
</mixed-citation></ref><?xmltex \hack{\newpage}?>
      <ref id="bib1.bib59"><label>59</label><?label 1?><mixed-citation>Xu, Y., Knudby, A., and Ho, H. C.: Estimating daily maximum air temperature
from MODIS in British Columbia, Canada, Int. J. Remote Sens., 35, 8108–8121,
<ext-link xlink:href="https://doi.org/10.1080/01431161.2014.978957" ext-link-type="DOI">10.1080/01431161.2014.978957</ext-link>, 2014.</mixed-citation></ref>
      <ref id="bib1.bib60"><label>60</label><?label 1?><mixed-citation>Yang, K. and He, J.: China meteorological forcing dataset (1979–2018),
National Tibetan Plateau Data Center [data set],
<ext-link xlink:href="https://doi.org/10.11888/AtmosphericPhysics.tpe.249369.file" ext-link-type="DOI">10.11888/AtmosphericPhysics.tpe.249369.file</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bib61"><label>61</label><?label 1?><mixed-citation>Yao, R., Wang, L., Huang, X., Li, L., Sun, J., Wu, X., and Jiang, W.:
Developing a temporally accurate air temperature dataset for Mainland China,
Sci. Total Environ., 706, 136037,
<ext-link xlink:href="https://doi.org/10.1016/j.scitotenv.2019.136037" ext-link-type="DOI">10.1016/j.scitotenv.2019.136037</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bib62"><label>62</label><?label 1?><mixed-citation>Zeng, L., Wardlow, B., Tadesse, T., Shan, J., Hayes, M., Li, D., and Xiang,
D.: Estimation of Daily Air Temperature Based on MODIS Land Surface
Temperature Products over the Corn Belt in the US, Remote Sens., 7, 951–970,
<ext-link xlink:href="https://doi.org/10.3390/rs70100951" ext-link-type="DOI">10.3390/rs70100951</ext-link>, 2015.</mixed-citation></ref>
      <ref id="bib1.bib63"><label>63</label><?label 1?><mixed-citation>Zhang, H., Zhang, F., Ye, M., Che, T., and Zhang, G.: Estimating daily air
temperatures over the Tibetan Plateau by dynamically integrating MODIS LST
data, J. Geophys. Res.-Atmos., 121, 11425–11441,
<ext-link xlink:href="https://doi.org/10.1002/2016jd025154" ext-link-type="DOI">10.1002/2016jd025154</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bib64"><label>64</label><?label 1?><mixed-citation>Zhang, H.: Estimation of daily average near-surface air temperature using
MODIS and AIRS data, 2017 2nd International Conference on Frontiers of
Sensors Technologies (ICFST),  377-381, 2017.</mixed-citation></ref>
      <ref id="bib1.bib65"><label>65</label><?label 1?><mixed-citation>Zhang, H., Zhang, F. A. N., Zhang, G., Ma, Y., Yang, K. U. N., and Ye, M.:
Daily air temperature estimation on glacier surfaces in the Tibetan Plateau
using MODIS LST data, J. Glaciol., 64, 132–147,
<ext-link xlink:href="https://doi.org/10.1017/jog.2018.6" ext-link-type="DOI">10.1017/jog.2018.6</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib66"><label>66</label><?label 1?><mixed-citation>Zhang, W., Huang, Y., Yu, Y., and Sun, W.: Empirical models for estimating
daily maximum, minimum and mean air temperatures with MODIS land surface
temperatures, Int. J. Remote Sens., 32, 9415–9440,
<ext-link xlink:href="https://doi.org/10.1080/01431161.2011.560622" ext-link-type="DOI">10.1080/01431161.2011.560622</ext-link>, 2011.</mixed-citation></ref>
      <ref id="bib1.bib67"><label>67</label><?label 1?><mixed-citation>Zhang, X., Wang, D., Liu, Q., Yao, Y., Jia, K., He, T., Jiang, B., Wei, Y.,
Ma, H., and Zhao, X.: An operational approach for generating the global land
surface downward shortwave radiation product from MODIS data, IEEE T.
Geosci. Remote, 57, 4636–4650, <ext-link xlink:href="https://doi.org/10.1109/TGRS.2019.2891945" ext-link-type="DOI">10.1109/TGRS.2019.2891945</ext-link>,
2019.</mixed-citation></ref>
      <ref id="bib1.bib68"><label>68</label><?label 1?><mixed-citation>Zhu, W., Lű, A., and Jia, S.: Estimation of daily maximum and minimum
air temperature using MODIS land surface temperature products, Remote Sens.
Environ., 130, 62–73, <ext-link xlink:href="https://doi.org/10.1016/j.rse.2012.10.034" ext-link-type="DOI">10.1016/j.rse.2012.10.034</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bib69"><label>69</label><?label 1?><mixed-citation>Zhu, W., Lű, A., Jia, S., Yan, J., and Mahmood, R.: Retrievals of
all-weather daytime air temperature from MODIS products, Remote Sens.
Environ., 189, 152–163, <ext-link xlink:href="https://doi.org/10.1016/j.rse.2016.11.011" ext-link-type="DOI">10.1016/j.rse.2016.11.011</ext-link>, 2017.</mixed-citation></ref>

  </ref-list></back>
    <!--<article-title-html>An all-sky 1&thinsp;km daily land surface air temperature product over mainland China for 2003–2019 from  MODIS and ancillary data</article-title-html>
<abstract-html><p>Surface air temperature (<i>T</i><sub>a</sub>), as an important
climate variable, has been used in a wide range of fields such as ecology,
hydrology, climatology, epidemiology, and environmental science. However,
ground measurements are limited by poor spatial representation and
inconsistency, and reanalysis and meteorological forcing datasets suffer
from coarse spatial resolution and inaccuracy. Previous studies using
satellite data have mainly estimated <i>T</i><sub>a</sub> under clear-sky conditions or
with limited temporal and spatial coverage. In this study, an all-sky daily
mean land <i>T</i><sub>a</sub> product at a 1&thinsp;km spatial resolution over mainland China for
2003–2019 has been generated mainly from the Moderate Resolution Imaging
Spectroradiometer (MODIS) products and the Global Land Data Assimilation
System (GLDAS) dataset. Three <i>T</i><sub>a</sub> estimation models based on random
forest were trained using ground measurements from 2384 stations for three
different clear-sky and cloudy-sky conditions. The random sample validation
results showed that the <i>R</i><sup>2</sup> and root-mean-square error (RMSE) values of the
three models ranged from 0.984 to 0.986 and from 1.342 to 1.440&thinsp;K,
respectively. We examined the spatiotemporal patterns and land cover type
dependences of model accuracy. Two cross-validation (CV) strategies of
leave-time-out (LTO) CV and leave-location-out (LLO) CV were also used to
evaluate the models. Finally, we developed the all-sky <i>T</i><sub>a</sub> dataset from
2003 to 2009 and compared it with the China Land Data Assimilation System
(CLDAS) dataset at a 0.0625° spatial resolution, the China
Meteorological Forcing Data (CMFD) dataset at a 0.1° spatial
resolution, and the GLDAS dataset at a 0.25° spatial resolution.
Validation accuracy of our product in 2010 was significantly better than
other datasets, with <i>R</i><sup>2</sup> and RMSE values of 0.992 and 1.010&thinsp;K,
respectively. In summary, the developed all-sky daily mean land <i>T</i><sub>a</sub>
dataset has achieved satisfactory accuracy and high spatial resolution
simultaneously, which fills the current dataset gap in this field and plays
an important role in the studies of climate change and the hydrological cycle.
This dataset is currently freely available at <a href="https://doi.org/10.5281/zenodo.4399453" target="_blank">https://doi.org/10.5281/zenodo.4399453</a>
(Chen et al., 2021b) and the University of Maryland
(<a href="http://glass.umd.edu/Ta_China/" target="_blank"/>, last access: 24 August 2021). A sub-dataset
that covers Beijing generated from this dataset is also publicly available
at <a href="https://doi.org/10.5281/zenodo.4405123" target="_blank">https://doi.org/10.5281/zenodo.4405123</a> (Chen et al., 2021a).</p></abstract-html>
<ref-html id="bib1.bib1"><label>1</label><mixed-citation>
Benali, A., Carvalho, A. C., Nunes, J. P., Carvalhais, N., and Santos, A.:
Estimating air surface temperature in Portugal using MODIS LST data, Remote
Sens. Environ., 124, 108–121, <a href="https://doi.org/10.1016/j.rse.2012.04.024" target="_blank">https://doi.org/10.1016/j.rse.2012.04.024</a>,
2012.
</mixed-citation></ref-html>
<ref-html id="bib1.bib2"><label>2</label><mixed-citation>Benavides, R., Montes, F., Rubio, A., and Osoro, K.: Geostatistical
modelling of air temperature in a mountainous region of Northern Spain, Agr.
Forest Meteorol., 146, 173–188,
<a href="https://doi.org/10.1016/j.agrformet.2007.05.014" target="_blank">https://doi.org/10.1016/j.agrformet.2007.05.014</a>, 2007.
</mixed-citation></ref-html>
<ref-html id="bib1.bib3"><label>3</label><mixed-citation>Bisht, G. and Bras, R. L.: Estimation of net radiation from the MODIS data
under all sky conditions: Southern Great Plains case study, Remote Sens.
Environ., 114, 1522–1534, <a href="https://doi.org/10.1016/j.rse.2010.02.007" target="_blank">https://doi.org/10.1016/j.rse.2010.02.007</a>, 2010.
</mixed-citation></ref-html>
<ref-html id="bib1.bib4"><label>4</label><mixed-citation>Borbas, E. and Menzel, P.: MODIS Atmosphere L2 Atmosphere Profile Product,
NASA MODIS Adaptive Processing System, Goddard Space Flight Center [data set], USA,
<a href="https://doi.org/10.5067/MODIS/MOD07_L2.006" target="_blank">https://doi.org/10.5067/MODIS/MOD07_L2.006</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib5"><label>5</label><mixed-citation>Breiman, L.: Bagging predictors, Mach. Learn., 24, 123–140, 1996.
</mixed-citation></ref-html>
<ref-html id="bib1.bib6"><label>6</label><mixed-citation>Breiman, L.: Random forests, Mach. Learn., 45, 5–32, 2001.
</mixed-citation></ref-html>
<ref-html id="bib1.bib7"><label>7</label><mixed-citation>Breiman, L., Friedman, J. H., Olshen, R. A., and Stone, C. J.:
Classification and Regression Trees, Wadsworth International Group, Belmont, California, USA, 1984.
</mixed-citation></ref-html>
<ref-html id="bib1.bib8"><label>8</label><mixed-citation>Chen, F., Liu, Y., Liu, Q., and Qin, F.: A statistical method based on
remote sensing for the estimation of air temperature in China, Int. J.
Climatol., 35, 2131-2143, <a href="https://doi.org/10.1002/joc.4113" target="_blank">https://doi.org/10.1002/joc.4113</a>, 2015.
</mixed-citation></ref-html>
<ref-html id="bib1.bib9"><label>9</label><mixed-citation>Chen, Y., Liang, S., Ma, H., Li, B., He, T., and Wang, Q.: An All-sky
0.01° Daily Surface Air Temperature Product over Beijing
(2003–2019), Zenodo [data set], <a href="https://doi.org/10.5281/zenodo.4405123" target="_blank">https://doi.org/10.5281/zenodo.4405123</a>, 2021a.
</mixed-citation></ref-html>
<ref-html id="bib1.bib10"><label>10</label><mixed-citation>Chen, Y., Liang, S., Ma, H., Li, B., He, T., and Wang, Q.: An All-sky 1&thinsp;km
Daily Surface Air Temperature Product over Mainland China, Zenodo [data set],
<a href="https://doi.org/10.5281/zenodo.4399453" target="_blank">https://doi.org/10.5281/zenodo.4399453</a>, 2021b.
</mixed-citation></ref-html>
<ref-html id="bib1.bib11"><label>11</label><mixed-citation>Emamifar, S., Rahimikhoob, A., and Noroozi, A. A.: Daily mean air
temperature estimation from MODIS land surface temperature products based on
M5 model tree, Int. J. Climatol., 33, 3174–3181,
<a href="https://doi.org/10.1002/joc.3655" target="_blank">https://doi.org/10.1002/joc.3655</a>, 2013.
</mixed-citation></ref-html>
<ref-html id="bib1.bib12"><label>12</label><mixed-citation>Famiglietti, C. A., Fisher, J. B., Halverson, G., and Borbas, E. E.: Global
validation of MODIS near-surface air and dew point temperatures, Geophys.
Res. Lett., 45, 7772–7780, <a href="https://doi.org/10.1029/2018GL077813" target="_blank">https://doi.org/10.1029/2018GL077813</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib13"><label>13</label><mixed-citation>Gelaro, R., McCarty, W., Suarez, M. J., Todling, R., Molod, A., Takacs, L.,
Randles, C. A., Darmenov, A., Bosilovich, M. G., Reichle, R., Wargan,&thinsp;K.,
Coy, L., Cullather, R., Draper, C., Akella, S., Buchard, V., Conaty, A., da
Silva, A. M., Gu, W., Kim, G.&thinsp;K., Koster, R., Lucchesi, R., Merkova, D.,
Nielsen, J. E., Partyka, G., Pawson, S., Putman, W., Rienecker, M.,
Schubert, S. D., Sienkiewicz, M., and Zhao, B.: The Modern-Era Retrospective
Analysis for Research and Applications, Version 2 (MERRA-2), J. Climate, 30,
5419–5454, <a href="https://doi.org/10.1175/jcli-d-16-0758.1" target="_blank">https://doi.org/10.1175/jcli-d-16-0758.1</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib14"><label>14</label><mixed-citation>Gislason, P. O., Benediktsson, J. A., and Sveinsson, J. R.: Random Forests
for land cover classification, Pattern Recogn. Lett., 27, 294–300,
<a href="https://doi.org/10.1016/j.patrec.2005.08.011" target="_blank">https://doi.org/10.1016/j.patrec.2005.08.011</a>, 2006.
</mixed-citation></ref-html>
<ref-html id="bib1.bib15"><label>15</label><mixed-citation>Goetz, S. J., Prince, S. D., and Small, J.: Advances in satellite remote
sensing of environmental variables for epidemiological applications, Adv.
Parasit., 47, 289–307, <a href="https://doi.org/10.1016/S0065-308X(00)47012-0" target="_blank">https://doi.org/10.1016/S0065-308X(00)47012-0</a>, 2000.
</mixed-citation></ref-html>
<ref-html id="bib1.bib16"><label>16</label><mixed-citation>Gong, P., Wang, J., Yu, L., Zhao, Y., Zhao, Y., Liang, L., Niu, Z., Huang,
X., Fu, H., and Liu, S.: Finer resolution observation and monitoring of
global land cover: First mapping results with Landsat TM and ETM+ data,
Int. J. Remote Sens., 34, 2607–2654,
<a href="https://doi.org/10.1080/01431161.2012.748992" target="_blank">https://doi.org/10.1080/01431161.2012.748992</a>, 2013.
</mixed-citation></ref-html>
<ref-html id="bib1.bib17"><label>17</label><mixed-citation>Good, E. J., Ghent, D. J., Bulgin, C. E., and Remedios, J. J.: A
spatiotemporal analysis of the relationship between near-surface air
temperature and satellite land surface temperatures using 17 years of data
from the ATSR series, J. Geophys. Res.-Atmos., 122, 9185–9210,
<a href="https://doi.org/10.1002/2017jd026880" target="_blank">https://doi.org/10.1002/2017jd026880</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib18"><label>18</label><mixed-citation>Guan, H., Zhang, X., Makhnin, O., and Sun, Z.: Mapping Mean Monthly
Temperatures over a Coastal Hilly Area Incorporating Terrain Aspect Effects,
J. Hydrometeorol., 14, 233–250, <a href="https://doi.org/10.1175/jhm-d-12-014.1" target="_blank">https://doi.org/10.1175/jhm-d-12-014.1</a>,
2013.
</mixed-citation></ref-html>
<ref-html id="bib1.bib19"><label>19</label><mixed-citation>Ham, J., Yangchi, C., Crawford, M. M., and Ghosh, J.: Investigation of the
random forest framework for classification of hyperspectral data, IEEE T.
Geosci. Remote, 43, 492–501, <a href="https://doi.org/10.1109/tgrs.2004.842481" target="_blank">https://doi.org/10.1109/tgrs.2004.842481</a>, 2005.
</mixed-citation></ref-html>
<ref-html id="bib1.bib20"><label>20</label><mixed-citation>Ishida, T. and Kawashima, S.: Use of cokriging to estimate surface
air-temperature from elevation, Theor. Appl. Climatol., 47, 147–157,
<a href="https://doi.org/10.1007/bf00867447" target="_blank">https://doi.org/10.1007/bf00867447</a>, 1993.
</mixed-citation></ref-html>
<ref-html id="bib1.bib21"><label>21</label><mixed-citation>Jang, J. D., Viau, A. A., and Anctil, F.: Neural network estimation of air
temperatures from AVHRR data, Int. J. Remote Sens., 25, 4541–4554,
<a href="https://doi.org/10.1080/01431160310001657533" target="_blank">https://doi.org/10.1080/01431160310001657533</a>, 2010.
</mixed-citation></ref-html>
<ref-html id="bib1.bib22"><label>22</label><mixed-citation>Jang,&thinsp;K., Kang, S., Kimball, J., and Hong, S.: Retrievals of All-Weather
Daily Air Temperature Using MODIS and AMSR-E Data, Remote Sens., 6,
8387–8404, <a href="https://doi.org/10.3390/rs6098387" target="_blank">https://doi.org/10.3390/rs6098387</a>, 2014.
</mixed-citation></ref-html>
<ref-html id="bib1.bib23"><label>23</label><mixed-citation>Khesali, E. and Mobasheri, M.: A method in near-surface estimation of air
temperature (NEAT) in times following the satellite passing time using MODIS
images, Adv. Space Res., 65, 2339–2347,
<a href="https://doi.org/10.1016/j.asr.2020.02.006" target="_blank">https://doi.org/10.1016/j.asr.2020.02.006</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib24"><label>24</label><mixed-citation>Kilibarda, M., Hengl, T., Heuvelink, G. B. M., Gräler, B., Pebesma, E.,
Perčec Tadić, M., and Bajat, B.: Spatio-temporal interpolation of
daily temperatures for global land areas at 1&thinsp;km resolution, J. Geophys.
Res.-Atmos., 119, 2294–2313, <a href="https://doi.org/10.1002/2013jd020803" target="_blank">https://doi.org/10.1002/2013jd020803</a>, 2014.
</mixed-citation></ref-html>
<ref-html id="bib1.bib25"><label>25</label><mixed-citation>Kurtzman, D. and Kadmon, R.: Mapping of temperature variables in Israel: a
comparison of different interpolation methods, Clim. Res., 13, 33–43,
<a href="https://doi.org/10.3354/cr013033" target="_blank">https://doi.org/10.3354/cr013033</a>, 1999.
</mixed-citation></ref-html>
<ref-html id="bib1.bib26"><label>26</label><mixed-citation>Li, L. and Zha, Y.: Estimating monthly average temperature by remote
sensing in China, Adv. Space Res., 63, 2345–2357,
<a href="https://doi.org/10.1016/j.asr.2018.12.039" target="_blank">https://doi.org/10.1016/j.asr.2018.12.039</a>, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib27"><label>27</label><mixed-citation>Li, X., Zhou, Y., Asrar, G. R., and Zhu, Z.: Developing a 1&thinsp;km resolution
daily air temperature dataset for urban and surrounding areas in the
conterminous United States, Remote Sens. Environ., 215, 74–84,
<a href="https://doi.org/10.1016/j.rse.2018.05.034" target="_blank">https://doi.org/10.1016/j.rse.2018.05.034</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib28"><label>28</label><mixed-citation>Liang, S.: Quantitative remote sensing of land surfaces, John Wiley &amp;
Sons, Inc., Hoboken, NJ, USA, 2004.
</mixed-citation></ref-html>
<ref-html id="bib1.bib29"><label>29</label><mixed-citation>Liang, S., Zhao, X., Liu, S., Yuan, W., Cheng, X., Xiao, Z., Zhang, X., Liu,
Q., Cheng, J., Tang, H., Qu, Y., Bo, Y., Qu, Y., Ren, H., Yu,&thinsp;K., and
Townshend, J.: A long-term Global LAnd Surface Satellite (GLASS) data-set
for environmental studies, Int. J. Digit. Earth, 6, 5–33,
<a href="https://doi.org/10.1080/17538947.2013.805262" target="_blank">https://doi.org/10.1080/17538947.2013.805262</a>, 2013.
</mixed-citation></ref-html>
<ref-html id="bib1.bib30"><label>30</label><mixed-citation>Liang, S., Wang, D., He, T., and Yu, Y.: Remote sensing of earth's energy
budget: synthesis and review, Int. J. Digit. Earth, 12, 737–780,
<a href="https://doi.org/10.1080/17538947.2019.1597189" target="_blank">https://doi.org/10.1080/17538947.2019.1597189</a>, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib31"><label>31</label><mixed-citation>Liang, S. and Wang, J.: Advanced remote sensing: terrestrial information
extraction and applications, 2nd Edn., Academic Press, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib32"><label>32</label><mixed-citation>Liang, S., Cheng, J., Jia,&thinsp;K., Jiang, B., Liu, Q., Xiao, Z., Yao, Y., Yuan,
W., Zhang, X., and Zhao, X.: The Global LAnd Surface Satellite (GLASS)
product suite, B. Am. Meteorol. Soc., 102, E323–E337,
<a href="https://doi.org/10.1175/BAMS-D-18-0341.1" target="_blank">https://doi.org/10.1175/BAMS-D-18-0341.1</a>, 2021.
</mixed-citation></ref-html>
<ref-html id="bib1.bib33"><label>33</label><mixed-citation>Lin, S., Moore, N. J., Messina, J. P., DeVisser, M. H., and Wu, J.:
Evaluation of estimating daily maximum and minimum air temperature with
MODIS data in east Africa, Int. J. Appl. Earth Obs., 18, 128–140,
<a href="https://doi.org/10.1016/j.jag.2012.01.004" target="_blank">https://doi.org/10.1016/j.jag.2012.01.004</a>, 2012.
</mixed-citation></ref-html>
<ref-html id="bib1.bib34"><label>34</label><mixed-citation>Liu, Q., Wang, L., Qu, Y., Liu, N., Liu, S., Tang, H., and Liang, S.:
Preliminary evaluation of the long-term GLASS albedo product, Int. J. Digit.
Earth, 6, 69–95, <a href="https://doi.org/10.1175/BAMS-D-18-0341.1" target="_blank">https://doi.org/10.1175/BAMS-D-18-0341.1</a>, 2013.
</mixed-citation></ref-html>
<ref-html id="bib1.bib35"><label>35</label><mixed-citation>Liu, R., Ma, Z., Liu, Y., Shao, Y., Zhao, W., and Bi, J.: Spatiotemporal
distributions of surface ozone levels in China from 2005 to 2017: A machine
learning approach, Environ. Int., 142, 105823,
<a href="https://doi.org/10.1016/j.envint.2020.105823" target="_blank">https://doi.org/10.1016/j.envint.2020.105823</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib36"><label>36</label><mixed-citation>
Ma, J., Zhou, J., Göttsche, F.-M., Liang, S., Wang, S., and Li, M.: A global long-term (1981–2000) land surface temperature product for NOAA AVHRR, Earth Syst. Sci. Data, 12, 3247–3268, <a href="https://doi.org/10.5194/essd-12-3247-2020" target="_blank">https://doi.org/10.5194/essd-12-3247-2020</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib37"><label>37</label><mixed-citation>Marzban, F., Sodoudi, S., and Preusker, R.: The influence of land-cover type
on the relationship between NDVI–LST and LST-Tair, Int. J. Remote Sens.,
39, 1377–1398, <a href="https://doi.org/10.1080/01431161.2017.1402386" target="_blank">https://doi.org/10.1080/01431161.2017.1402386</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib38"><label>38</label><mixed-citation>
McGovern, A., Lagerquist, R., Gagne, D. J., Jergensen, G. E., Elmore, K. L., Homeyer, C. R., and Smith, T.: Making the black box more transparent: Understanding the physical implications of machine learning, B. Am. Meteorol. Soc., 100, 2175–2199, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib39"><label>39</label><mixed-citation>Meyer, H., Katurji, M., Appelhans, T., Müller, M., Nauss, T., Roudier,
P., and Zawar-Reza, P.: Mapping Daily Air Temperature for Antarctica Based
on MODIS LST, Remote Sens., 8, 732, <a href="https://doi.org/10.3390/rs8090732" target="_blank">https://doi.org/10.3390/rs8090732</a>, 2016.
</mixed-citation></ref-html>
<ref-html id="bib1.bib40"><label>40</label><mixed-citation>Noi, P., Degener, J., and Kappas, M.: Comparison of Multiple Linear
Regression, Cubist Regression, and Random Forest Algorithms to Estimate
Daily Air Surface Temperature from Dynamic Combinations of MODIS LST Data,
Remote Sens., 9, 398, <a href="https://doi.org/10.3390/rs9050398" target="_blank">https://doi.org/10.3390/rs9050398</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib41"><label>41</label><mixed-citation>Ploton, P., Mortier, F., Rejou-Mechain, M., Barbier, N., Picard, N., Rossi,
V., Dormann, C., Cornu, G., Viennois, G., Bayol, N., Lyapustin, A.,
Gourlet-Fleury, S., and Pelissier, R.: Spatial validation reveals poor
predictive performance of large-scale ecological mapping models, Nat.
Commun., 11, 4540, <a href="https://doi.org/10.1038/s41467-020-18321-y" target="_blank">https://doi.org/10.1038/s41467-020-18321-y</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib42"><label>42</label><mixed-citation>Prihodko, L. and Goward, S. N.: Estimation of air temperature from remotely
sensed surface observations, Remote Sens. Environ., 60, 335–346,
<a href="https://doi.org/10.1016/S0034-4257(96)00216-7" target="_blank">https://doi.org/10.1016/S0034-4257(96)00216-7</a>, 1997.
</mixed-citation></ref-html>
<ref-html id="bib1.bib43"><label>43</label><mixed-citation>Quinlan, J. R.:
Induction of decision trees, Mach. Learn., 1, 81–106, 1986.
</mixed-citation></ref-html>
<ref-html id="bib1.bib44"><label>44</label><mixed-citation>Quinlan, J. R.: C4.5 : programs for machine learning, Morgan Kaufmann
Publishers Inc., 1992.
</mixed-citation></ref-html>
<ref-html id="bib1.bib45"><label>45</label><mixed-citation>Rao, Y., Liang, S., and Yu, Y.: Land Surface Air Temperature Data Are
Considerably Different Among BEST-LAND, CRU-TEM4v, NASA-GISS, and NOAA-NCEI,
J. Geophys. Res.-Atmos., 123, 5881–5900,
<a href="https://doi.org/10.1029/2018jd028355" target="_blank">https://doi.org/10.1029/2018jd028355</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib46"><label>46</label><mixed-citation>Rao, Y., Liang, S., Wang, D., Yu, Y., Song, Z., Zhou, Y., Shen, M., and Xu,
B.: Estimating daily average surface air temperature using satellite land
surface temperature and top-of-atmosphere radiation products over the
Tibetan Plateau, Remote Sens. Environ., 234, 111462,
<a href="https://doi.org/10.1016/j.rse.2019.111462" target="_blank">https://doi.org/10.1016/j.rse.2019.111462</a>, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib47"><label>47</label><mixed-citation>Rodell, M., Houser, P. R., Jambor, U., Gottschalck, J., Mitchell,&thinsp;K., Meng,
C. J., Arsenault,&thinsp;K., Cosgrove, B., Radakovich, J., Bosilovich, M., Entin,
J.&thinsp;K., Walker, J. P., Lohmann, D., and Toll, D.: The Global Land Data
Assimilation System, B. Am. Meteorol. Soc., 85, 381–394,
<a href="https://doi.org/10.1175/bams-85-3-381" target="_blank">https://doi.org/10.1175/bams-85-3-381</a>, 2004.
</mixed-citation></ref-html>
<ref-html id="bib1.bib48"><label>48</label><mixed-citation>Rosenfeld, A., Dorman, M., Schwartz, J., Novack, V., Just, A. C., and Kloog,
I.: Estimating daily minimum, maximum, and mean near surface air temperature
using hybrid satellite models across Israel, Environ. Res., 159, 297–312,
<a href="https://doi.org/10.1016/j.envres.2017.08.017" target="_blank">https://doi.org/10.1016/j.envres.2017.08.017</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib49"><label>49</label><mixed-citation>Schwingshackl, C., Hirschi, M., and Seneviratne, S. I.: Global Contributions
of Incoming Radiation and Land Surface Conditions to Maximum Near-Surface
Air Temperature Variability and Trend, Geophys. Res. Lett., 45, 5034–5044,
<a href="https://doi.org/10.1029/2018GL077794" target="_blank">https://doi.org/10.1029/2018GL077794</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib50"><label>50</label><mixed-citation>Shen, H., Jiang, Y., Li, T., Cheng, Q., Zeng, C., and Zhang, L.: Deep
learning-based air temperature mapping by fusing remote sensing, station,
simulation and socioeconomic data, Remote Sens. Environ., 240, 111692,
<a href="https://doi.org/10.1016/j.rse.2020.111692" target="_blank">https://doi.org/10.1016/j.rse.2020.111692</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib51"><label>51</label><mixed-citation>Shi, C., Xie, Z., Qian, H., Liang, M., and Yang, X.: China land soil
moisture EnKF data assimilation based on satellite remote sensing data, Sci.
China Earth Sci., 54, 1430–1440, <a href="https://doi.org/10.1007/s11430-010-4160-3" target="_blank">https://doi.org/10.1007/s11430-010-4160-3</a>,
2011.
</mixed-citation></ref-html>
<ref-html id="bib1.bib52"><label>52</label><mixed-citation>Stisen, S., Sandholt, I., Nørgaard, A., Fensholt, R., and Eklundh, L.:
Estimation of diurnal air temperature using MSG SEVIRI data in West Africa,
Remote Sens. Environ., 110, 262–274,
<a href="https://doi.org/10.1016/j.rse.2007.02.025" target="_blank">https://doi.org/10.1016/j.rse.2007.02.025</a>, 2007.
</mixed-citation></ref-html>
<ref-html id="bib1.bib53"><label>53</label><mixed-citation>Sun, Y. J., Wang, J. F., Zhang, R. H., Gillies, R. R., Xue, Y., and Bo, Y.
C.: Air temperature retrieval from remote sensing data based on
thermodynamics, Theor. Appl. Climatol., 80, 37–48,
<a href="https://doi.org/10.1007/s00704-004-0079-y" target="_blank">https://doi.org/10.1007/s00704-004-0079-y</a>, 2004.
</mixed-citation></ref-html>
<ref-html id="bib1.bib54"><label>54</label><mixed-citation>Vancutsem, C., Ceccato, P., Dinku, T., and Connor, S. J.: Evaluation of
MODIS land surface temperature data to estimate air temperature in different
ecosystems over Africa, Remote Sens. Environ., 114, 449–465,
<a href="https://doi.org/10.1016/j.rse.2009.10.002" target="_blank">https://doi.org/10.1016/j.rse.2009.10.002</a>, 2010.
</mixed-citation></ref-html>
<ref-html id="bib1.bib55"><label>55</label><mixed-citation>Vogt, J. V., Viau, A. A., and Paquet, F.: Mapping regional air temperature
fields using satellite-derived surface skin temperatures, Int. J. Climatol.,
17, 1559–1579, 1997.
</mixed-citation></ref-html>
<ref-html id="bib1.bib56"><label>56</label><mixed-citation>Wan, Z., Hook, S., and Hulley, G.: MOD11A1 MODIS/Terra Land Surface
Temperature/Emissivity Daily L3 Global 1km SIN Grid, NASA LP DAAC [data set],
<a href="https://doi.org/10.5067/MODIS/MOD11A1.006" target="_blank">https://doi.org/10.5067/MODIS/MOD11A1.006</a>, 2015.
</mixed-citation></ref-html>
<ref-html id="bib1.bib57"><label>57</label><mixed-citation>Xiao, Q., Chang, H. H., Geng, G., and Liu, Y.: An Ensemble Machine-Learning
Model To Predict Historical PM<sub>2.5</sub> Concentrations in China from Satellite
Data, Environ. Sci. Technol., 52, 13260–13269,
<a href="https://doi.org/10.1021/acs.est.8b02917" target="_blank">https://doi.org/10.1021/acs.est.8b02917</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib58"><label>58</label><mixed-citation>Xiao, Z., Liang, S., Wang, J., Chen, P., Yin, X., Zhang, L., and Song, J.:
Use of General Regression Neural Networks for Generating the GLASS Leaf Area
Index Product From Time-Series MODIS Surface Reflectance, IEEE T. Geosci.
Remote, 52, 209–223, <a href="https://doi.org/10.1109/tgrs.2013.2237780" target="_blank">https://doi.org/10.1109/tgrs.2013.2237780</a>, 2014.

</mixed-citation></ref-html>
<ref-html id="bib1.bib59"><label>59</label><mixed-citation>Xu, Y., Knudby, A., and Ho, H. C.: Estimating daily maximum air temperature
from MODIS in British Columbia, Canada, Int. J. Remote Sens., 35, 8108–8121,
<a href="https://doi.org/10.1080/01431161.2014.978957" target="_blank">https://doi.org/10.1080/01431161.2014.978957</a>, 2014.
</mixed-citation></ref-html>
<ref-html id="bib1.bib60"><label>60</label><mixed-citation>Yang,&thinsp;K. and He, J.: China meteorological forcing dataset (1979–2018),
National Tibetan Plateau Data Center [data set],
<a href="https://doi.org/10.11888/AtmosphericPhysics.tpe.249369.file" target="_blank">https://doi.org/10.11888/AtmosphericPhysics.tpe.249369.file</a>, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib61"><label>61</label><mixed-citation>Yao, R., Wang, L., Huang, X., Li, L., Sun, J., Wu, X., and Jiang, W.:
Developing a temporally accurate air temperature dataset for Mainland China,
Sci. Total Environ., 706, 136037,
<a href="https://doi.org/10.1016/j.scitotenv.2019.136037" target="_blank">https://doi.org/10.1016/j.scitotenv.2019.136037</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib62"><label>62</label><mixed-citation>Zeng, L., Wardlow, B., Tadesse, T., Shan, J., Hayes, M., Li, D., and Xiang,
D.: Estimation of Daily Air Temperature Based on MODIS Land Surface
Temperature Products over the Corn Belt in the US, Remote Sens., 7, 951–970,
<a href="https://doi.org/10.3390/rs70100951" target="_blank">https://doi.org/10.3390/rs70100951</a>, 2015.
</mixed-citation></ref-html>
<ref-html id="bib1.bib63"><label>63</label><mixed-citation>Zhang, H., Zhang, F., Ye, M., Che, T., and Zhang, G.: Estimating daily air
temperatures over the Tibetan Plateau by dynamically integrating MODIS LST
data, J. Geophys. Res.-Atmos., 121, 11425–11441,
<a href="https://doi.org/10.1002/2016jd025154" target="_blank">https://doi.org/10.1002/2016jd025154</a>, 2016.
</mixed-citation></ref-html>
<ref-html id="bib1.bib64"><label>64</label><mixed-citation>Zhang, H.: Estimation of daily average near-surface air temperature using
MODIS and AIRS data, 2017 2nd International Conference on Frontiers of
Sensors Technologies (ICFST),  377-381, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib65"><label>65</label><mixed-citation>Zhang, H., Zhang, F. A. N., Zhang, G., Ma, Y., Yang,&thinsp;K. U. N., and Ye, M.:
Daily air temperature estimation on glacier surfaces in the Tibetan Plateau
using MODIS LST data, J. Glaciol., 64, 132–147,
<a href="https://doi.org/10.1017/jog.2018.6" target="_blank">https://doi.org/10.1017/jog.2018.6</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib66"><label>66</label><mixed-citation>Zhang, W., Huang, Y., Yu, Y., and Sun, W.: Empirical models for estimating
daily maximum, minimum and mean air temperatures with MODIS land surface
temperatures, Int. J. Remote Sens., 32, 9415–9440,
<a href="https://doi.org/10.1080/01431161.2011.560622" target="_blank">https://doi.org/10.1080/01431161.2011.560622</a>, 2011.
</mixed-citation></ref-html>
<ref-html id="bib1.bib67"><label>67</label><mixed-citation>Zhang, X., Wang, D., Liu, Q., Yao, Y., Jia,&thinsp;K., He, T., Jiang, B., Wei, Y.,
Ma, H., and Zhao, X.: An operational approach for generating the global land
surface downward shortwave radiation product from MODIS data, IEEE T.
Geosci. Remote, 57, 4636–4650, <a href="https://doi.org/10.1109/TGRS.2019.2891945" target="_blank">https://doi.org/10.1109/TGRS.2019.2891945</a>,
2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib68"><label>68</label><mixed-citation>Zhu, W., Lű, A., and Jia, S.: Estimation of daily maximum and minimum
air temperature using MODIS land surface temperature products, Remote Sens.
Environ., 130, 62–73, <a href="https://doi.org/10.1016/j.rse.2012.10.034" target="_blank">https://doi.org/10.1016/j.rse.2012.10.034</a>, 2013.
</mixed-citation></ref-html>
<ref-html id="bib1.bib69"><label>69</label><mixed-citation>Zhu, W., Lű, A., Jia, S., Yan, J., and Mahmood, R.: Retrievals of
all-weather daytime air temperature from MODIS products, Remote Sens.
Environ., 189, 152–163, <a href="https://doi.org/10.1016/j.rse.2016.11.011" target="_blank">https://doi.org/10.1016/j.rse.2016.11.011</a>, 2017.
</mixed-citation></ref-html>--></article>
