<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing with OASIS Tables v3.0 20080202//EN" "https://jats.nlm.nih.gov/nlm-dtd/publishing/3.0/journalpub-oasis3.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:oasis="http://docs.oasis-open.org/ns/oasis-exchange/table" xml:lang="en" dtd-version="3.0" article-type="research-article">
  <front>
    <journal-meta><journal-id journal-id-type="publisher">HESS</journal-id><journal-title-group>
    <journal-title>Hydrology and Earth System Sciences</journal-title>
    <abbrev-journal-title abbrev-type="publisher">HESS</abbrev-journal-title><abbrev-journal-title abbrev-type="nlm-ta">Hydrol. Earth Syst. Sci.</abbrev-journal-title>
  </journal-title-group><issn pub-type="epub">1607-7938</issn><publisher>
    <publisher-name>Copernicus Publications</publisher-name>
    <publisher-loc>Göttingen, Germany</publisher-loc>
  </publisher></journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.5194/hess-30-5647-2026</article-id><title-group><article-title>Hydrochemistry and modeling nitrate concentration in farmland groundwater under different hydrological seasons by integrating hybrid quantum-classical ML, virtual sample generation and AlphaEarth Foundation</article-title><alt-title>Hydrochemistry and modeling nitrate concentration in farmland groundwater</alt-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author" equal-contrib="yes" corresp="no" rid="aff1 aff2 aff3">
          <name><surname>Xu</surname><given-names>Junjie</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" equal-contrib="yes" corresp="no" rid="aff4">
          <name><surname>Wei</surname><given-names>Xin</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="yes" rid="aff1 aff5">
          <name><surname>Yu</surname><given-names>Yilei</given-names></name>
          <email>yileiyu@hbu.edu.cn</email>
        <ext-link>https://orcid.org/0000-0001-7789-1321</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff2">
          <name><surname>Yang</surname><given-names>Lihu</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff6">
          <name><surname>Zhai</surname><given-names>Yuanzheng</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff7">
          <name><surname>Lv</surname><given-names>Cuicui</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff1">
          <name><surname>Song</surname><given-names>Xianfang</given-names></name>
          
        </contrib>
        <aff id="aff1"><label>1</label><institution>College of Life Sciences, Hebei University, Baoding, Hebei, 071000, China</institution>
        </aff>
        <aff id="aff2"><label>2</label><institution>Key Laboratory of Land Water Cycle and Surface Processes, Institute of Geographic Sciences and Natural Resources Research, Chinese Academy of Sciences, Beijing, 100101, China</institution>
        </aff>
        <aff id="aff3"><label>3</label><institution>University of Chinese Academy of Sciences, Beijing, 100049, China</institution>
        </aff>
        <aff id="aff4"><label>4</label><institution>College of Ecology and Environment, Institute of Disaster Prevention Science and Technology, Sanhe, Hebei, 065201, China</institution>
        </aff>
        <aff id="aff5"><label>5</label><institution>Engineering Research Center of Groundwater Pollution Control and Remediation, Ministry of Education of China, Beijing Normal University, Beijing, 100875, China</institution>
        </aff>
        <aff id="aff6"><label>6</label><institution>College of Water Sciences, Beijing Normal University, 100875, Beijing, China</institution>
        </aff>
        <aff id="aff7"><label>7</label><institution>Xiong'an Institute of Innovation, Xiong'an, 071899, China</institution>
        </aff><author-comment content-type="econtrib"><p>These authors contributed equally to this work.</p></author-comment>
      </contrib-group>
      <author-notes><corresp id="corr1">Yilei Yu (yileiyu@hbu.edu.cn)</corresp></author-notes><pub-date><day>8</day><month>September</month><year>2026</year></pub-date>
      
      <volume>30</volume>
      <issue>17</issue>
      <fpage>5647</fpage><lpage>5683</lpage>
      <history>
        <date date-type="received"><day>18</day><month>January</month><year>2026</year></date>
           <date date-type="rev-request"><day>29</day><month>January</month><year>2026</year></date>
           <date date-type="rev-recd"><day>5</day><month>August</month><year>2026</year></date>
           <date date-type="accepted"><day>30</day><month>August</month><year>2026</year></date>
      </history>
      <permissions>
        <copyright-statement>Copyright: © 2026 Junjie Xu et al.</copyright-statement>
        <copyright-year>2026</copyright-year>
      <license license-type="open-access"><license-p>This work is licensed under the Creative Commons Attribution 4.0 International License. To view a copy of this licence, visit <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link></license-p></license></permissions><self-uri xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026.html">This article is available from https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026.html</self-uri><self-uri xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026.pdf">The full text article is available as a PDF file from https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026.pdf</self-uri>
      <abstract><title>Abstract</title>

      <p id="d2e181">Precise seasonal prediction of groundwater nitrate concentrations in intensive agricultural areas faces challenges such as data sparsity, strong spatiotemporal heterogeneity, and complex hydro-biogeochemical processes. To address these issues, this study proposes an integrated prediction framework combining hybrid quantum-classical machine learning, advanced virtual sample generation (t-SNE-GMM-KNN), and remote sensing foundation model semantic embedding (AEF). Modeling was conducted across the 2022–2023 normal, dry, and wet seasons in Xiong'an New Area. Hydrochemical types were dominated by Ca-Mg-HCO<inline-formula><mml:math id="M1" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, controlled by mineral dissolution and evaporation. Nitrate concentrations were highest in the dry season (mean 42.93 mg L<sup>−1</sup>), driven by evaporative concentration. Spatially, high-value zones shifted: southeast (normal), central (dry), and northwest (wet). MixSIAR modeling based on isotopes indicated domestic sewage and livestock manure (74.1 %) as dominant sources. The t-SNE-GMM-KNN strategy mitigated small-sample bias while preserving nonlinear structure. When virtual samples were augmented to 10-fold, the Random Forest <inline-formula><mml:math id="M3" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> in the dry season increased from 0.284 to <inline-formula><mml:math id="M4" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.85. Furthermore, a hybrid quantum-classical Random Forest exhibited superior robustness for data sparsity, achieving peak performance in the normal season (<inline-formula><mml:math id="M5" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.962, RMSE <inline-formula><mml:math id="M6" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 5.73 mg L<sup>−1</sup>). Additionally, using only AEF embeddings achieved screening-level accuracy (<inline-formula><mml:math id="M8" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> up to 0.860), providing a feasible rapid survey scheme for extensive unmonitored regions, where field sampling is impractical, whereas the field-parameter-based model serves primarily to elucidate hydrochemical driving mechanisms and enables rapid on-site nitrate estimation using portable water quality meters. Correlation analysis identified TDS and EC as persistent top predictors (<inline-formula><mml:math id="M9" display="inline"><mml:mrow><mml:mi>r</mml:mi><mml:mi mathvariant="italic">&gt;</mml:mi></mml:mrow></mml:math></inline-formula> 0.8). This comprehensive framework offers a robust solution for seasonal nitrate prediction by distinguishing between mechanistic analysis, field-operational estimation, and large-scale screening and sustainable water management.</p>
  </abstract>
    
<funding-group>
<award-group id="gs1">
<funding-source>National Natural Science Foundation of China</funding-source>
<award-id>41601037</award-id>
</award-group>
</funding-group>
</article-meta>
  </front>
<body>
      

<sec id="Ch1.S1" sec-type="intro">
  <label>1</label><title>Introduction</title>
      <p id="d2e289">Nitrate (NO<inline-formula><mml:math id="M10" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) contamination in groundwater poses a serious threat to drinking water safety and ecosystem health, particularly in intensively managed agricultural regions (Wang et al., 2023). In China, groundwater nitrate pollution is a growing concern, national monitoring data from 2013 to 2017 revealed a nitrate exceedance rate exceeding 10 % relative to the Class III limit (88 mg L<sup>−1</sup> as NO<inline-formula><mml:math id="M12" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) of the Chinese Groundwater Quality Standard (GB/T 14848-2017), with Hebei Province reporting an alarming rate of 31.66 % in 2017. Over recent decades, escalating nitrate concentrations in surface and groundwater have been driven by intensified fertilizer use in agriculture, along with discharges of industrial and domestic wastewater (Zhang et al., 2018). Severe nitrate exceedances are especially prevalent in northern and northwestern China (Gu et al., 2013), where key contributors include domestic and industrial effluents, nitrification of soil organic nitrogen, and synthetic fertilizer application (Han et al., 2016). For instance, in the North China Plain, shallow groundwater nitrate exceedance rates range from 9.5 % to 34.1 % relative to the WHO drinking water guideline of 50 mg L<sup>−1</sup> NO<inline-formula><mml:math id="M14" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, and a rising trend persists at the regional scale, particularly in agricultural areas (Wang et al., 2018). In monsoonal temperate regions, seasonal shifts in precipitation, evapotranspiration, and groundwater recharge profoundly influence the transport, dilution, and accumulation of nitrate, leading to pronounced intra-annual variability in its concentration and spatial distribution (Gao et al., 2023). Consequently, understanding and forecasting nitrate dynamics across hydrological seasons is essential for informed groundwater management and pollution mitigation, but remains a formidable challenge due to the nonlinearity, high dimensionality, and data scarcity inherent in such systems (Deng et al., 2023).</p>
      <p id="d2e352">Traditional monitoring and modeling approaches face three critical limitations. First, field sampling campaigns though providing high-fidelity hydrochemical data are inherently sparse in space and time, especially for large-scale or rapidly changing environments, which are time-consuming, labor-intensive, and costly, limiting the spatial and temporal coverage of data (Cai et al., 2025). Second, while process-based models incorporate physical mechanisms, they require extensive parameterization and are computationally prohibitive for dynamic, multi-season forecasting at farm-to-regional scales (Feng et al., 2022). Hydrological seasonal variations (normal, dry, and wet seasons) significantly influence the migration and transformation of nitrogen in the soil-groundwater system (Chen et al., 2025). For instance, concentrated rainfall during the wet season (accounting for 60 %–80 % of annual precipitation) can promote the leaching of surface nitrogen into groundwater, leading to a 25-fold increase in stream nitrate concentrations during storm events compared to baseflow (Sebestyen et al., 2014), meanwhile, intense evaporation in the dry season leads to the accumulation of nitrate in shallow aquifers, where concentrations can exceed the US EPA drinking water standard of 10 mg L<sup>−1</sup> by 2–3 times (Liu et al., 2025c). These seasonal differences result in distinct hydrochemical characteristics and nitrate concentration distributions, increasing the complexity of prediction models (Wu et al., 2025). Third, even advanced machine learning (ML) techniques such as Random Forest (RF), despite their robustness to nonlinearity and multicollinearity, still rely heavily on sufficient representative samples to capture the multi-modal distribution and tail behavior of environmental variables, particularly for heavy-tailed pollutants like NO<inline-formula><mml:math id="M16" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> (Luo et al., 2022). Moreover, the small sample sizes obtained from discrete sampling often lead to data sparsity and skewed distributions, reducing the model's generalization ability by 30 %–50 % when applied to unmonitored areas and compromising the robustness and generalization ability of machine learning (ML) models trained on such data (Thunyawatcharakul et al., 2025; Wang et al., 2024).</p>
      <p id="d2e379">Despite these recognized limitations of conventional approaches, existing solutions remain fragmented and insufficient for seasonal groundwater nitrate prediction. Specifically, current virtual sample generation methods, whether statistical (e.g., SMOTE, GMM) or deep learning-based (e.g., VAEs, GANs), suffer from a critical paradox: they either fail to preserve the complex non-linear manifold structure of high-dimensional geochemical data or inherently require large training datasets that are unavailable in typical environmental monitoring scenarios (Farnia et al., 2023; Tung et al., 2023). Meanwhile, while remote sensing foundation models such as Google's AlphaEarth Foundation (AEF) have demonstrated success in surface parameter estimation, their applicability to subsurface groundwater quality prediction, particularly for dynamic seasonal modeling, remains entirely untested (Tollefson, 2025; Li et al., 2025a). Furthermore, although quantum machine learning (QML) theoretically offers advantages in capturing complex non-linear relationships through exponentially high-dimensional Hilbert space mapping, its practical efficacy in small-sample environmental datasets, especially when hybridized with classical ensemble methods, has not been systematically evaluated (Hong and Lopez, 2025; Oliveira Santos et al., 2025). Consequently, no existing framework simultaneously addresses the triple challenge of data sparsity, seasonal heterogeneity, and high-dimensional feature extraction for groundwater nitrate forecasting. To bridge these gaps, this study integrates three emerging methodological frontiers: advanced virtual sample generation, remote sensing foundation models, and quantum-enhanced machine learning. Recent advances in these domains offer partial solutions. Gaussian Mixture Models (GMM) and deep generative frameworks (e.g., VAEs, GANs) have shown promise in enriching training data, with GMM achieving an average similarity of 83.0 % between unmixed chemical spectra and ground truth in geochemical analysis (Farnia et al., 2023; Tung et al., 2023), yet these approaches exhibit critical constraints: they often fail to preserve the non-linear manifold structure of high-dimensional geochemical space or require large training sets, precisely what is lacking. Non-linear dimensionality reduction methods, such as t-SNE, excel at revealing latent clusters corresponding to distinct hydrological processes, with a classification accuracy of 92 % for annual daily hydrograph clustering in mountainous watersheds, yet lack explicit generative mechanisms (Wang et al., 2025; Tang and Carey, 2022). Meanwhile, the rise of foundation models in Earth observation exemplified by Google's AlphaEarth Foundation (AEF), offers unprecedented opportunities: its 64-dimensional semantic embeddings, derived from multi-sensor satellite time series (including Sentinel-2, Landsat, and Sentinel-1), implicitly encode land use, vegetation phenology, soil moisture, and anthropogenic footprints at 10 m resolution (Tollefson, 2025). These features have been successfully applied in land use classification and crop monitoring, but their potential for predicting groundwater nitrate concentrations, especially across different hydrological seasons remains underexplored (Li and Yu, 2026). Quantum machine learning (QML) further opens a new frontier. Parameterized Quantum Circuits (PQCs) can map classical inputs into exponentially high-dimensional quantum Hilbert spaces, generating entangled feature representations that reveal complex, non-linear patterns inaccessible to classical kernels (Hong and Lopez, 2025). For ozone concentration forecasting, a hybrid QML model achieved an <inline-formula><mml:math id="M17" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> of 94.12 % for 1-hour forecasts and 75.62 % for 6-hour forecasts, outperforming classical persistence models by a forecast skill of 31.01 %–57.46 % (Oliveira Santos et al., 2025). Recently, Saberian et al. (2025) developed HydroQuantum, a quantum-driven Python package implementing Quantum LSTM (QLSTM), hybrid quantum-classical LSTM, and Variational Quantum Circuits (VQC) for daily streamflow and stream water temperature simulations across the continental US. Their work demonstrated that quantum-enhanced architectures can effectively capture temporal dependencies in hydrological time series, particularly in snowmelt-driven and regulated basins, providing valuable methodological context for quantum-driven hydrological modeling. Crucially, analytical quantum feature extraction via Pauli-<inline-formula><mml:math id="M18" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> expectation values avoids the noisy sampling overhead of near-term quantum hardware, reducing computational latency by <inline-formula><mml:math id="M19" display="inline"><mml:mo>∼</mml:mo></mml:math></inline-formula> 80 % compared to sampling-based methods and making it viable for small-sample environmental modeling (Gujju et al., 2024).</p>
      <p id="d2e407">Furthermore, identifying the sources and controlling factors of nitrate pollution is crucial for improving prediction accuracy and guiding targeted pollution control measures. Isotopic analysis (<inline-formula><mml:math id="M20" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N-NO<inline-formula><mml:math id="M21" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M22" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-NO<inline-formula><mml:math id="M23" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) combined with the MixSIAR model has proven effective in quantitatively apportioning nitrate sources (Tian et al., 2025). Meanwhile, Bayesian models and SHapley Additive exPlanations (SHAP) analysis can reveal the key environmental variables driving nitrate concentration changes, enhancing the interpretability of prediction models (Alam et al., 2025a). Despite these advancements, several gaps persist in the current research: (1) Few studies have integrated hybrid quantum-classical ML with virtual sample augmentation to address small-sample challenges in seasonal nitrate prediction; (2) The potential of AEF remote sensing semantic features for groundwater nitrate prediction remains untested, particularly in comparison with in-situ measured parameters; (3) The combined effects of hydrological seasonal variations, nitrate source apportionment, and key environmental drivers on prediction model performance require systematic investigation.</p>
      <p id="d2e457">The North China Plain, an important agricultural production region in China, is characterized by high nitrogen input intensity and significant seasonal hydrological variations, making it an area prone to groundwater nitrate pollution (Liu et al., 2025a; Hou et al., 2025). Conducting field-scale research on nitrate pollution in this region is of great significance for the protection of regional water resources. The Xiong'an New Area was specifically selected as the core study site due to its unique combination of national strategic ecological importance, representative hydrogeological conditions, and severe pollution characteristics that exemplify regional challenges. Quantitatively, Hebei Province reported a groundwater nitrate exceedance rate of 31.66 % in 2017, significantly higher than the national average, with Xiong'an situated in the heart of this high-risk zone (Xiong et al., 2025). Furthermore, the study site represents a typical high-intensity agricultural system with annual nitrogen inputs ranging from 540 to 660 kg N ha<sup>−1</sup>, coupled with a shallow groundwater table (5.0–20.0 m) highly susceptible to surface loading (Xu et al., 2021). Moreover, existing investigations reveal severe anthropogenic impacts including legacy nitrogen accumulation (<inline-formula><mml:math id="M25" display="inline"><mml:mo lspace="0mm">∼</mml:mo></mml:math></inline-formula> 320 kg N ha<sup>−1</sup> yr<sup>−1</sup> surplus) and sewage irrigation contributing up to 58.3 % of groundwater chemical signatures. This setting provides an ideal natural laboratory for testing seasonal prediction frameworks, as the distinct monsoonal climate (60 %–80 % precipitation concentrated in summer) creates pronounced hydrochemical contrasts between dry, wet, and normal seasons, directly challenging model robustness under data sparsity (Xu et al., 2021). A mechanistic understanding of how nitrate concentrations vary across these hydrological seasons (normal, dry, wet) and their controlling factors is crucial for regional water resource management. To address the critical gaps outlined above, this study establishes three clear objectives: (1) to develop a t-SNE-GMM-KNN virtual sample generation strategy that preserves geochemical manifold structure while expanding small datasets for robust seasonal modeling; (2) to construct a hybrid quantum-classical Random Forest architecture that enhances feature discriminability without quantum hardware limitations, specifically targeting the high variability and data scarcity characteristic of dry-season nitrate prediction; and (3) to evaluate the feasibility of AlphaEarth Foundation (AEF) semantic embeddings as standalone predictors for rapid groundwater nitrate screening in unmonitored regions. The relevance of this research lies in its provision of a unified methodological framework that transcends single-site application, offering a scalable solution for seasonal groundwater quality prediction in intensively managed agricultural landscapes worldwide. Practically, the results enable three distinct operational modes: (i) mechanistic analysis using field-measurable parameters (pH, EC, TDS, etc.) to elucidate hydrochemical driving processes; (ii) rapid on-site nitrate estimation using portable multi-parameter meters when laboratory analysis is unavailable; and (iii) large-scale regional risk assessment using remote sensing embeddings for areas lacking monitoring infrastructure. By distinguishing between these application scenarios, the framework provides actionable decision-support tools for water resource managers to implement targeted pollution control measures, optimize monitoring network design, and prioritize remediation efforts across hydrological seasons.</p>
      <p id="d2e503">Compared with existing research on groundwater quality prediction based on multiple ML algorithms and large datasets, the novelty of this paper lies in three aspects: (1) the t-SNE-GMM-KNN strategy preserves the non-linear manifold structure of high-dimensional geochemical space during generation; (2) we systematically explore the application of hybrid quantum-classical machine learning in groundwater quality prediction, evaluating its robustness and feature discriminability in small-sample seasonal dynamics; and (3) we conduct a comparative analysis of prediction performance using traditional field observation data versus the first-time application of AlphaEarth Foundation (AEF) semantic embeddings for nitrate estimation.</p>
</sec>
<sec id="Ch1.S2">
  <label>2</label><title>Materials and Methods</title>
<sec id="Ch1.S2.SS1">
  <label>2.1</label><title>Study area</title>
      <p id="d2e521">The North China Plain is one of China's most important agricultural production bases. This study focuses on the Xiong'an New Area, situated in the central part of Hebei Province, as a representative research site within this plain. Located in the core region defined by Beijing, Tianjin, and Baoding, it boasts an advantageous geographical position, with straight-line distances of 105 km to both Beijing and Tianjin, and 30 km to Baoding. Its geographical coordinates range from 38°43<sup>′</sup> to 39°10<sup>′</sup> N latitude and from 115°38<sup>′</sup> to 116°20<sup>′</sup> E longitude, covering an area of approximately 1770 km<sup>2</sup>. The specific study area is an unmanned farm located in Xieyeqiao Village, Nanzhang Town, Rongcheng County, within the Xiong'an New Area (Fig. 1). The farm covers an area of 3000 ha and primarily cultivates two main grain crops: wheat and corn.</p>

      <fig id="F1" specific-use="star"><label>Figure 1</label><caption><p id="d2e571">Study area map showing the sampling location and sampling points. The administrative boundaries are based on the standard map (approval no. (2023)2766) from the Standard Map Service of the Ministry of Natural Resources of China; the satellite basemap is from © Google Earth (Imagery © 2026 Airbus, CNES/Airbus, MaxarTechnologies).</p></caption>
          <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f01.png"/>

        </fig>

      <p id="d2e580">Cultivated land, predominantly dryland, occupies a large proportion of Xiong'an New Area, where traditional practices involve high application rates of nitrogen fertilizer and manure. The study site follows these practices, with an annual nitrogen application rate of 540–660 kg (N) ha<sup>−1</sup> yr<sup>−1</sup>, mainly as urea (46 % N), posing a high risk of groundwater nitrogen pollution; the dense surrounding rural population adds further input from domestic sewage discharge. Irrigation is scheduled by crop phenology: wheat receives muddy water irrigation before sowing and at the overwintering, reviving, and jointing stages, while maize receives a single irrigation after sowing. The region has a temperate continental monsoon climate, with a mean annual temperature of 12.6 °C (January average <inline-formula><mml:math id="M35" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>5 °C; July average 26.1 °C). The mean annual precipitation is 480.8 mm, of which nearly 80 % falls between June and September (Fig. 2). The multi-year average evaporation is 1661.1 mm and the average water surface evaporation is 1,761.7 mm. The mean annual sunshine duration and wind speed are 2335.2 h and 1.7 m s<sup>−1</sup>, respectively, and the frost-free period averages 204 d. The soil texture is predominantly silt loam, with high-clay interlayers (clay and silty clay) at depths of 2–8.5 m. In the thick vadose zone, nitrogen exists predominantly as organic nitrogen (<inline-formula><mml:math id="M37" display="inline"><mml:mo lspace="0mm">∼</mml:mo></mml:math></inline-formula> 97 % of total nitrogen), and the shallow layer at 3–6 m depth stores about half of the total nitrate reserves in the North China Plain (Li et al., 2025b). Groundwater occurs mainly in Quaternary unconsolidated porous aquifers (sampling well depths: 70–120 m); the shallow aquifer group consists of silty fine to medium sand with medium water abundance (single-well yields: <inline-formula><mml:math id="M38" display="inline"><mml:mo>∼</mml:mo></mml:math></inline-formula> 1000–3000 m<sup>3</sup> d<sup>−1</sup>). The depth to the shallow groundwater table ranges from 5.0 to 35.0 m, with fluctuations driven by precipitation infiltration and agricultural extraction. Precipitation is the primary recharge source for farmland groundwater, and artificial extraction for irrigation is the main discharge pathway (Huang and Chen, 2012). Detailed climate and hydrological data are given in Table 1.</p>

      <fig id="F2" specific-use="star"><label>Figure 2</label><caption><p id="d2e665">Temperature and precipitation trends in study area (1952–2025).</p></caption>
          <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f02.png"/>

        </fig>

<table-wrap id="T1" specific-use="star"><label>Table 1</label><caption><p id="d2e677">Summary of key climatic and hydrological variables in the study area.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="4">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:colspec colnum="3" colname="col3" align="left"/>
     <oasis:colspec colnum="4" colname="col4" align="left"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Category</oasis:entry>
         <oasis:entry colname="col2">Variable</oasis:entry>
         <oasis:entry colname="col3">Value/Range</oasis:entry>
         <oasis:entry colname="col4">Unit</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">Temperature</oasis:entry>
         <oasis:entry colname="col2">Mean annual temperature</oasis:entry>
         <oasis:entry colname="col3">12.6</oasis:entry>
         <oasis:entry colname="col4">°C</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean January temperature</oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M41" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>5.0</oasis:entry>
         <oasis:entry colname="col4">°C</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean July temperature</oasis:entry>
         <oasis:entry colname="col3">26.1</oasis:entry>
         <oasis:entry colname="col4">°C</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Precipitation</oasis:entry>
         <oasis:entry colname="col2">Mean annual precipitation</oasis:entry>
         <oasis:entry colname="col3">480.8</oasis:entry>
         <oasis:entry colname="col4">mm</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Rainy season (Jun–Sep) contribution</oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M42" display="inline"><mml:mo>∼</mml:mo></mml:math></inline-formula> 80</oasis:entry>
         <oasis:entry colname="col4">% of annual total</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Evaporation</oasis:entry>
         <oasis:entry colname="col2">Mean annual evaporation</oasis:entry>
         <oasis:entry colname="col3">1661.10</oasis:entry>
         <oasis:entry colname="col4">mm</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean annual water surface evaporation</oasis:entry>
         <oasis:entry colname="col3">1761.70</oasis:entry>
         <oasis:entry colname="col4">mm</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Solar and Wind</oasis:entry>
         <oasis:entry colname="col2">Mean annual sunshine duration</oasis:entry>
         <oasis:entry colname="col3">2335.20</oasis:entry>
         <oasis:entry colname="col4">h</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean annual wind speed</oasis:entry>
         <oasis:entry colname="col3">1.7</oasis:entry>
         <oasis:entry colname="col4">m s<sup>−1</sup></oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Frost-free period</oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M44" display="inline"><mml:mo>∼</mml:mo></mml:math></inline-formula> 204</oasis:entry>
         <oasis:entry colname="col4">d</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Soil and Vadose Zone</oasis:entry>
         <oasis:entry colname="col2">Dominant soil texture</oasis:entry>
         <oasis:entry colname="col3">Silt loam</oasis:entry>
         <oasis:entry colname="col4">–</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Organic N proportion in vadose zone</oasis:entry>
         <oasis:entry colname="col3"><inline-formula><mml:math id="M45" display="inline"><mml:mo>∼</mml:mo></mml:math></inline-formula> 97</oasis:entry>
         <oasis:entry colname="col4">% of total N</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Peak nitrate storage depth</oasis:entry>
         <oasis:entry colname="col3">3–6</oasis:entry>
         <oasis:entry colname="col4">m</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Groundwater</oasis:entry>
         <oasis:entry colname="col2">Aquifer type</oasis:entry>
         <oasis:entry colname="col3">Quaternary unconsolidated porous</oasis:entry>
         <oasis:entry colname="col4">–</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Sampling well depth</oasis:entry>
         <oasis:entry colname="col3">70–120</oasis:entry>
         <oasis:entry colname="col4">m</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Shallow water table depth</oasis:entry>
         <oasis:entry colname="col3">5.0–35.0</oasis:entry>
         <oasis:entry colname="col4">m</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Single-well yield</oasis:entry>
         <oasis:entry colname="col3">1000–3000</oasis:entry>
         <oasis:entry colname="col4">m<sup>3</sup> d<sup>−1</sup></oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Dominant recharge source</oasis:entry>
         <oasis:entry colname="col3">Precipitation infiltration</oasis:entry>
         <oasis:entry colname="col4">–</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Dominant discharge pathway</oasis:entry>
         <oasis:entry colname="col3">Agricultural extraction</oasis:entry>
         <oasis:entry colname="col4">–</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

</sec>
<sec id="Ch1.S2.SS2">
  <label>2.2</label><title>Data collection and measurements</title>
<sec id="Ch1.S2.SS2.SSS1">
  <label>2.2.1</label><title>Field sampling data and laboratory analysis</title>
      <p id="d2e1060">Field investigations and the collection of hydrochemical and isotopic samples were conducted in the study area from 2022 to 2023. The temporal resolution for groundwater sampling was designed as seasonal snapshots, with campaigns occurring in October 2022 (normal season), April 2023 (dry season), and August 2023 (wet season), representing a temporal interval of approximately 4–6 months between surveys. A total of 66, 65, and 50 groundwater samples were collected in October 2022, April 2023, and August 2023, respectively. All groundwater samples were obtained from existing agricultural irrigation wells within the study area. In-situ physicochemical parameters (<inline-formula><mml:math id="M48" display="inline"><mml:mi>T</mml:mi></mml:math></inline-formula>, pH, TDS, DO, EC, and ORP) were measured using a Hach HQ400 multi-parameter meter (Li et al., 2022), and major ions, nitrogen species, and stable isotopes (<inline-formula><mml:math id="M49" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>H, <inline-formula><mml:math id="M50" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O, <inline-formula><mml:math id="M51" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N-NO<inline-formula><mml:math id="M52" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M53" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-NO<inline-formula><mml:math id="M54" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) were analyzed in the laboratory following standard protocols. Detailed sampling procedures, quality control measures, and analytical methods are provided in the Supplement (Sects. S1 and S2).</p>
</sec>
<sec id="Ch1.S2.SS2.SSS2">
  <label>2.2.2</label><title>Google AlphaEarth Foundation</title>
      <p id="d2e1147">To compare with predictions based on in-situ field sampling data and to validate the accuracy of remote sensing-based nitrate prediction, this study incorporates the Google AlphaEarth Foundation (AEF) dataset. AEF provides a 64-dimensional surface semantic embedding vector (A00–A63) at 10 m spatial and annual temporal resolution, generated by pre-training on multi-source satellite imagery (e.g., Sentinel-2, Landsat). These embeddings implicitly encode environmental semantics such as land cover, vegetation dynamics, soil moisture, and human activity intensity (Alvarez et al., 2025; Tollefson, 2025). The 64-dimensional AEF vectors were extracted at each sampling point via the Google Earth Engine platform, and Principal Component Analysis (PCA) was applied to retain the minimum number of components explaining at least 95 % of the cumulative variance. The PCA-reduced features served as model inputs. Details of the AEF data processing workflow are provided in the Supplement (Sect. S3).</p>
</sec>
</sec>
<sec id="Ch1.S2.SS3">
  <label>2.3</label><title>Statistical methods for source apportionment and driver analysis</title>
<sec id="Ch1.S2.SS3.SSS1">
  <label>2.3.1</label><title>MixSIAR model and isotopic composition of nitrate sources</title>
      <p id="d2e1166">MixSIAR incorporates prior information on end-member values and their errors and distributions, and estimates the proportional contribution of each source via Markov Chain Monte Carlo (MCMC) iteration (Stock et al., 2018). In this study, it was applied to apportion five potential NO<inline-formula><mml:math id="M55" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> sources: precipitation (NP), soil organic nitrogen (SON), synthetic NH<inline-formula><mml:math id="M56" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mo>+</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> fertilizer (NHF), synthetic NO<inline-formula><mml:math id="M57" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> fertilizer (NOF), and domestic sewage and manure (DSM), with the source end-member values listed in Table 2 (Mao et al., 2023; Gao et al., 2023; Torres-Martínez et al., 2021).</p>

<table-wrap id="T2"><label>Table 2</label><caption><p id="d2e1208">Summary statistics of <inline-formula><mml:math id="M58" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O and <inline-formula><mml:math id="M59" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N for potential nitrate sources.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="5">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right" colsep="1"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:thead>
       <oasis:row>
         <oasis:entry colname="col1">Sources</oasis:entry>
         <oasis:entry rowsep="1" namest="col2" nameend="col3" align="center" colsep="1"><inline-formula><mml:math id="M60" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-NO<inline-formula><mml:math id="M61" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry rowsep="1" namest="col4" nameend="col5" align="center"><inline-formula><mml:math id="M62" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N-NO<inline-formula><mml:math id="M63" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula></oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean</oasis:entry>
         <oasis:entry colname="col3">SD</oasis:entry>
         <oasis:entry colname="col4">Mean</oasis:entry>
         <oasis:entry colname="col5">SD</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">NP</oasis:entry>
         <oasis:entry colname="col2">57.2</oasis:entry>
         <oasis:entry colname="col3">6.9</oasis:entry>
         <oasis:entry colname="col4">0.6</oasis:entry>
         <oasis:entry colname="col5">1.5</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">NHF</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M64" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>4.1</oasis:entry>
         <oasis:entry colname="col3">2.7</oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M65" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>2.1</oasis:entry>
         <oasis:entry colname="col5">0.7</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">NOF</oasis:entry>
         <oasis:entry colname="col2">21.7</oasis:entry>
         <oasis:entry colname="col3">2.9</oasis:entry>
         <oasis:entry colname="col4">0.2</oasis:entry>
         <oasis:entry colname="col5">2.3</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SON</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M66" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>2.7</oasis:entry>
         <oasis:entry colname="col3">4.4</oasis:entry>
         <oasis:entry colname="col4">3.8</oasis:entry>
         <oasis:entry colname="col5">1.8</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">DSM</oasis:entry>
         <oasis:entry colname="col2">6.1</oasis:entry>
         <oasis:entry colname="col3">1.6</oasis:entry>
         <oasis:entry colname="col4">17.4</oasis:entry>
         <oasis:entry colname="col5">3.9</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

</sec>
<sec id="Ch1.S2.SS3.SSS2">
  <label>2.3.2</label><title>Pearson correlation and Bayesian linear regression model</title>
      <p id="d2e1442">Pearson correlation coefficients (<inline-formula><mml:math id="M67" display="inline"><mml:mi>r</mml:mi></mml:math></inline-formula>) were calculated to quantify the linear relationships between nitrate concentrations and individual hydrochemical parameters (Su et al., 2025). We performed a two-tailed significance test for each correlation coefficient to determine whether the linear correlation was statistically significant. The significance criteria were set as: <inline-formula><mml:math id="M68" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mi mathvariant="italic">&lt;</mml:mi><mml:mn mathvariant="normal">0.001</mml:mn></mml:mrow></mml:math></inline-formula> (highly significant), 0.001 <inline-formula><mml:math id="M69" display="inline"><mml:mo>≤</mml:mo></mml:math></inline-formula> <inline-formula><mml:math id="M70" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mi mathvariant="italic">&lt;</mml:mi><mml:mn mathvariant="normal">0.01</mml:mn></mml:mrow></mml:math></inline-formula> (very significant), 0.01 <inline-formula><mml:math id="M71" display="inline"><mml:mo>≤</mml:mo></mml:math></inline-formula> <inline-formula><mml:math id="M72" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mi mathvariant="italic">&lt;</mml:mi><mml:mn mathvariant="normal">0.05</mml:mn></mml:mrow></mml:math></inline-formula> (significant), and <inline-formula><mml:math id="M73" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mo>≥</mml:mo><mml:mn mathvariant="normal">0.05</mml:mn></mml:mrow></mml:math></inline-formula> (not significant) (Ma et al., 2025).</p>
      <p id="d2e1515">Bayesian linear regression was used to infer parameter posterior distributions, providing uncertainty quantification and reducing overfitting under small-sample conditions. <inline-formula><mml:math id="M74" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula>-score standardized nitrate concentration was regressed on the standardized predictors, with residuals assumed normal (mean 0, variance <inline-formula><mml:math id="M75" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">σ</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>). Weakly informative priors were used (Charizanos and Demirhan, 2023): a Student-<inline-formula><mml:math id="M76" display="inline"><mml:mi>t</mml:mi></mml:math></inline-formula> prior for the intercept, standard normal priors for the coefficients, and an exponential prior for the residual standard deviation. Posterior inference was performed via Markov Chain Monte Carlo (MCMC) sampling in the brms package in R (Addy et al., 2024). The probability of direction (pd), the proportion of each coefficient's posterior on the same side of 0 (Weller et al., 2022), quantified directional effects (0.5–1; higher values indicate stronger evidence).</p>
</sec>
</sec>
<sec id="Ch1.S2.SS4">
  <label>2.4</label><title>t-SNE-GMM-KNN virtual sample generation and quality assessment</title>
<sec id="Ch1.S2.SS4.SSS1">
  <label>2.4.1</label><title>t-SNE-GMM-KNN: based on nonlinear structure modeling in feature space</title>
      <p id="d2e1559">To mitigate overfitting and poor generalization caused by data sparsity and skewed distributions in small-sample modeling, we propose a three-stage virtual sample generation strategy: t-SNE-Gaussian Mixture Sampling with KNN inverse mapping, which preserves the non-linear manifold structure and multi-modal distribution of the original high-dimensional feature space while generating physically plausible and statistically consistent virtual samples. Virtual samples are generated independently for each hydrological season (normal, dry, wet) within the hydrochemical feature space, i.e., the space spanned by the in-situ water-quality parameters and nitrate concentration, without temporal extrapolation or cross-season fusion, thereby augmenting sample density within each season while preserving the inherent seasonal hydrochemical heterogeneity. The workflow is as follows:</p>
      <p id="d2e1562"><list list-type="order">
              <list-item>

      <p id="d2e1567"><italic>Data standardization.</italic></p>

      <p id="d2e1571">All input features are <inline-formula><mml:math id="M77" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula>-score standardized to eliminate scale differences and stabilize the subsequent dimensionality reduction (Jamshidi et al., 2022). Notably, the target variable (NO<inline-formula><mml:math id="M78" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) is standardized alongside the predictors (pH, <inline-formula><mml:math id="M79" display="inline"><mml:mi>T</mml:mi></mml:math></inline-formula>, EC, DO, ORP, Salt, TDS) to preserve the joint predictor-target distribution during generation.</p>
              </list-item>
              <list-item>

      <p id="d2e1603"><italic>t-SNE non-linear dimensionality reduction.</italic></p>

      <p id="d2e1607">t-Distributed Stochastic Neighbor Embedding (t-SNE) is employed to map the high-dimensional feature space into a low-dimensional latent space (<inline-formula><mml:math id="M80" display="inline"><mml:mrow><mml:mi>d</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn></mml:mrow></mml:math></inline-formula>) (Islam et al., 2023; Liu et al., 2021). Compared to linear methods such as PCA, t-SNE better preserves the local non-linear manifold structure of hydrochemical data, which is critical for capturing clustered subgroups corresponding to distinct hydrological and contamination processes (Wang et al., 2025). The perplexity was set to 10, suitable for small sample sizes (<inline-formula><mml:math id="M81" display="inline"><mml:mrow><mml:mi>n</mml:mi><mml:mi mathvariant="italic">&lt;</mml:mi><mml:mn mathvariant="normal">100</mml:mn></mml:mrow></mml:math></inline-formula>) while balancing local neighborhood preservation against global structure (Kobak and Berens, 2019). PCA initialization was used to ensure reproducibility and accelerate convergence, with a maximum of 1000 iterations and a gradient descent tolerance of <inline-formula><mml:math id="M82" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">5</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>.</p>
              </list-item>
              <list-item>

      <p id="d2e1655"><italic>GMM clustering and optimal component selection.</italic></p>

      <p id="d2e1659">In the t-SNE-reduced low-dimensional space, a Gaussian Mixture Model (GMM) is constructed to characterize the probability density distribution of the data (Jia et al., 2022). The GMM assumes that the data are generated from a linear combination of several Gaussian distributions. The GMM captures the multimodal nature of the hydrochemical data (e.g., distinct clusters corresponding to different contamination levels or redox conditions). The weights, means, and covariance matrices of each Gaussian component are estimated via the Expectation-Maximization (EM) algorithm, thereby accurately capturing the complex distribution patterns of the data (Yan and Huang, 2023). To avoid subjectively setting the number of clusters, the Bayesian Information Criterion (BIC) is used to automatically optimize the number of components, <inline-formula><mml:math id="M83" display="inline"><mml:mi>K</mml:mi></mml:math></inline-formula>, within the range (1 to 7, where the upper bound was set to approximately <inline-formula><mml:math id="M84" display="inline"><mml:mrow><mml:mi>n</mml:mi><mml:mo>/</mml:mo><mml:mn mathvariant="normal">10</mml:mn></mml:mrow></mml:math></inline-formula> to prevent overfitting given the small sample size) (Ghodba et al., 2025):

                    <disp-formula id="Ch1.E1" content-type="numbered"><label>1</label><mml:math id="M85" display="block"><mml:mrow><mml:mi mathvariant="normal">BIC</mml:mi><mml:mo>(</mml:mo><mml:mi>K</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mi>log⁡</mml:mi><mml:mi>L</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mi>K</mml:mi></mml:msub><mml:mi>log⁡</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:math></disp-formula>

                  where <inline-formula><mml:math id="M86" display="inline"><mml:mi>L</mml:mi></mml:math></inline-formula> is the model's likelihood, <inline-formula><mml:math id="M87" display="inline"><mml:mi>k</mml:mi></mml:math></inline-formula> is the total number of free parameters for a <inline-formula><mml:math id="M88" display="inline"><mml:mi>K</mml:mi></mml:math></inline-formula>-component model, and <inline-formula><mml:math id="M89" display="inline"><mml:mi>n</mml:mi></mml:math></inline-formula> is the sample size. The value of <inline-formula><mml:math id="M90" display="inline"><mml:mi>K</mml:mi></mml:math></inline-formula> corresponding to the minimum BIC is selected as the optimal number of components, ensuring a balance between goodness-of-fit and model complexity.</p>
              </list-item>
              <list-item>

      <p id="d2e1756"><italic>Virtual sample generation and inverse mapping.</italic></p>

      <p id="d2e1760">Virtual points were randomly sampled from the optimal GMM's joint probability distribution in the 2D t-SNE latent space, thereby inheriting the multi-modality and covariance structure of the original data. Generation factors of 1–10<inline-formula><mml:math id="M91" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> the original sample size were tested, with 10<inline-formula><mml:math id="M92" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> yielding optimal model performance. The sampled points were then mapped back to the original 8-dimensional hydrochemical feature space (including NO<inline-formula><mml:math id="M93" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) using a <inline-formula><mml:math id="M94" display="inline"><mml:mi>K</mml:mi></mml:math></inline-formula>-Nearest Neighbors (KNN) regressor fitted on the original t-SNE embeddings (inputs) and standardized hydrochemical data (outputs) (Niu et al., 2025), with n_neighbors<inline-formula><mml:math id="M95" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula>min(5, n_original-1) to balance inverse-mapping smoothness against local fidelity and avoid over-smoothing extreme values. After inverse standardization, each virtual sample contains synchronized values of all predictors and the target NO<inline-formula><mml:math id="M96" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentration, ensuring the one-to-one predictor-target correspondence required for RF modeling.</p>
              </list-item>
              <list-item>

      <p id="d2e1819"><italic>Physical constraints and quality control.</italic></p>

      <p id="d2e1823">For the target variable NO<inline-formula><mml:math id="M97" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, a non-negativity constraint was imposed post-generation to prevent non-physical values from the regression approximation, while other variables were allowed to fluctuate within reasonable ranges without hard clipping. Consistency between virtual and measured samples was validated by comparing means, standard deviations, coefficients of variation, extreme-value ranges, and boxplot distributions, confirming no systematic bias or outliers. The virtual samples, each containing both predictor variables and nitrate concentration, were concatenated with the measured data to form an augmented training set, in which the 7 water-quality parameters (pH, <inline-formula><mml:math id="M98" display="inline"><mml:mi>T</mml:mi></mml:math></inline-formula>, EC, DO, ORP, Salt, TDS) served as inputs (<inline-formula><mml:math id="M99" display="inline"><mml:mi>X</mml:mi></mml:math></inline-formula>) and NO<inline-formula><mml:math id="M100" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> as the target (<inline-formula><mml:math id="M101" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula>), thereby increasing sample size while preserving the geochemical integrity of feature-target relationships. The physical plausibility and distributional consistency of the virtual samples were validated using the Kolmogorov-Smirnov (KS) test, Jensen-Shannon (JS) divergence, and Maximum Mean Discrepancy (MMD), which assess distributional fidelity from local, overall, and high-dimensional perspectives, respectively.</p>
              </list-item>
            </list></p>
</sec>
<sec id="Ch1.S2.SS4.SSS2">
  <label>2.4.2</label><title>Validation metrics for virtual sample quality assessment</title>
      <p id="d2e1881"><list list-type="order">
              <list-item>

      <p id="d2e1886"><italic>Kolmogorov-Smirnov (KS) Test.</italic></p>

      <p id="d2e1890">The two-sample KS test was employed to quantify the maximum vertical distance between empirical cumulative distribution functions (ECDFs) of observed and virtual samples (Li and Yu, 2026). For each hydrochemical parameter, the test statistic <inline-formula><mml:math id="M102" display="inline"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi mathvariant="normal">KS</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is defined as:

                    <disp-formula id="Ch1.E2" content-type="numbered"><label>2</label><mml:math id="M103" display="block"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi mathvariant="normal">KS</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="normal">sup</mml:mi><mml:mi>x</mml:mi></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>m</mml:mi></mml:msub><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:math></disp-formula>

                  where <inline-formula><mml:math id="M104" display="inline"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M105" display="inline"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>m</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> represent the ECDFs of observed (<inline-formula><mml:math id="M106" display="inline"><mml:mi>n</mml:mi></mml:math></inline-formula> samples) and virtual (<inline-formula><mml:math id="M107" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula> samples) datasets, respectively. The <inline-formula><mml:math id="M108" display="inline"><mml:mi>P</mml:mi></mml:math></inline-formula>-value tests the null hypothesis that both samples derive from the same underlying distribution. Following established criteria for environmental data synthesis (Farnia et al., 2023; Tung et al., 2023), <inline-formula><mml:math id="M109" display="inline"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi mathvariant="normal">KS</mml:mi></mml:msub><mml:mi mathvariant="italic">&lt;</mml:mi><mml:mn mathvariant="normal">0.15</mml:mn></mml:mrow></mml:math></inline-formula> was adopted as the strict threshold for high fidelity and <inline-formula><mml:math id="M110" display="inline"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi mathvariant="normal">KS</mml:mi></mml:msub><mml:mi mathvariant="italic">&lt;</mml:mi><mml:mn mathvariant="normal">0.20</mml:mn></mml:mrow></mml:math></inline-formula> as the acceptable limit for parameters with inherent geochemical heterogeneity (Michalek et al., 2023; Li et al., 2023).</p>
              </list-item>
              <list-item>

      <p id="d2e2028"><italic>Jensen-Shannon (JS) Divergence.</italic></p>

      <p id="d2e2032">To measure the symmetric similarity between the observed distribution <inline-formula><mml:math id="M111" display="inline"><mml:mi>P</mml:mi></mml:math></inline-formula> and virtual distribution <inline-formula><mml:math id="M112" display="inline"><mml:mi>Q</mml:mi></mml:math></inline-formula> in a bounded metric [0, 1], the JS divergence was computed (Van Katwyk et al., 2023):

                    <disp-formula id="Ch1.E3" content-type="numbered"><label>3</label><mml:math id="M113" display="block"><mml:mrow><mml:mi mathvariant="normal">JS</mml:mi><mml:mo>(</mml:mo><mml:mi>P</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi>Q</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">1</mml:mn><mml:mn mathvariant="normal">2</mml:mn></mml:mfrac></mml:mstyle><mml:mi mathvariant="normal">KL</mml:mi><mml:mo>(</mml:mo><mml:mi>P</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi mathvariant="script">M</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">1</mml:mn><mml:mn mathvariant="normal">2</mml:mn></mml:mfrac></mml:mstyle><mml:mi mathvariant="normal">KL</mml:mi><mml:mo>(</mml:mo><mml:mi>Q</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi mathvariant="script">M</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>

                  where <inline-formula><mml:math id="M114" display="inline"><mml:mrow><mml:mi>M</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula>1/2(<inline-formula><mml:math id="M115" display="inline"><mml:mrow><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>Q</mml:mi></mml:mrow></mml:math></inline-formula>) represents the midpoint distribution, and KL denotes the Kullback-Leibler divergence. Unlike KL divergence, JS divergence is symmetric and bounded, making it particularly suitable for evaluating generative models in environmental applications (Balogun et al., 2024). The JS divergence was calculated using histogram-based probability density estimation (20 bins) with Laplace smoothing to handle zero-frequency bins. JS <inline-formula><mml:math id="M116" display="inline"><mml:mi mathvariant="italic">&lt;</mml:mi></mml:math></inline-formula> 0.05 indicates high distributional consistency, 0.05-0.10 moderate similarity, and JS <inline-formula><mml:math id="M117" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.10 significant distributional drift that may compromise model training integrity (Farnia et al., 2023) (Farnia et al., 2023).</p>
              </list-item>
              <list-item>

      <p id="d2e2155"><italic>Maximum Mean Discrepancy (MMD).</italic></p>

      <p id="d2e2159">Given the high-dimensional nature of hydrochemical feature space (<inline-formula><mml:math id="M118" display="inline"><mml:mrow><mml:mi>d</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 8 parameters), we employed MMD in the Reproducing Kernel Hilbert Space (RKHS) to capture non-linear distributional differences without parametric assumptions (Kalinin et al., 2025). Using the Radial Basis Function (RBF) kernel <inline-formula><mml:math id="M119" display="inline"><mml:mrow><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>exp⁡</mml:mi><mml:mo>(</mml:mo><mml:mo>-</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:mi>y</mml:mi><mml:mo>|</mml:mo><mml:msup><mml:mo>|</mml:mo><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>/</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:msup><mml:mi mathvariant="italic">σ</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> with bandwidth <inline-formula><mml:math id="M120" display="inline"><mml:mi mathvariant="italic">σ</mml:mi></mml:math></inline-formula> set via the median heuristic, the squared MMD between observed samples <inline-formula><mml:math id="M121" display="inline"><mml:mrow><mml:mi>X</mml:mi><mml:mo>=</mml:mo><mml:msubsup><mml:mfenced open="{" close="}"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mfenced><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> and virtual samples <inline-formula><mml:math id="M122" display="inline"><mml:mrow><mml:mi>Y</mml:mi><mml:mo>=</mml:mo><mml:mo mathvariant="italic">{</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:msubsup><mml:mo mathvariant="italic">}</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> was estimated as (Li et al., 2025a):

                    <disp-formula id="Ch1.E4" content-type="numbered"><label>4</label><mml:math id="M123" display="block"><mml:mtable class="split" rowspacing="0.2ex" displaystyle="true" columnalign="right left"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="normal">MMD</mml:mi><mml:mo>[</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mi>Y</mml:mi><mml:mo>]</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>=</mml:mo><mml:mfenced close="" open="["><mml:mrow><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">1</mml:mn><mml:mrow><mml:msup><mml:mi>m</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mstyle><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:msubsup><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mfenced></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">2</mml:mn><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:mstyle><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:msup><mml:mfenced close="]" open=""><mml:mrow><mml:mo>+</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">1</mml:mn><mml:mrow><mml:msup><mml:mi>n</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mstyle><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mfenced><mml:mstyle scriptlevel="+1"><mml:mfrac><mml:mn mathvariant="normal">1</mml:mn><mml:mn mathvariant="normal">2</mml:mn></mml:mfrac></mml:mstyle></mml:msup></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>

                  MMD has demonstrated superior performance in validating synthetic environmental data compared to parametric tests, particularly for small-sample geochemical datasets where underlying distribution assumptions are often violated (An et al., 2025).</p>
              </list-item>
            </list></p>
</sec>
</sec>
<sec id="Ch1.S2.SS5">
  <label>2.5</label><title>Hybrid quantum-classical random forest</title>
      <p id="d2e2487">In this study, Random Forest (RF) was adopted as the baseline model. RF builds numerous decision trees through bootstrap sampling and random feature selection and integrates their predictions (Abderzak et al., 2025), effectively suppressing overfitting and improving generalization for environmental data with small samples, high dimensionality, non-linearity, and multicollinearity. Hyperparameters were configured based on a preliminary grid search and domain expertise: n_estimators<inline-formula><mml:math id="M124" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula>100, max_depth<inline-formula><mml:math id="M125" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula>5, min_samples_split<inline-formula><mml:math id="M126" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula>6, min_samples_leaf<inline-formula><mml:math id="M127" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula>3. Feature importance was quantified by the mean decrease in Gini impurity to identify the critical hydrogeochemical drivers (Kaur et al., 2025). A hybrid quantum-enhanced RF model was then constructed: standardized input features were encoded by a Parameterized Quantum Circuit (PQC) into quantum features with non-linear entanglement properties (Naresh and Reddi, 2025), which were concatenated with the original features and fed into an RF regressor (Lamichhane and Rawat, 2025). This hybrid framework retains the ensemble-learning strengths of classical RF while improving the capture of complex non-linear relationships in small-sample datasets, without relying on physical quantum hardware.</p>
      <p id="d2e2518"><list list-type="order">
            <list-item>

      <p id="d2e2523"><italic>Quantum Feature Encoding.</italic></p>

      <p id="d2e2527">Classical data were mapped into a quantum Hilbert space via single-qubit <inline-formula><mml:math id="M128" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula>-gate and two-qubit CZ-gate entanglement operations, using the ZFeatureMap provided by Qiskit as the encoding circuit (Vedavyasa and Kumar, 2025). The number of qubits was set equal to the number of input features, and the repetition layers (reps) were set to 1 to avoid circuit-depth-induced noise while maintaining expressibility. Its Hamiltonian form is given by (Khalil et al., 2025):

                  <disp-formula id="Ch1.E5" content-type="numbered"><label>5</label><mml:math id="M129" display="block"><mml:mrow><mml:mtable class="split" rowspacing="0.2ex" displaystyle="true" columnalign="right left"><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>U</mml:mi><mml:mi mathvariant="normal">ZMap</mml:mi></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mo>∏</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>R</mml:mi></mml:msubsup><mml:mfenced open="[" close="]"><mml:mrow><mml:msubsup><mml:mo>⊗</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>d</mml:mi></mml:msubsup><mml:msub><mml:mi>H</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>⋅</mml:mo><mml:mi>exp⁡</mml:mi><mml:mfenced open="(" close=")"><mml:mrow><mml:mo>-</mml:mo><mml:mi>i</mml:mi><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mo>⊆</mml:mo><mml:mo mathvariant="italic">{</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mi>d</mml:mi><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:munder><mml:msub><mml:mi mathvariant="italic">ϕ</mml:mi><mml:mi>S</mml:mi></mml:msub><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo><mml:msub><mml:mo>⊗</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>∈</mml:mo><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>Z</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mfenced></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>

                where <inline-formula><mml:math id="M130" display="inline"><mml:mrow><mml:msub><mml:mi>U</mml:mi><mml:mi mathvariant="normal">ZMap</mml:mi></mml:msub><mml:mo>(</mml:mo><mml:mi>x</mml:mi></mml:mrow></mml:math></inline-formula>) is the unitary operator of the encoding circuit, <inline-formula><mml:math id="M131" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> is the normalized input feature vector, <inline-formula><mml:math id="M132" display="inline"><mml:mi>R</mml:mi></mml:math></inline-formula> is the number of repetition layers, <inline-formula><mml:math id="M133" display="inline"><mml:mi>d</mml:mi></mml:math></inline-formula> is the number of qubits, <inline-formula><mml:math id="M134" display="inline"><mml:mrow><mml:mi>H</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:math></inline-formula> the Hadamard gate on the <inline-formula><mml:math id="M135" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula>th qubit, <inline-formula><mml:math id="M136" display="inline"><mml:mrow><mml:mi>S</mml:mi><mml:mo>⊆</mml:mo><mml:mo mathvariant="italic">{</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mi>d</mml:mi><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula> is the subset of qubit indices involved in entanglement, <inline-formula><mml:math id="M137" display="inline"><mml:mrow><mml:mi mathvariant="italic">ϕ</mml:mi><mml:mi>S</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is the data-dependent rotation angle, and <inline-formula><mml:math id="M138" display="inline"><mml:mrow><mml:mi>Z</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:math></inline-formula> is the Pauli-<inline-formula><mml:math id="M139" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> operator on the <inline-formula><mml:math id="M140" display="inline"><mml:mi>j</mml:mi></mml:math></inline-formula>th qubit.</p>

      <p id="d2e2776">For each sample <inline-formula><mml:math id="M141" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula>, the quantum state  <inline-formula><mml:math id="M142" display="inline"><mml:mrow><mml:mo>|</mml:mo><mml:mi mathvariant="italic">ψ</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula> was constructed and the Pauli-<inline-formula><mml:math id="M143" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> expectation value of each qubit was calculated analytically (Liao et al., 2024):

                  <disp-formula id="Ch1.E6" content-type="numbered"><label>6</label><mml:math id="M144" display="block"><mml:mrow><mml:mo>〈</mml:mo><mml:msub><mml:mi>Z</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>〉</mml:mo><mml:mo>=</mml:mo><mml:mo>〈</mml:mo><mml:mi mathvariant="italic">ψ</mml:mi><mml:mo>(</mml:mo><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mo>)</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mi>Z</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>|</mml:mo><mml:mi mathvariant="italic">ψ</mml:mi><mml:mo>(</mml:mo><mml:mi mathvariant="bold-italic">x</mml:mi><mml:mo>)</mml:mo><mml:mo>〉</mml:mo><mml:mo>=</mml:mo><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn mathvariant="normal">0</mml:mn><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>

                where <inline-formula><mml:math id="M145" display="inline"><mml:mrow><mml:mo>〈</mml:mo><mml:mi>Z</mml:mi><mml:mi>i</mml:mi><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula> is the expectation value of the Pauli-<inline-formula><mml:math id="M146" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> operator for the <inline-formula><mml:math id="M147" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula>th qubit, a dimensionless scalar in [<inline-formula><mml:math id="M148" display="inline"><mml:mo lspace="0mm">-</mml:mo></mml:math></inline-formula>1, <inline-formula><mml:math id="M149" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>1]; <inline-formula><mml:math id="M150" display="inline"><mml:mrow><mml:mo>|</mml:mo><mml:mi mathvariant="italic">ψ</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula> is the quantum state encoding <inline-formula><mml:math id="M151" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> after applying <inline-formula><mml:math id="M152" display="inline"><mml:mrow><mml:msub><mml:mi>U</mml:mi><mml:mi mathvariant="normal">ZMap</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>; and <inline-formula><mml:math id="M153" display="inline"><mml:mrow><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn mathvariant="normal">0</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M154" display="inline"><mml:mrow><mml:mi>P</mml:mi><mml:mo>(</mml:mo><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> are the probabilities of measuring the <inline-formula><mml:math id="M155" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula>th qubit in states <inline-formula><mml:math id="M156" display="inline"><mml:mrow><mml:mo>|</mml:mo><mml:mn mathvariant="normal">0</mml:mn><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M157" display="inline"><mml:mrow><mml:mo>|</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula>, with <inline-formula><mml:math id="M158" display="inline"><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>∈</mml:mo><mml:mo mathvariant="italic">{</mml:mo><mml:mn mathvariant="normal">0</mml:mn><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula> the binary measurement outcome.</p>
            </list-item>
            <list-item>

      <p id="d2e3073"><italic>Feature Fusion and Modeling.</italic></p>

      <p id="d2e3077">The original <inline-formula><mml:math id="M159" display="inline"><mml:mi>n</mml:mi></mml:math></inline-formula>-dimensional raw features are concatenated with the <inline-formula><mml:math id="M160" display="inline"><mml:mi>n</mml:mi></mml:math></inline-formula>-dimensional quantum <inline-formula><mml:math id="M161" display="inline"><mml:mrow><mml:mo>〈</mml:mo><mml:mi>Z</mml:mi><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula> features to form a 2<inline-formula><mml:math id="M162" display="inline"><mml:mi>n</mml:mi></mml:math></inline-formula>-dimensional hybrid feature vector x_aug <inline-formula><mml:math id="M163" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> [x_raw; <inline-formula><mml:math id="M164" display="inline"><mml:mrow><mml:mo>〈</mml:mo><mml:mi>Z</mml:mi><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula>] (Cowlessur et al., 2025). This fusion retains the original physical information of the hydrochemical parameters while introducing non-linear entangled quantum features, expanding the feature space to enhance the model's ability to capture complex non-linear relationships. The augmented vector is fed into a random forest regressor with hyperparameters identical to the classical RF baseline to ensure a fair comparison:

                  <disp-formula id="Ch1.E7" content-type="numbered"><label>7</label><mml:math id="M165" display="block"><mml:mrow><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo mathvariant="normal" stretchy="false">^</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">1</mml:mn><mml:mi>M</mml:mi></mml:mfrac></mml:mstyle><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:msub><mml:mi mathvariant="normal">Tree</mml:mi><mml:mi>m</mml:mi></mml:msub><mml:mo>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi mathvariant="normal">aug</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>

                where <inline-formula><mml:math id="M166" display="inline"><mml:mover accent="true"><mml:mi>y</mml:mi><mml:mo stretchy="false" mathvariant="normal">^</mml:mo></mml:mover></mml:math></inline-formula> is the predicted nitrate concentration, <inline-formula><mml:math id="M167" display="inline"><mml:mi>M</mml:mi></mml:math></inline-formula> is the number of decision trees, Tree<sub><italic>m</italic></sub> is the <inline-formula><mml:math id="M169" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula>th tree regressor, and <inline-formula><mml:math id="M170" display="inline"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi mathvariant="normal">aug</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is the augmented hybrid feature vector defined above. The hyperparameter settings are the same as those for the classical random forest method described above.</p>
            </list-item>
          </list></p>
</sec>
<sec id="Ch1.S2.SS6">
  <label>2.6</label><title>Evaluation methods and prediction process</title>
<sec id="Ch1.S2.SS6.SSS1">
  <label>2.6.1</label><title>SHAP analysis</title>
      <p id="d2e3240">The SHapley Additive exPlanations (SHAP) method was used for local and global explainability analysis (Merabet et al., 2025). Three visualizations were employed: the summary plot, revealing the overall ranking and distribution of feature importance across all samples (global perspective); the dependence plot, revealing nonlinear relationships and interaction effects between individual predictors and predicted nitrate concentration (conditional dependence); and the waterfall plot, revealing the contribution decomposition of each feature for a representative sample (local attribution) (Alam et al., 2025b; Li et al., 2024; Hollmann et al., 2025).</p>
</sec>
<sec id="Ch1.S2.SS6.SSS2">
  <label>2.6.2</label><title>Leave-One-Out Cross-Validation (LOOCV) and model evaluation indicators</title>
      <p id="d2e3251">Given the limited sample size in each hydrological season, model performance was evaluated using Leave-One-Out Cross-Validation (LOOCV) (Ren et al., 2021). Prior to LOOCV, hyperparameters were tuned via 5-fold cross-validation grid search (n_estimators [100, 150], max_depth [5, 6, 7], min_samples_split [4, 6]) and then fixed for the LOOCV evaluation. To assess prediction uncertainty, 90 % prediction intervals were estimated using quantile regression forests, with coverage probability calculated to validate their reliability; in addition, bootstrap resampling (1000 iterations) was used to derive 95 % confidence intervals for all evaluation metrics, which are reported alongside the LOOCV metrics. Model accuracy was quantified by <inline-formula><mml:math id="M171" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, RMSE, and MAE (Gul et al., 2025).</p>
</sec>
<sec id="Ch1.S2.SS6.SSS3">
  <label>2.6.3</label><title>Model robustness and uncertainty analysis</title>
      <p id="d2e3273">To evaluate the response of predicted nitrate concentrations to input perturbations and verify the robustness of the feature importance ranking, a multi-dimensional sensitivity analysis was conducted, including a SHAP-based global sensitivity index, one-at-a-time (OAT) perturbation tests, and elasticity coefficient calculations (Di Santo et al., 2025). First, from the SHAP values obtained via TreeExplainer, the first-order and total-order sensitivity indices were calculated to measure the independent and interactive contributions of each feature (Huang et al., 2021). The first-order sensitivity index for feature <inline-formula><mml:math id="M172" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula> is defined as the normalized mean absolute SHAP value:

              <disp-formula id="Ch1.E8" content-type="numbered"><label>8</label><mml:math id="M173" display="block"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mn mathvariant="normal">1</mml:mn><mml:mi>n</mml:mi></mml:mfrac></mml:mstyle><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi mathvariant="italic">ϕ</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>p</mml:mi></mml:msubsup><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mn mathvariant="normal">1</mml:mn><mml:mi>n</mml:mi></mml:mfrac></mml:mstyle><mml:msubsup><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mo>|</mml:mo><mml:msubsup><mml:mi mathvariant="italic">ϕ</mml:mi><mml:mi>k</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula>

            where <inline-formula><mml:math id="M174" display="inline"><mml:mrow><mml:msubsup><mml:mi mathvariant="italic">ϕ</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is the SHAP value of feature <inline-formula><mml:math id="M175" display="inline"><mml:mi>i</mml:mi></mml:math></inline-formula> for sample <inline-formula><mml:math id="M176" display="inline"><mml:mi>j</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="M177" display="inline"><mml:mi>n</mml:mi></mml:math></inline-formula> is the number of samples, and <inline-formula><mml:math id="M178" display="inline"><mml:mi>p</mml:mi></mml:math></inline-formula> is the number of features. The total-order sensitivity index <inline-formula><mml:math id="M179" display="inline"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi mathvariant="normal">Ti</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, capturing both main and interaction effects, was approximated using the diagonal elements of the SHAP interaction values:</p>
      <p id="d2e3438">Second, an OAT perturbation test was conducted to assess local model stability, by varying each target feature within <inline-formula><mml:math id="M180" display="inline"><mml:mo>±</mml:mo></mml:math></inline-formula>30 % of its measured range while fixing the other inputs; perturbation sensitivity was quantified as the relative change rate of the model's <inline-formula><mml:math id="M181" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> across perturbation levels (Anderson and Lucas, 2018). Finally, the elasticity coefficient, defined as the ratio of the percentage change in the predicted output to the percentage change in the input, was used to quantify the directional response of the predictions to per-unit changes in each parameter. An absolute value greater (less) than 1 indicates an elastic (inelastic) response. Monte Carlo simulation was specifically used to analyze the propagation of input measurement errors to the final predictions (Dega et al., 2023).</p>
</sec>
<sec id="Ch1.S2.SS6.SSS4">
  <label>2.6.4</label><title>Standardized prediction workflow</title>
      <p id="d2e3467">To systematically evaluate how data inputs, virtual sample generation, and modeling strategy jointly affect seasonal nitrate prediction, we established a standardized, fully reproducible pipeline (Fig. 3), in which the output of each step serves as the direct input of the next: (1) Data inputs and preprocessing. Two complementary input feature sets were prepared from the seasonally partitioned observations (66, 65, and 50 samples for the normal, dry, and wet seasons, respectively): (i) in-situ field water-quality parameters and (ii) AlphaEarth Foundation (AEF) embeddings compressed by PCA. Both sets were <inline-formula><mml:math id="M182" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula>-score standardized independently per season; the raw observations were used without interpolation or smoothing to preserve intrinsic variability. These standardized features, together with the measured NO<inline-formula><mml:math id="M183" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations, constitute the sole data basis for all subsequent steps. (2) Virtual sample generation and validation. Each standardized seasonal dataset was fed into the t-SNE–GMM–KNN generator to produce virtual samples at generation factors of 1–10<inline-formula><mml:math id="M184" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>. Only virtual datasets whose distribution consistency and physical plausibility passed the validation tests were allowed to enter model training, thereby explicitly linking generation quality to downstream evaluation. (3) Model training. Classical Random Forest (RF) and quantum-enhanced RF were trained under unified hyperparameters on two tracks of training sets built from the validated samples: the original samples alone, and the original samples combined with 1–10<inline-formula><mml:math id="M185" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> virtual samples. For the quantum-enhanced RF, input features were encoded via parameterized quantum circuits to generate <inline-formula><mml:math id="M186" display="inline"><mml:mrow><mml:mo>〈</mml:mo><mml:mi>Z</mml:mi><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula> quantum features, which were concatenated with the original features (<inline-formula><mml:math id="M187" display="inline"><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mi>n</mml:mi></mml:mrow></mml:math></inline-formula> inputs in total). (4) Model evaluation. Every trained model was evaluated using leave-one-out cross-validation on the measured samples. (5) Interpretability analysis. SHAP-based multi-scale interpretation (summary, dependence, and waterfall plots) was conducted and cross-verified against Bayesian modeling and Pearson correlation analysis.</p>

      <fig id="F3" specific-use="star"><label>Figure 3</label><caption><p id="d2e3528">Process diagram for constructing prediction framework.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f03.png"/>

          </fig>

      <p id="d2e3537">All computations were performed on an Apple M2 Pro (10-core CPU, 16 GB RAM) using Python 3.9, with scikit-learn 1.7.2, qiskit 1.4.5, shap 0.49.1, pandas 2.3.3, numpy 2.2.6, scipy 1.15.3, joblib 1.5.2, and matplotlib 3.9.0. All random operations were seeded (random_state <inline-formula><mml:math id="M188" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 42) to ensure deterministic results. The complete source code, covering quantum <inline-formula><mml:math id="M189" display="inline"><mml:mrow><mml:mo>〈</mml:mo><mml:mi>Z</mml:mi><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula> feature extraction, model training, LOOCV evaluation, bootstrap uncertainty quantification, and sensitivity analysis, is publicly available at GitHub.</p>
</sec>
</sec>
</sec>
<sec id="Ch1.S3">
  <label>3</label><title>Results</title>
<sec id="Ch1.S3.SS1">
  <label>3.1</label><title>Seasonal hydrochemical controls of nitrate distribution in farmland groundwater</title>
<sec id="Ch1.S3.SS1.SSS1">
  <label>3.1.1</label><title>Hydrochemical characteristics and water types</title>
      <p id="d2e3583">Groundwater pH was weakly alkaline in the normal water period with minimal variation and near-neutral in the dry and wet seasons, with a minimum of 5.77 in the wet season indicating occasional acidic water (Table S1 in the Supplement). Temperature varied markedly among seasons but remained stable within each (CV <inline-formula><mml:math id="M190" display="inline"><mml:mo>≈</mml:mo></mml:math></inline-formula> 0.06). EC, salinity, and TDS followed consistent seasonal patterns, peaking in the dry season and reaching minima in the wet season. Redox indicators were highly variable: DO was slightly higher in the wet season, whereas ORP remained low across all seasons with large standard deviations. Among major ions, Ca<sup>2+</sup>, Mg<sup>2+</sup>, Na<sup>+</sup>, Cl<sup>−</sup>, and SO<inline-formula><mml:math id="M195" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> all peaked in the dry season, while HCO<inline-formula><mml:math id="M196" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> was highest in the wet season and K<sup>+</sup> remained low. NO<inline-formula><mml:math id="M198" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations were highest in the dry season and lowest in the wet season, far exceeding those of NO<inline-formula><mml:math id="M199" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> and NH<inline-formula><mml:math id="M200" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mo>+</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>. Most variables were right-skewed, with notable extreme values for NO<inline-formula><mml:math id="M201" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> in the dry season (maximum <inline-formula><mml:math id="M202" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 358.58, mean <inline-formula><mml:math id="M203" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 42.93 mg L<sup>−1</sup>), Cl<sup>−</sup> in the dry season (maximum <inline-formula><mml:math id="M206" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 241.36, mean <inline-formula><mml:math id="M207" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 24.90 mg L<sup>−1</sup>), and F<sup>−</sup> in the normal period (maximum <inline-formula><mml:math id="M210" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 13.17, mean <inline-formula><mml:math id="M211" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 3.70 mg L<sup>−1</sup>). The Piper diagram (Fig. 4) shows that dry-season samples cluster tightly in the Ca-HCO<inline-formula><mml:math id="M213" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> facies, indicating dominant control by carbonate dissolution. In the wet season, the Ca-Mg-HCO<inline-formula><mml:math id="M214" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> type remains dominant but some samples shift toward sulfate and chloride types, reflecting leaching inputs of surface pollutants via rainfall infiltration. In the normal season, the hydrochemical types are most dispersed, showing mixed HCO<inline-formula><mml:math id="M215" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>-Cl facies. Overall, the groundwater hydrochemistry in the study area is jointly controlled by precipitation-evaporation dynamics and carbonate weathering.</p>

      <fig id="F4" specific-use="star"><label>Figure 4</label><caption><p id="d2e3857">Piper diagram classifying the hydrochemical facies of the analyzed groundwater.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f04.png"/>

          </fig>

</sec>
<sec id="Ch1.S3.SS1.SSS2">
  <label>3.1.2</label><title>Sources and controlling factors of ions in groundwater</title>
      <p id="d2e3874">The Gibbs diagram shows that the groundwater in the study area is primarily controlled by rock weathering during the normal, dry, and wet seasons, indicating the dominance of water-rock interaction (Fig. 5). The ratio of <inline-formula><mml:math id="M216" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(Na<inline-formula><mml:math id="M217" display="inline"><mml:mrow><mml:msup><mml:mi/><mml:mo>+</mml:mo></mml:msup><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula> K<sup>+</sup>) to <inline-formula><mml:math id="M219" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>Cl<sup>−</sup> (Fig. 6a) shows that the vast majority of sample points plot above the <inline-formula><mml:math id="M221" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line, indicating that Na<sup>+</sup> and K<sup>+</sup> are primarily sourced from the dissolution of evaporite rocks. In the relationships between <inline-formula><mml:math id="M224" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(Ca<inline-formula><mml:math id="M225" display="inline"><mml:mrow><mml:msup><mml:mi/><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>+</mml:mo></mml:mrow></mml:msup><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula> Mg<sup>2+</sup>) and <inline-formula><mml:math id="M227" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>HCO<inline-formula><mml:math id="M228" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, and between <inline-formula><mml:math id="M229" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(Ca<inline-formula><mml:math id="M230" display="inline"><mml:mrow><mml:msup><mml:mi/><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>+</mml:mo></mml:mrow></mml:msup><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula> Mg<sup>2+</sup>) and <inline-formula><mml:math id="M232" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(HCO<inline-formula><mml:math id="M233" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula> SO<inline-formula><mml:math id="M234" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>) (Fig. 6b–c), samples from all periods plot above the <inline-formula><mml:math id="M235" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line, confirming that Ca<sup>2+</sup> and Mg<sup>2+</sup> mainly originate from the dissolution of carbonate minerals. Furthermore, the <inline-formula><mml:math id="M238" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>Ca<sup>2+</sup>-<inline-formula><mml:math id="M240" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>Mg<sup>2+</sup> relationship (Fig. 6d) helps identify the types of mineral dissolution. Samples from the dry season are concentrated below the <inline-formula><mml:math id="M242" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">2</mml:mn></mml:mrow></mml:math></inline-formula> line, indicating a dominance of magnesium-poor mineral dissolution, with cation exchange causing a relative depletion of Ca<sup>2+</sup>. Samples from the normal and wet seasons are stably distributed between the <inline-formula><mml:math id="M244" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M245" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">2</mml:mn></mml:mrow></mml:math></inline-formula> lines, reflecting that dolomite dissolution has reached equilibrium while calcite remains in a state of non-equilibrium dissolution, continuously supplying Ca<sup>2+</sup>. In the relationship between <inline-formula><mml:math id="M247" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(SO<inline-formula><mml:math id="M248" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula> Cl<sup>−</sup>) and <inline-formula><mml:math id="M250" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>HCO<inline-formula><mml:math id="M251" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> (Fig. 6e), the distribution of sample points on both sides of the <inline-formula><mml:math id="M252" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line suggests that groundwater ions have dual contributions from both evaporite and carbonate rocks. Conversely, in the <inline-formula><mml:math id="M253" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>Ca<sup>2+</sup> versus <inline-formula><mml:math id="M255" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>SO<inline-formula><mml:math id="M256" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> relationship (Fig. 6f), samples generally plot above the <inline-formula><mml:math id="M257" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>:</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line, which excludes gypsum as a primary source of Ca<sup>2+</sup> and indicates that Ca<sup>2+</sup> is mainly derived from the dissolution of carbonate minerals. Therefore, the chemical composition of groundwater in the study area is primarily controlled by the dissolution of carbonate minerals, and is also influenced by hydrological seasonal variations and cation exchange processes.</p>

      <fig id="F5" specific-use="star"><label>Figure 5</label><caption><p id="d2e4358">Gibbs diagrams of the groundwater samples in different hydrological seasons.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f05.png"/>

          </fig>

      <fig id="F6" specific-use="star"><label>Figure 6</label><caption><p id="d2e4369">Plots of ion ratio relationship.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f06.png"/>

          </fig>

      <p id="d2e4379">The Chloro-Alkaline Index (CAI) was used to assess cation exchange and adsorption between groundwater and sediments, where negative values indicate exchange and more negative values reflect stronger intensity. This was further examined through the relationship between [<inline-formula><mml:math id="M260" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(Ca<inline-formula><mml:math id="M261" display="inline"><mml:mrow><mml:msup><mml:mi/><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>+</mml:mo></mml:mrow></mml:msup><mml:mo>)</mml:mo><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula> <inline-formula><mml:math id="M262" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(Mg<inline-formula><mml:math id="M263" display="inline"><mml:mrow><mml:msup><mml:mi/><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>+</mml:mo></mml:mrow></mml:msup><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi mathvariant="italic">γ</mml:mi></mml:mrow></mml:math></inline-formula>(HCO<inline-formula><mml:math id="M264" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi mathvariant="italic">γ</mml:mi></mml:mrow></mml:math></inline-formula>(SO<inline-formula><mml:math id="M265" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>)] and [<inline-formula><mml:math id="M266" display="inline"><mml:mi mathvariant="italic">γ</mml:mi></mml:math></inline-formula>(Na<inline-formula><mml:math id="M267" display="inline"><mml:mrow><mml:msup><mml:mi/><mml:mo>+</mml:mo></mml:msup><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi mathvariant="italic">γ</mml:mi></mml:mrow></mml:math></inline-formula>(Cl<sup>−</sup>)] (Fig. 7). Samples from the normal and wet seasons plotted near the slope <inline-formula><mml:math id="M269" display="inline"><mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula> line (slopes of <inline-formula><mml:math id="M270" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>1.52 and <inline-formula><mml:math id="M271" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>1.36, respectively), confirming active cation exchange consistent with the CAI results; exchange was most pronounced in the rainy season (<inline-formula><mml:math id="M272" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.41), leading to Na<sup>+</sup> enrichment. In contrast, the dry-season slope of 0.55 indicates negligible cation exchange during this period.</p>

      <fig id="F7" specific-use="star"><label>Figure 7</label><caption><p id="d2e4549">Relationship diagram of groundwater (Ca<inline-formula><mml:math id="M274" display="inline"><mml:mrow><mml:msup><mml:mi/><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>+</mml:mo></mml:mrow></mml:msup><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula>Mg<sup>2+</sup>-SO<inline-formula><mml:math id="M276" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>-HCO<inline-formula><mml:math id="M277" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>) and (Na<sup>+</sup>-Cl<sup>−</sup>) along with CAI-1 and CAI-2 correlation diagrams.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f07.png"/>

          </fig>

</sec>
<sec id="Ch1.S3.SS1.SSS3">
  <label>3.1.3</label><title>Spatial distribution dynamics of groundwater depth and nitrate driven by seasonal hydrological processes</title>
      <p id="d2e4639">The spatial distribution of groundwater depth reflects the regional hydraulic gradient and groundwater flow direction, whereas the spatial variability of nitrate concentration is closely associated with flow paths, pollution source inputs, and hydrological processes (Fig. 8). During the normal season, the groundwater depth distribution is relatively uniform. The eastern region of the farm, characterized by shallower depths, serves as a recharge zone, with groundwater flowing towards the deeper western region. At this time, nitrate concentration are relatively dispersed, with high-concentration zones located in the southeastern part of the farm. In the dry season, the groundwater depth becomes deeper, and the flow direction shifts from the eastern and western sides towards the central area. During this period, nitrate concentration reach their annual peak (mean: 42.93 mg L<sup>−1</sup>). The distribution of nitrate exhibits a higher degree of spatial coincidence with the groundwater flow direction, indicating that enhanced evaporative concentration during the dry season leads to the further accumulation of flow-transported pollutants in the discharge zone. In the wet season, the groundwater depth further becomes shallower, and groundwater flows from the southeastern region towards the northwestern region. nitrate concentration drop to their annual minimum (mean: 27.14 mg L<sup>−1</sup>). High-concentration areas are distributed in the northwest, overlapping with regions of deeper groundwater depth. It is inferred that precipitation infiltration during the rainy season dilutes the groundwater nitrate; as dilution is the dominant process during infiltration, the nitrate concentration exhibits a decreasing trend along the groundwater flow direction.</p>

      <fig id="F8" specific-use="star"><label>Figure 8</label><caption><p id="d2e4668">Spatial distribution of nitrate concentration and groundwater depth in different seasons. From top to bottom: normal water period, dry season, and wet season.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f08.png"/>

          </fig>


</sec>
</sec>
<sec id="Ch1.S3.SS2">
  <label>3.2</label><title>Groundwater recharge sources and pollution source identification</title>
<sec id="Ch1.S3.SS2.SSS1">
  <label>3.2.1</label><title>Stable hydrogen and oxygen isotope composition of water</title>
      <p id="d2e4695">During the normal season, the mean values of groundwater <inline-formula><mml:math id="M282" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D and <inline-formula><mml:math id="M283" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O were <inline-formula><mml:math id="M284" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>61.31 ‰ and <inline-formula><mml:math id="M285" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>7.31 ‰, with ranges of <inline-formula><mml:math id="M286" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>73.40 ‰ to <inline-formula><mml:math id="M287" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>53.52 ‰ and <inline-formula><mml:math id="M288" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>10.25 ‰ to <inline-formula><mml:math id="M289" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>2.82 ‰, respectively. The <inline-formula><mml:math id="M290" display="inline"><mml:mi>d</mml:mi></mml:math></inline-formula>-excess values ranged from <inline-formula><mml:math id="M291" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>38.07 ‰ to 17.30 ‰, with a mean of <inline-formula><mml:math id="M292" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>2.80 ‰. In the dry season, the mean groundwater <inline-formula><mml:math id="M293" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D and <inline-formula><mml:math id="M294" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O values were <inline-formula><mml:math id="M295" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>71.05 ‰ and <inline-formula><mml:math id="M296" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>9.74 ‰, with ranges of <inline-formula><mml:math id="M297" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>76.93 ‰ to <inline-formula><mml:math id="M298" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>60.55 ‰ and <inline-formula><mml:math id="M299" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>10.75 ‰ to <inline-formula><mml:math id="M300" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>7.86 ‰, respectively. The <inline-formula><mml:math id="M301" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">17</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O values ranged from <inline-formula><mml:math id="M302" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>5.60 ‰ to <inline-formula><mml:math id="M303" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>2.79 ‰, averaging <inline-formula><mml:math id="M304" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>5.01 ‰, while the <inline-formula><mml:math id="M305" display="inline"><mml:mi>d</mml:mi></mml:math></inline-formula>-excess varied from 0.01 ‰ to 13.69 ‰, with a mean of 6.89 ‰. During the wet season, the mean groundwater <inline-formula><mml:math id="M306" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D and <inline-formula><mml:math id="M307" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O were <inline-formula><mml:math id="M308" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>74.43 ‰ and <inline-formula><mml:math id="M309" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>9.99 ‰, with ranges of <inline-formula><mml:math id="M310" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>76.84 ‰ to <inline-formula><mml:math id="M311" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>70.91 ‰ and <inline-formula><mml:math id="M312" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>10.70 ‰ to <inline-formula><mml:math id="M313" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>8.73 ‰, respectively. The <inline-formula><mml:math id="M314" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">17</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O values were between <inline-formula><mml:math id="M315" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>5.72 ‰ and <inline-formula><mml:math id="M316" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>4.72 ‰, with a mean of <inline-formula><mml:math id="M317" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>5.28 ‰, and the <inline-formula><mml:math id="M318" display="inline"><mml:mi>d</mml:mi></mml:math></inline-formula>-excess ranged from <inline-formula><mml:math id="M319" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>3.67 ‰ to 9.80 ‰, averaging 5.50 ‰. The <inline-formula><mml:math id="M320" display="inline"><mml:mi>d</mml:mi></mml:math></inline-formula>-excess during the dry season was the highest among the three periods, while it was the lowest during the normal period, indicating significant variations in <inline-formula><mml:math id="M321" display="inline"><mml:mi>d</mml:mi></mml:math></inline-formula>-excess across different seasons. The <inline-formula><mml:math id="M322" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">17</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O values in the dry season were higher than those in the wet and normal periods, which is a direct reflection of the impact of precipitation variations on the isotopic composition of the water body.</p>
      <p id="d2e5015">The isotopic values of precipitation <inline-formula><mml:math id="M323" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D and <inline-formula><mml:math id="M324" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O ranged from <inline-formula><mml:math id="M325" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>97.78 ‰ to <inline-formula><mml:math id="M326" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>20.22 ‰ and from <inline-formula><mml:math id="M327" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>13.48 ‰ to <inline-formula><mml:math id="M328" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>1.96 ‰, with mean values of <inline-formula><mml:math id="M329" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>55.36 ‰ and <inline-formula><mml:math id="M330" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>7.60 ‰, respectively. Overall, the <inline-formula><mml:math id="M331" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O and <inline-formula><mml:math id="M332" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D values of precipitation in the study area fall within the global ranges of <inline-formula><mml:math id="M333" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>50 ‰ to 10 ‰ and <inline-formula><mml:math id="M334" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>350 ‰ to 50 ‰. The Local Meteoric Water Line (LMWL) for the study area is defined by the equation: <inline-formula><mml:math id="M335" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D <inline-formula><mml:math id="M336" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 6.2 <inline-formula><mml:math id="M337" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-8.2 (Fig. 9). Specifically, the equations for the normal, dry, and wet seasons are: <inline-formula><mml:math id="M338" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D <inline-formula><mml:math id="M339" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 1.07 <inline-formula><mml:math id="M340" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-53.50, <inline-formula><mml:math id="M341" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D <inline-formula><mml:math id="M342" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 4.20 <inline-formula><mml:math id="M343" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-30.09, and <inline-formula><mml:math id="M344" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D <inline-formula><mml:math id="M345" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 1.95 <inline-formula><mml:math id="M346" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-54.97, respectively. The slope of the annual LMWL is lower than that of the Global Meteoric Water Line (GMWL) proposed by Craig in 1964 (<inline-formula><mml:math id="M347" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D <inline-formula><mml:math id="M348" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 8 <inline-formula><mml:math id="M349" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O<inline-formula><mml:math id="M350" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>10), as well as lower than the China Meteoric Water Line (CMWL) (<inline-formula><mml:math id="M351" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D <inline-formula><mml:math id="M352" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 7.9 <inline-formula><mml:math id="M353" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O<inline-formula><mml:math id="M354" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>8.2). The stable hydrogen and oxygen isotopic characteristics of groundwater samples from the three periods all exhibit a discrete, linear distribution and plot below both GMWL and LMWL. This phenomenon reveals that the water isotopes have undergone strong fractionation during evaporation in the normal, wet, and dry seasons. Furthermore, the stable hydrogen and oxygen isotope data for the dry and normal seasons are mainly concentrated in the lower-left region of the plot, indicating relative isotopic depletion during these two periods. During the normal period, the <inline-formula><mml:math id="M355" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D and <inline-formula><mml:math id="M356" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O values exhibit a high degree of dispersion and are widely distributed in the upper-central part of the scatter plot. This reflects that the stable hydrogen and oxygen isotopes are relatively enriched and have a wide range of variation during the normal season.</p>

      <fig id="F9"><label>Figure 9</label><caption><p id="d2e5299"><inline-formula><mml:math id="M357" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O <inline-formula><mml:math id="M358" display="inline"><mml:mo>/</mml:mo></mml:math></inline-formula> <inline-formula><mml:math id="M359" display="inline"><mml:mi mathvariant="italic">δ</mml:mi></mml:math></inline-formula>D relationship of groundwater samples in different hydrological seasons.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f09.png"/>

          </fig>

</sec>
<sec id="Ch1.S3.SS2.SSS2">
  <label>3.2.2</label><title>Identification of nitrate sources using isotopes and MixSIAR model</title>
      <p id="d2e5340">During the normal water period, the nitrogen and oxygen isotopic compositions in groundwater exhibit a wide range of variation. The <inline-formula><mml:math id="M360" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N-NO<inline-formula><mml:math id="M361" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> values range from 5.6 ‰ to 24.52 ‰ (average: 18.22 ‰), while the <inline-formula><mml:math id="M362" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-NO<inline-formula><mml:math id="M363" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> values range from <inline-formula><mml:math id="M364" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>6.33 ‰ to 6.23 ‰ (average: 0.22 ‰). In the low water period, the range of <inline-formula><mml:math id="M365" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N-NO<inline-formula><mml:math id="M366" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> values expands to 3.2 ‰–21.96 ‰ (average: 12.19 ‰), and the <inline-formula><mml:math id="M367" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-NO<inline-formula><mml:math id="M368" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> values range from <inline-formula><mml:math id="M369" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>9.58 ‰ to 8.04 ‰ (average: 0.65 ‰). Previous studies have established characteristic <inline-formula><mml:math id="M370" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-NO<inline-formula><mml:math id="M371" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> ranges for different nitrate sources: atmospheric deposition (23 ‰–75 ‰), nitrate fertilizers (18 ‰–24 ‰), and products of nitrification (<inline-formula><mml:math id="M372" display="inline"><mml:mo lspace="0mm">-</mml:mo></mml:math></inline-formula>10 ‰–10 ‰). The data points are predominantly concentrated within the zone of animal manure and domestic wastewater, indicating that nitrate is primarily derived from these sources, with soil nitrogen as a secondary contributor.</p>
      <p id="d2e5481">The MixSIAR model was employed to quantitatively apportion the sources of groundwater nitrate nitrogen. According to the average contributions from each source, the five pollution sources in the study area were ranked as follows: DSM (74.1 %) <inline-formula><mml:math id="M373" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> SON (20.9 %) <inline-formula><mml:math id="M374" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> NHF (4.2 %) <inline-formula><mml:math id="M375" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> NOF (0.6 %) <inline-formula><mml:math id="M376" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> NP (0.2 %) (Fig. 10). This indicates that the primary contributor to groundwater NO<inline-formula><mml:math id="M377" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>-N in the study area was manure and sewage, followed by soil nitrogen. The influences of atmospheric precipitation and chemical fertilizers on groundwater nitrate were negligible. The quantitative results from the MixSIAR analysis are consistent with the qualitative findings, confirming that manure and sewage, along with soil nitrogen, are the dominant sources of nitrate pollution in the study area.</p>

      <fig id="F10" specific-use="star"><label>Figure 10</label><caption><p id="d2e5526"><bold>(a)</bold> distributions of <inline-formula><mml:math id="M378" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N-NO<inline-formula><mml:math id="M379" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M380" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O-NO<inline-formula><mml:math id="M381" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> values in the study area. <bold>(b)</bold> proportional contributions of the main NO<inline-formula><mml:math id="M382" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> sources evaluated by the MixSIAR model. Note: boxplots denote the 25th, 50th and 75th percentiles.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f10.png"/>

          </fig>

</sec>
</sec>
<sec id="Ch1.S3.SS3">
  <label>3.3</label><title>Bayesian model analysis and correlation analysis</title>
      <p id="d2e5608">In the normal season, the Bayesian model identified Mg<sup>2+</sup> as the central driver of NO<inline-formula><mml:math id="M384" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, consistent with its strong correlation (Fig. 11). Na<sup>+</sup> showed a significant negative effect despite only a weak correlation (<inline-formula><mml:math id="M386" display="inline"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.39), suggesting its variation reflects hydrological processes rather than direct involvement in NO<inline-formula><mml:math id="M387" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> transformation. Although TDS and EC correlated strongly with NO<inline-formula><mml:math id="M388" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> (<inline-formula><mml:math id="M389" display="inline"><mml:mi>r</mml:mi></mml:math></inline-formula> <inline-formula><mml:math id="M390" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.8), their probabilities of direction (pd <inline-formula><mml:math id="M391" display="inline"><mml:mi mathvariant="italic">&lt;</mml:mi></mml:math></inline-formula> 80 %) indicate indirect effects through collinearity with other ions. In the dry season, SO<inline-formula><mml:math id="M392" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> emerged as the primary positive driver, agreeing with its high correlation with NO<inline-formula><mml:math id="M393" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> (<inline-formula><mml:math id="M394" display="inline"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.96), whereas Na<sup>+</sup> and Ca<sup>2+</sup> exerted significant negative effects despite weak positive correlations, likely because their enrichment reflects evaporation while NO<inline-formula><mml:math id="M397" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> derives from anthropogenic inputs, with no direct causal link. In the wet season, Cl<sup>−</sup> and Mg<sup>2+</sup> were the dominant factors with clear directional effects, matching their correlations with NO<inline-formula><mml:math id="M400" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> (<inline-formula><mml:math id="M401" display="inline"><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn mathvariant="normal">0.78</mml:mn></mml:mrow></mml:math></inline-formula> and 0.84, respectively) and confirming their direct influence. TDS and EC again showed wide posterior distributions and low pd values, indicating effects attributable to collinearity with Mg<sup>2+</sup> and Cl<sup>−</sup> rather than independent contributions.</p>

      <fig id="F11" specific-use="star"><label>Figure 11</label><caption><p id="d2e5842">Factor effects and Pearson coefficients of physicochemical variables on NO<inline-formula><mml:math id="M404" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> at different. periods. The left part of each subgraph shows the relative importance and posterior distribution of each environmental variable to NO<inline-formula><mml:math id="M405" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> after the Bayesian model operation. The red area represents the probability density of the positive effect, and the blue area represents the probability density of the negative effect. The percentage values beside the distribution represent the Probability of Direction (pd). The right part of each subgraph is the heat map of the correlation analysis.</p></caption>
          <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f11.png"/>

        </fig>

</sec>
<sec id="Ch1.S3.SS4">
  <label>3.4</label><title>Model performance evaluation</title>
<sec id="Ch1.S3.SS4.SSS1">
  <label>3.4.1</label><title>Virtual data analysis</title>
      <p id="d2e5890">To address the modeling bias arising from limited measured samples, this study constructed virtual datasets at 1–10 times the original scale based on a strategy combining t-SNE dimensionality reduction, GMM clustering sampling, and KNN inverse mapping, to enhance the robustness of model training. Taking the 10<inline-formula><mml:math id="M406" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> virtual dataset as an example, the statistical characteristics (Table 3) show that the virtual samples effectively reproduced the central tendency and dispersion of the original data. For the normal season, the mean NO<inline-formula><mml:math id="M407" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentration was 30.41 mg L<sup>−1</sup> (vs. observed mean of 33.67), with a standard deviation of 28.78 (vs. 35.83) and a coefficient of variation (CV) of 0.95 (vs. 1.06). In the dry season, the maximum value of the virtual samples reached 178.09 mg L<sup>−1</sup>, while this did not fully replicate the extreme high values (observed maximum of 358.58 mg L<sup>−1</sup>), it effectively expanded the range of the heavy-tailed distribution. For the wet season, although the CV for all indicators was slightly lower than the measured values, their ranges (8.47–80.37 vs. 4.15–98.36 mg L<sup>−1</sup>) still showed a high degree of overlap, indicating that no systematic distortion was introduced.</p>

<table-wrap id="T3" specific-use="star"><label>Table 3</label><caption><p id="d2e5964">Statistical characteristics of different virtual samples.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="10">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:colspec colnum="5" colname="col5" align="right"/>
     <oasis:colspec colnum="6" colname="col6" align="right"/>
     <oasis:colspec colnum="7" colname="col7" align="right"/>
     <oasis:colspec colnum="8" colname="col8" align="right"/>
     <oasis:colspec colnum="9" colname="col9" align="right"/>
     <oasis:colspec colnum="10" colname="col10" align="right"/>
     <oasis:thead>
       <oasis:row>
         <oasis:entry colname="col1">Periods</oasis:entry>
         <oasis:entry colname="col2"/>
         <oasis:entry colname="col3">pH</oasis:entry>
         <oasis:entry colname="col4"><inline-formula><mml:math id="M412" display="inline"><mml:mi>T</mml:mi></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col5">EC</oasis:entry>
         <oasis:entry colname="col6">DO</oasis:entry>
         <oasis:entry colname="col7">ORP</oasis:entry>
         <oasis:entry colname="col8">Salt</oasis:entry>
         <oasis:entry colname="col9">TDS</oasis:entry>
         <oasis:entry colname="col10">NO<inline-formula><mml:math id="M413" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula></oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Unit</oasis:entry>
         <oasis:entry colname="col3"/>
         <oasis:entry colname="col4">°</oasis:entry>
         <oasis:entry colname="col5"><inline-formula><mml:math id="M414" display="inline"><mml:mrow class="unit"><mml:mi mathvariant="normal">µ</mml:mi></mml:mrow></mml:math></inline-formula>s cm<sup>−1</sup></oasis:entry>
         <oasis:entry colname="col6">mg L<sup>−1</sup></oasis:entry>
         <oasis:entry colname="col7">mv</oasis:entry>
         <oasis:entry colname="col8">ppt</oasis:entry>
         <oasis:entry colname="col9">mg L<sup>−1</sup></oasis:entry>
         <oasis:entry colname="col10">mg L<sup>−1</sup></oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">Normal season</oasis:entry>
         <oasis:entry colname="col2">Max</oasis:entry>
         <oasis:entry colname="col3">8.32</oasis:entry>
         <oasis:entry colname="col4">16.2</oasis:entry>
         <oasis:entry colname="col5">963.6</oasis:entry>
         <oasis:entry colname="col6">9.31</oasis:entry>
         <oasis:entry colname="col7">101.02</oasis:entry>
         <oasis:entry colname="col8">0.42</oasis:entry>
         <oasis:entry colname="col9">628.6</oasis:entry>
         <oasis:entry colname="col10">124.03</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"><inline-formula><mml:math id="M419" display="inline"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 660</oasis:entry>
         <oasis:entry colname="col2">Min</oasis:entry>
         <oasis:entry colname="col3">7.95</oasis:entry>
         <oasis:entry colname="col4">13.88</oasis:entry>
         <oasis:entry colname="col5">378.2</oasis:entry>
         <oasis:entry colname="col6">3.52</oasis:entry>
         <oasis:entry colname="col7"><inline-formula><mml:math id="M420" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>48.28</oasis:entry>
         <oasis:entry colname="col8">0.12</oasis:entry>
         <oasis:entry colname="col9">245</oasis:entry>
         <oasis:entry colname="col10">5.10</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean</oasis:entry>
         <oasis:entry colname="col3">8.18</oasis:entry>
         <oasis:entry colname="col4">14.72</oasis:entry>
         <oasis:entry colname="col5">527.31</oasis:entry>
         <oasis:entry colname="col6">6.66</oasis:entry>
         <oasis:entry colname="col7">2.09</oasis:entry>
         <oasis:entry colname="col8">0.20</oasis:entry>
         <oasis:entry colname="col9">342.72</oasis:entry>
         <oasis:entry colname="col10">30.41</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">SD</oasis:entry>
         <oasis:entry colname="col3">0.07</oasis:entry>
         <oasis:entry colname="col4">0.54</oasis:entry>
         <oasis:entry colname="col5">148.31</oasis:entry>
         <oasis:entry colname="col6">1.66</oasis:entry>
         <oasis:entry colname="col7">25.97</oasis:entry>
         <oasis:entry colname="col8">0.08</oasis:entry>
         <oasis:entry colname="col9">96.92</oasis:entry>
         <oasis:entry colname="col10">28.78</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">CV</oasis:entry>
         <oasis:entry colname="col3">0.01</oasis:entry>
         <oasis:entry colname="col4">0.04</oasis:entry>
         <oasis:entry colname="col5">0.28</oasis:entry>
         <oasis:entry colname="col6">0.25</oasis:entry>
         <oasis:entry colname="col7">12.44</oasis:entry>
         <oasis:entry colname="col8">0.41</oasis:entry>
         <oasis:entry colname="col9">0.28</oasis:entry>
         <oasis:entry colname="col10">0.95</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Dry season</oasis:entry>
         <oasis:entry colname="col2">Max</oasis:entry>
         <oasis:entry colname="col3">8.19</oasis:entry>
         <oasis:entry colname="col4">17.59</oasis:entry>
         <oasis:entry colname="col5">987.8</oasis:entry>
         <oasis:entry colname="col6">8.10</oasis:entry>
         <oasis:entry colname="col7">69.18</oasis:entry>
         <oasis:entry colname="col8">0.44</oasis:entry>
         <oasis:entry colname="col9">641.8</oasis:entry>
         <oasis:entry colname="col10">178.09</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"><inline-formula><mml:math id="M421" display="inline"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 650</oasis:entry>
         <oasis:entry colname="col2">Min</oasis:entry>
         <oasis:entry colname="col3">7.04</oasis:entry>
         <oasis:entry colname="col4">14.44</oasis:entry>
         <oasis:entry colname="col5">417.6</oasis:entry>
         <oasis:entry colname="col6">2.68</oasis:entry>
         <oasis:entry colname="col7"><inline-formula><mml:math id="M422" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>59.14</oasis:entry>
         <oasis:entry colname="col8">0.132</oasis:entry>
         <oasis:entry colname="col9">267.8</oasis:entry>
         <oasis:entry colname="col10">5.19</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean</oasis:entry>
         <oasis:entry colname="col3">7.35</oasis:entry>
         <oasis:entry colname="col4">15.49</oasis:entry>
         <oasis:entry colname="col5">661.79</oasis:entry>
         <oasis:entry colname="col6">6.414</oasis:entry>
         <oasis:entry colname="col7">0.52</oasis:entry>
         <oasis:entry colname="col8">0.27</oasis:entry>
         <oasis:entry colname="col9">429.90</oasis:entry>
         <oasis:entry colname="col10">37.75</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">SD</oasis:entry>
         <oasis:entry colname="col3">0.44</oasis:entry>
         <oasis:entry colname="col4">0.69</oasis:entry>
         <oasis:entry colname="col5">170.65</oasis:entry>
         <oasis:entry colname="col6">1.24</oasis:entry>
         <oasis:entry colname="col7">29.61</oasis:entry>
         <oasis:entry colname="col8">0.09</oasis:entry>
         <oasis:entry colname="col9">111.16</oasis:entry>
         <oasis:entry colname="col10">33.13</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">CV</oasis:entry>
         <oasis:entry colname="col3">0.06</oasis:entry>
         <oasis:entry colname="col4">0.04</oasis:entry>
         <oasis:entry colname="col5">0.26</oasis:entry>
         <oasis:entry colname="col6">0.19</oasis:entry>
         <oasis:entry colname="col7">57.21</oasis:entry>
         <oasis:entry colname="col8">0.34</oasis:entry>
         <oasis:entry colname="col9">0.26</oasis:entry>
         <oasis:entry colname="col10">0.88</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Wet season</oasis:entry>
         <oasis:entry colname="col2">Max</oasis:entry>
         <oasis:entry colname="col3">8.21</oasis:entry>
         <oasis:entry colname="col4">18.88</oasis:entry>
         <oasis:entry colname="col5">872.8</oasis:entry>
         <oasis:entry colname="col6">8.54</oasis:entry>
         <oasis:entry colname="col7">56.4</oasis:entry>
         <oasis:entry colname="col8">0.37</oasis:entry>
         <oasis:entry colname="col9">569</oasis:entry>
         <oasis:entry colname="col10">80.37</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"><inline-formula><mml:math id="M423" display="inline"><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 500</oasis:entry>
         <oasis:entry colname="col2">Min</oasis:entry>
         <oasis:entry colname="col3">6.31</oasis:entry>
         <oasis:entry colname="col4">16.12</oasis:entry>
         <oasis:entry colname="col5">395.8</oasis:entry>
         <oasis:entry colname="col6">5.34</oasis:entry>
         <oasis:entry colname="col7"><inline-formula><mml:math id="M424" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>77.64</oasis:entry>
         <oasis:entry colname="col8">0.13</oasis:entry>
         <oasis:entry colname="col9">257.6</oasis:entry>
         <oasis:entry colname="col10">8.47</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Mean</oasis:entry>
         <oasis:entry colname="col3">7.38</oasis:entry>
         <oasis:entry colname="col4">16.93</oasis:entry>
         <oasis:entry colname="col5">535.7</oasis:entry>
         <oasis:entry colname="col6">7.06</oasis:entry>
         <oasis:entry colname="col7">13.75</oasis:entry>
         <oasis:entry colname="col8">0.20</oasis:entry>
         <oasis:entry colname="col9">348.24</oasis:entry>
         <oasis:entry colname="col10">25.85</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">SD</oasis:entry>
         <oasis:entry colname="col3">0.33</oasis:entry>
         <oasis:entry colname="col4">0.62</oasis:entry>
         <oasis:entry colname="col5">131.02</oasis:entry>
         <oasis:entry colname="col6">0.78</oasis:entry>
         <oasis:entry colname="col7">28.85</oasis:entry>
         <oasis:entry colname="col8">0.08</oasis:entry>
         <oasis:entry colname="col9">86.05</oasis:entry>
         <oasis:entry colname="col10">19.88</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">CV</oasis:entry>
         <oasis:entry colname="col3">0.04</oasis:entry>
         <oasis:entry colname="col4">0.04</oasis:entry>
         <oasis:entry colname="col5">0.24</oasis:entry>
         <oasis:entry colname="col6">0.11</oasis:entry>
         <oasis:entry colname="col7">2.10</oasis:entry>
         <oasis:entry colname="col8">0.38</oasis:entry>
         <oasis:entry colname="col9">0.25</oasis:entry>
         <oasis:entry colname="col10">0.77</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <p id="d2e6680">Standardized multivariate boxplots (Fig. 12) visually confirm that the median, interquartile range (IQR), whisker length, and outlier distribution of the virtual data for each period were highly similar to the measured data, demonstrating that the central tendency and dispersion characteristics were well-preserved. Hydrological seasonal characteristics, such as high EC, TDS, Cl<sup>−</sup>, and NO<inline-formula><mml:math id="M426" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> in the dry season and low, concentrated NO<inline-formula><mml:math id="M427" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> in the wet season, were also accurately preserved. Although the number of some newly added outliers slightly increased, they all fell within physically reasonable ranges, with no non-physical solutions, such as negative concentrations or out-of-bounds pH values, occurring. Figure 13 presents a comparison of nitrate concentration frequency distributions between the original and virtual datasets across normal, dry, and wet periods. The distributional comparison indicates that the proposed t-SNE <inline-formula><mml:math id="M428" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> GMM <inline-formula><mml:math id="M429" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> KNN inverse mapping virtual sample generation strategy maintains the core features of the NO<inline-formula><mml:math id="M430" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> distribution for each hydrological period, while simultaneously improving sample representation in sparse areas and intervals of high variability. Therefore, the t-SNE <inline-formula><mml:math id="M431" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> GMM method effectively captured the non-linear structure and extreme value information of the original data, and can provide reliable data support for subsequent model training.</p>

      <fig id="F12" specific-use="star"><label>Figure 12</label><caption><p id="d2e6753">Box plots of the observed and virtual variable data at different periods.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f12.png"/>

          </fig>

      <fig id="F13" specific-use="star"><label>Figure 13</label><caption><p id="d2e6764">Comparison of nitrate concentration distribution in the original and virtual datasets under. different periods.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f13.png"/>

          </fig>

      <p id="d2e6773">The Kolmogorov-Smirnov (KS) test results indicate that the dry season exhibited the lowest mean KS statistic (0.1214 <inline-formula><mml:math id="M432" display="inline"><mml:mo>±</mml:mo></mml:math></inline-formula> 0.0387), with 80.0 % of the features falling below the strict threshold of 0.15 (Fig. S2 in the Supplement). This demonstrates that the virtual samples during this period highly reproduced the distribution characteristics of the observed data. The normal season followed, with a mean KS statistic of 0.1278 <inline-formula><mml:math id="M433" display="inline"><mml:mo>±</mml:mo></mml:math></inline-formula> 0.0356, where 72.5 % of the features had KS values below 0.15. The distribution similarity during the wet season was relatively lower, characterized by a mean KS statistic of 0.1657 <inline-formula><mml:math id="M434" display="inline"><mml:mo>±</mml:mo></mml:math></inline-formula> 0.0542; although 37.5 % of the features had KS values below 0.15, 70.0 % still satisfied the acceptable criterion of KS <inline-formula><mml:math id="M435" display="inline"><mml:mi mathvariant="italic">&lt;</mml:mi></mml:math></inline-formula> 0.20. The P-values from the KS test provided a quantitative assessment of the statistical significance regarding distributional differences. When <inline-formula><mml:math id="M436" display="inline"><mml:mi>P</mml:mi></mml:math></inline-formula> <inline-formula><mml:math id="M437" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.05, the null hypothesis could not be rejected, indicating no significant difference between the distributions of the virtual and observed samples (Figs. S2 and 14). Jensen-Shannon (JS) divergence analysis further corroborated these findings (Fig. 15). The mean JS divergence across all three hydrological periods remained below the threshold of 0.05, demonstrating high consistency with the original distributions. These low JS divergence values indicate that the probability distributions of the synthetic samples closely approximated the observed data, with no significant distributional shift observed. Notably, the JS divergence values for Salt and NO<inline-formula><mml:math id="M438" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> were relatively higher across all hydrological periods. Evaluation results using the Maximum Mean Discrepancy (MMD) showed that MMD values approaching zero and remaining within the confidence interval indicate no significant distributional differences between the synthetic and observed data in the Reproducing Kernel Hilbert Space (RKHS), confirming that the virtual samples introduced no systematic bias (Fig. 16).</p>

      <fig id="F14"><label>Figure 14</label><caption><p id="d2e6833">Mean KS <inline-formula><mml:math id="M439" display="inline"><mml:mi>P</mml:mi></mml:math></inline-formula>-values across different sample generation factors during normal, dry, and wet seasons. The dashed red line indicates the significance level (<inline-formula><mml:math id="M440" display="inline"><mml:mrow><mml:mi mathvariant="italic">α</mml:mi><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.05). Values above this threshold suggest no significant distributional difference between virtual and observed samples.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f14.png"/>

          </fig>

      <fig id="F15" specific-use="star"><label>Figure 15</label><caption><p id="d2e6862">Boxplots of JS divergence for water quality features across hydrological seasons. The dashed red line denotes the strict similarity threshold (JS <inline-formula><mml:math id="M441" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 0.05). Lower values indicate higher distributional consistency.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f15.png"/>

          </fig>

      <fig id="F16"><label>Figure 16</label><caption><p id="d2e6880">Variation of Maximum Mean Discrepancy (MMD) with sample generation factor across seasons. MMD values approaching zero and remaining within the confidence interval confirm the absence of systematic bias in the virtual samples within the Reproducing Kernel Hilbert Space (RKHS).</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f16.png"/>

          </fig>

      <p id="d2e6889">Analysis of the impact of sample generation multipliers revealed that a multiplier of 7 to 10 times yielded the optimal performance across all hydrological periods. Within this range, both KS statistics and JS divergence remained at low levels with minimal fluctuation, indicating the most stable distribution similarity. In contrast, sample generation with multipliers of 1 to 3 times exhibited larger fluctuations in some indicators, demonstrating insufficient stability. Based on a comprehensive evaluation, this study recommends adopting a sample generation multiplier of at least 7 to ensure the quality of virtual samples. Feature-level analysis revealed the response characteristics of different water quality parameters to the data augmentation strategy. The KS statistics for ORP, Temperature, and EC were generally low (mean <inline-formula><mml:math id="M442" display="inline"><mml:mi mathvariant="italic">&lt;</mml:mi></mml:math></inline-formula> 0.12), indicating excellent performance in virtual sample generation for these features. In contrast, Salt, NO<inline-formula><mml:math id="M443" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, and TDS exhibited relatively higher KS statistics (mean <inline-formula><mml:math id="M444" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.18). Nevertheless, the MMD values for all features remained within the confidence interval, suggesting that even for features with high KS values, the virtual samples did not introduce significant distributional bias. These diagnostics confirm that the t-SNE-GMM-KNN strategy effectively mitigates small sample bias without introducing systematic distributional distortion.</p>
</sec>
<sec id="Ch1.S3.SS4.SSS2">
  <label>3.4.2</label><title>Prediction based on on-site measured water quality data</title>
      <p id="d2e6926">In the normal season, baseline and quantum enhanced RF models achieved <inline-formula><mml:math id="M445" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> of 0.6727 and 0.6602, respectively, with wide bootstrap 95 % CIs reflecting the limited sample size (Table S2). Augmenting virtual samples from 1<inline-formula><mml:math id="M446" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> to 10<inline-formula><mml:math id="M447" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> the original size steadily raised <inline-formula><mml:math id="M448" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> above 0.958 (quantum enhanced RF reaching 0.9622, 95 % CI: [0.9744, 0.9881]), with gains plateauing beyond approximately 500 virtual samples. The dry season showed the poorest baseline performance (<inline-formula><mml:math id="M449" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.2841, 95 % CI: [0.2884, 0.7276]; Tables S2–S3), attributable to high NO<inline-formula><mml:math id="M450" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> variability and outlier sensitivity under data sparsity. Virtual sampling markedly improved accuracy: <inline-formula><mml:math id="M451" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> reached 0.527 at 1<inline-formula><mml:math id="M452" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation and 0.854 (95 % CI: [0.8431, 0.9221]) at 8<inline-formula><mml:math id="M453" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>. Although the quantum enhanced model slightly lagged at 2<inline-formula><mml:math id="M454" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation or less, both models converged to high accuracy, confirming that virtual samples effectively mitigated data sparsity and skewed distributions. The wet season yielded the best baseline performance (<inline-formula><mml:math id="M455" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.7331, 95 % CI: [0.6715, 0.8926]; Table S4), owing to lower NO<inline-formula><mml:math id="M456" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations and smaller spatial variability. Augmentation further raised accuracy to <inline-formula><mml:math id="M457" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.962 at 4<inline-formula><mml:math id="M458" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation and a stable 0.977 at 10<inline-formula><mml:math id="M459" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> (RMSE as low as 3.03 mg L<sup>−1</sup>; 95 % CI: [0.9787, 0.9875]). Quantum enhanced and classical RF performed nearly identically, indicating limited marginal benefit from quantum feature encoding when data quality is high and relationships are more linear.</p>
      <p id="d2e7088">In the normal season (Fig. 17A1–A2), both models trained on the 66 original samples showed marked deviations between predicted and observed values, with highly dispersed predictions and medians far from the observed median, consistent with the low <inline-formula><mml:math id="M461" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> of small sample modeling. As virtual samples increased, predicted distributions gradually converged toward the observed values and the boxplot IQR and whiskers narrowed, indicating substantially improved stability and accuracy. At 10<inline-formula><mml:math id="M462" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation (660 virtual samples), predicted and observed boxplots nearly overlapped, with RMSE reduced to 6.02 mg L<sup>−1</sup>; the quantum enhanced model slightly outperformed classical RF (<inline-formula><mml:math id="M464" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.9622, bootstrap 95 % CI: [0.9744, 0.9881]), confirming robust accuracy despite the small original sample. In the dry season (Fig. 17B1–B2), both models systematically overestimated NO<inline-formula><mml:math id="M465" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> on the original data owing to its extremely high and skewed concentrations, placing predicted boxplots entirely above the observed values (<inline-formula><mml:math id="M466" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.28, 95 % CI: [0.2884, 0.7276]). Virtual sampling brought substantial improvement: predicted medians and ranges began converging from 1<inline-formula><mml:math id="M467" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation, and at 8<inline-formula><mml:math id="M468" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> or higher the models captured the high concentration intervals, with RF reaching <inline-formula><mml:math id="M469" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.8542 (95 % CI: [0.8431, 0.9221]) at 9<inline-formula><mml:math id="M470" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>. Classical RF slightly outperformed the quantum enhanced model at low generation levels, but their performance converged as sample size grew, demonstrating that virtual sample generation effectively alleviates challenges from data sparsity and extreme values. In the wet season (Fig. 17C1–C2), predictions already agreed well with observations on the original data, and virtual samples further narrowed the IQR and concentrated predictions within the observed distribution. At 10<inline-formula><mml:math id="M471" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation, agreement was exceptionally high (<inline-formula><mml:math id="M472" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.977, RMSE <inline-formula><mml:math id="M473" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 3.03 mg L<sup>−1</sup>), demonstrating excellent predictive performance.</p>

      <fig id="F17" specific-use="star"><label>Figure 17</label><caption><p id="d2e7236">Comparison of observed and predicted NO<inline-formula><mml:math id="M475" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations across data generation levels for random forest and quantum feature-enhanced random forest models in normal, dry, and wet seasons.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f17.png"/>

          </fig>

</sec>
<sec id="Ch1.S3.SS4.SSS3">
  <label>3.4.3</label><title>Prediction based on AlphaEarth Foundation Embeddings</title>
      <p id="d2e7265">To explore the potential of remote sensing semantic embedding features in predicting groundwater nitrate concentrations, this section employs the 64-dimensional surface semantic vectors derived from the Google AlphaEarth Foundation (AEF) dataset as model input variables. We reduced the variables through principal component analysis to preserve <inline-formula><mml:math id="M476" display="inline"><mml:mo>≥</mml:mo></mml:math></inline-formula> 95 % variance.</p>
      <p id="d2e7275">In the normal season, models trained on the original samples performed poorly, with <inline-formula><mml:math id="M477" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> of 0.167 (RF) and 0.119 (quantum enhanced RF) and RMSE as high as 32.89 and 33.82 mg L<sup>−1</sup>, respectively (Table S5), indicating that AEF embedding features alone cannot adequately capture the hydrological processes of this period under limited sample size. Virtual sampling markedly improved performance: at 10<inline-formula><mml:math id="M479" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation, RF reached <inline-formula><mml:math id="M480" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.860 with RMSE reduced to 10.73 mg L<sup>−1</sup> and a narrowed bootstrap 95 % CI of [0.8792, 0.9326], while the quantum enhanced RF achieved <inline-formula><mml:math id="M482" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.844 with a consistent trend, confirming statistically stable performance even with indirect remote sensing inputs. Boxplot comparison (Fig. 18A) shows that initial predictions severely overestimated low to medium concentrations while underestimating high value tails; as sample size expanded, predicted distributions progressively converged toward the observed values, with much improved agreement in median and IQR. This confirms that virtual samples effectively enhanced the capability of AEF features to represent nonlinear patterns.</p>

      <fig id="F18" specific-use="star"><label>Figure 18</label><caption><p id="d2e7349">Comparison of observation and prediction of NO<inline-formula><mml:math id="M483" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentration by random forest and quantum featution-enhanced random forest models at data enhancement levels in normal, dry and wet seasons: based on AlphaEarth Foundation as the input variable.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f18.png"/>

          </fig>

      <p id="d2e7371">In the dry season, the original dataset yielded a negative <inline-formula><mml:math id="M484" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, reflecting the weak generalization of AEF features under high variability and heavy tailed distributions (Table S6). Performance improved steadily with virtual sampling: <inline-formula><mml:math id="M485" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> rose to 0.039 at 1<inline-formula><mml:math id="M486" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, 0.641 at 8<inline-formula><mml:math id="M487" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> (RMSE <inline-formula><mml:math id="M488" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 21.05 mg L<sup>−1</sup>, bootstrap 95 % CI: [0.6385, 0.8638]), and 0.674 at 10<inline-formula><mml:math id="M490" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> (RMSE <inline-formula><mml:math id="M491" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 19.86 mg L<sup>−1</sup>). The quantum enhanced RF slightly outperformed standard RF at high expansion levels, suggesting that quantum encoding helps mitigate the influence of extreme values and enhances robustness (Fig. 18B). Prediction distributions show that the initial model entirely failed to capture the high concentration clustering of dry season NO<inline-formula><mml:math id="M493" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula>, whereas after data generation the predicted boxplots progressively covered the true high value intervals, with tail behavior aligning with observations. In the wet season, although baseline <inline-formula><mml:math id="M494" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> was also negative, improvement was the most rapid (Table S7): <inline-formula><mml:math id="M495" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> reached 0.5 at only 2<inline-formula><mml:math id="M496" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> expansion, 0.685 at 5<inline-formula><mml:math id="M497" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, and stabilized at 0.784 (RF, RMSE <inline-formula><mml:math id="M498" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 8.27 mg L<sup>−1</sup>) and 0.781 (quantum enhanced RF) at 10<inline-formula><mml:math id="M500" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>. Boxplots confirm that the initially dispersed and systematically biased predictions rapidly converged to the dense intervals of observed values, with final IQR and whisker ranges showing high overlap.</p>
      <p id="d2e7531">Compared to modeling results based on in-situ observation data, the predictive performance based on AlphaEarth Foundation embedding features was generally lower. Under the same virtual sample generation multiplier, the maximum <inline-formula><mml:math id="M501" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> for the normal, dry, and wet seasons were approximately 10.27 %, 17.37 %, and 19.33 % lower, respectively. This indicates that measured water quality parameters more directly reflect the key processes of nitrogen migration and transformation. However, given that AEF can be obtained globally without the need for field sampling, it offers a feasible alternative for the rapid screening of groundwater nitrate risks in large-scale unmonitored areas.</p>
</sec>
</sec>
<sec id="Ch1.S3.SS5">
  <label>3.5</label><title>Feature importance analysis</title>
      <p id="d2e7555">The dominant predictive factors vary across different seasons, and the virtual sample generation strategy influences both the stability of feature importance and model performance. There are distinct differences in the key driving factors for each season, which aligns with the results of the Bayesian models and correlation analysis (Fig. 19). In the normal season, TDS, EC, Salt, and DO are the most important predictive variables, with their importance significantly higher than that of other parameters. In the dry season, TDS, EC, Salt, and pH exhibit the highest importance. In the wet season, the importance of TDS, EC, Salt, and ORP is most prominent. With the increase in the number of virtual samples, the ranking of feature importance tends to stabilize. For instance, in the dry season, when the sample size increased from the original 65 to 715, the importance of TDS and EC continued to rise and eventually stabilized. Comparing the RF and quantum-enhanced models, quantum enhancement did not fundamentally alter the ranking of feature importance; however, it slightly increased the importance of certain variables or made them more stable, demonstrating the effectiveness of quantum feature encoding as a means of information enhancement.</p>

      <fig id="F19" specific-use="star"><label>Figure 19</label><caption><p id="d2e7560">Input feature importance of the classical Random Forest (RF) and quantum-enhanced RF models for seasonal nitrate prediction using in-situ measured water quality parameters. Panels <bold>(a)</bold>–<bold>(c)</bold> correspond to the normal, dry, and wet seasons, respectively. Columns represent the original, 1<inline-formula><mml:math id="M502" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, 5<inline-formula><mml:math id="M503" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, and 10<inline-formula><mml:math id="M504" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> virtual-sample datasets.</p></caption>
          <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f19.png"/>

        </fig>

      <p id="d2e7596">Figure 20 presents the feature importance of the RF and quantum enhanced RF models using only the 64 dimensional AEF embedding vectors as inputs. Although these abstract features cannot be assigned explicit physical meanings, their importance rankings reveal which remote sensing semantics are critical for nitrate prediction. Compared with in situ parameters, AEF feature importance fluctuates considerably across seasons and data volumes, with no consistent core feature set, indicating that AEF embeddings, despite their rich environmental information, correlate only weakly with nitrate and require large data volumes to establish a robust mapping. Seasonally, A05, A07, and A00 ranked highest in the normal season (likely encoding land use, soil moisture, or vegetation cover); A08, A06, and A05 in the dry season (associated with surface dryness, bare land, or human activity intensity, matching the spatial distribution of high concentration pollution sources); and A02, A03, and A05 in the wet season (related to runoff, vegetation growth, or soil water content, reflecting rainfall driven pollutant migration). Virtual sample generation proved crucial for stabilizing these rankings: the initially chaotic importance ordering on the original data gradually clarified as virtual samples increased, highlighting core features and further demonstrating the effectiveness of the generation strategy for small sample modeling. The quantum enhanced model showed a similar importance distribution to classical RF but occasionally assigned slightly higher weights to certain features, suggesting that quantum feature encoding helps extract more discriminative information from the high dimensional remote sensing semantic space and marginally refines feature selection.</p>

      <fig id="F20" specific-use="star"><label>Figure 20</label><caption><p id="d2e7602">Importance of AlphaEarth Foundation (AEF) embedding features for seasonal nitrate prediction using the classical Random Forest (RF) and quantum-enhanced RF models. Panels <bold>(a)</bold>–<bold>(c)</bold> correspond to the normal, dry, and wet seasons, respectively. Columns represent the original, 1<inline-formula><mml:math id="M505" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, 5<inline-formula><mml:math id="M506" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, and 10<inline-formula><mml:math id="M507" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> virtual-sample datasets.</p></caption>
          <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f20.png"/>

        </fig>

      <p id="d2e7638">Figure 21 presents a local feature attribution analysis for representative samples predicting the highest and lowest NO<inline-formula><mml:math id="M508" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations using SHAP waterfall plots. Regardless of whether the classical RF or the quantum-enhanced RF model is used, samples predicting high NO<inline-formula><mml:math id="M509" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations are driven by a set of features with positive contributions (red bars). In the normal season, for the highest NO<inline-formula><mml:math id="M510" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> sample with a predicted value of 161.17 mg L<sup>−1</sup>, features A05, A09, and A06 contributed the highest positive values, with A05 making the largest contribution and serving as the key factor driving the prediction to a high level. In the dry season, for the sample with a predicted value as high as 358.58 mg L<sup>−1</sup>, features A12, A00, and A07 were the main positive driving factors, with A12 contributing most prominently. In the wet season, for the sample with a predicted value of 98.36 mg L<sup>−1</sup>, features A03, A02, and A04 provided the main positive contributions, with A03 contributing the most. For samples predicting low NO<inline-formula><mml:math id="M514" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations, model decisions mainly rely on features with negative contributions (blue bars). The role of these features is to pull the predicted value down from the baseline <inline-formula><mml:math id="M515" display="inline"><mml:mrow><mml:mo>(</mml:mo><mml:mi>E</mml:mi><mml:mo>[</mml:mo><mml:mi>f</mml:mi><mml:mo>(</mml:mo><mml:mi>X</mml:mi><mml:mo>)</mml:mo><mml:mo>]</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. In the normal season, for the lowest NO<inline-formula><mml:math id="M516" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> sample with a predicted value of 2.39 mg L<sup>−1</sup>, features A05, A04, and A00 exhibited strong negative contributions, with A05 showing the largest negative contribution. In the dry season, for the sample with a predicted value of only 0.10 mg L<sup>−1</sup>, features A09, A12, and A06 were the main negative driving factors, with A09 contributing the most negatively. In the wet season, for the sample with a predicted value of 4.15 mg L<sup>−1</sup>, features A03, A06, and A00 provided the main negative contributions, with A03 contributing the most negatively.</p>

      <fig id="F21" specific-use="star"><label>Figure 21</label><caption><p id="d2e7801">SHAP waterfall-based local feature attribution for representative samples with the highest and lowest observed NO<inline-formula><mml:math id="M520" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations. Panels <bold>(a)</bold>–<bold>(c)</bold> correspond to the normal, dry, and wet seasons, respectively. In each panel, the left group shows the classical RF model and the right group shows the quantum-enhanced RF model; within each group, the left and right plots correspond to the highest- and lowest-concentration samples, respectively.</p></caption>
          <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f21.png"/>

        </fig>

</sec>
<sec id="Ch1.S3.SS6">
  <label>3.6</label><title>Sensitivity analysis of key parameters</title>
      <p id="d2e7836">To quantitatively evaluate the response intensity of the model's nitrate concentration prediction results to the perturbation of input hydrochemical parameters, and to verify the robustness of the model and the reliability of feature importance ranking, this study conducted a multi-dimensional sensitivity analysis based on the Sobol global sensitivity index (first-order and total-order), one-at-a-time (OAT) feature perturbation test, and elasticity coefficient calculation. The analysis was performed for both classical Random Forest and hybrid quantum-classical Random Forest models under different virtual sample generation gradients (original, 1<inline-formula><mml:math id="M521" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, 5<inline-formula><mml:math id="M522" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>, 10<inline-formula><mml:math id="M523" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>) across normal, dry, and wet hydrological seasons, to systematically reveal the key controlling parameters of the model and the stability of parameter sensitivity under data generation.</p>
<sec id="Ch1.S3.SS6.SSS1">
  <label>3.6.1</label><title>Global sensitivity analysis based on Sobol' index</title>
      <p id="d2e7867">In the normal season, the total order Sobol indices of the classical RF consistently ranked TDS (0.2882 to 0.2933), EC (0.2701 to 0.2839), and Salt (0.2411 to 0.2612) as the three most sensitive parameters across all generation gradients (Fig. 22), in close agreement with the Gini and SHAP importance rankings. The first order index of TDS remained above 0.29 in the original data and stabilized at 0.2938 after 10<inline-formula><mml:math id="M524" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation, indicating a stable and dominant independent contribution to nitrate prediction. In contrast, pH, <inline-formula><mml:math id="M525" display="inline"><mml:mi>T</mml:mi></mml:math></inline-formula>, DO, and ORP showed total order indices below 0.08, reflecting limited influence. The quantum enhanced RF yielded the same overall ranking, with the sensitivity of TDS, EC, and Salt further stabilized at 10<inline-formula><mml:math id="M526" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation (0.2586, 0.2516, and 0.2416, respectively). Notably, the total order sensitivity of ORP rose from 0.0313 (original) to 0.0608 (10<inline-formula><mml:math id="M527" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>), suggesting that quantum feature encoding strengthens the model's perception of redox contributions to nitrate dynamics and that virtual sampling improves the stability of sensitivity estimates for low contribution parameters.</p>

      <fig id="F22" specific-use="star"><label>Figure 22</label><caption><p id="d2e7900">Sensitivity analysis of key parameters during the normal season.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f22.png"/>

          </fig>

      <p id="d2e7909">In the dry season, characterized by highly skewed nitrate distributions and strong evaporative concentration, parameter sensitivity differed from the normal season (Fig. 23). For classical RF on the original data, the five most sensitive parameters were TDS (0.2713), Salt (0.1907), DO (0.1766), EC (0.1739), and pH (0.1146). The notably higher pH sensitivity than in other seasons indicates that the acid-base environment strongly controls nitrate accumulation under intense evaporation. With increasing virtual sample generation, the sensitivity of TDS, EC, and Salt gradually stabilized, while DO remained high (0.1596 to 0.2165), confirming its persistent control over nitrate as an indicator of nitrification intensity in the oxidizing dry season environment. After 10<inline-formula><mml:math id="M528" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation, the quantum enhanced RF showed higher sensitivity for pH (0.1015) and DO (0.1665) than classical RF, suggesting the hybrid quantum-classical framework is more robust in capturing nonlinear hydrochemistry-nitrate relationships under the highly heterogeneous, small sample conditions of the dry season.</p>

      <fig id="F23" specific-use="star"><label>Figure 23</label><caption><p id="d2e7922">Sensitivity analysis of key parameters during the dry season.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f23.png"/>

          </fig>

      <p id="d2e7931">In the wet season dominated by precipitation dilution and infiltration, the parameter sensitivity pattern was more concentrated (Fig. 24). For the classical RF model, TDS (total-order index 0.2819–0.3179), Salt (0.2565–0.2810), and EC (0.2478–0.2492) maintained an absolute dominant position in global sensitivity across all generation gradients, with the sum of their total-order indices exceeding 75 % of the total sensitivity of all parameters. The total-order sensitivity of ORP ranked fourth (0.1153–0.1435), which was higher than that in the normal and dry seasons, reflecting that the redox potential change caused by rainfall infiltration is a key secondary factor affecting nitrate leaching and migration in the wet season. The sensitivity of pH, <inline-formula><mml:math id="M529" display="inline"><mml:mi>T</mml:mi></mml:math></inline-formula>, and DO was extremely low, with total-order indices all below 0.03, indicating that these parameters have limited independent influence on nitrate prediction under the dilution effect of heavy precipitation. The quantum-enhanced RF model showed a consistent sensitivity ranking with the classical RF, and the total-order sensitivity of ORP further increased to 0.1343 after 10x generation, verifying the reliability of ORP as a key driving parameter in the wet season.</p>

      <fig id="F24" specific-use="star"><label>Figure 24</label><caption><p id="d2e7943">Sensitivity analysis of key parameters during the wet season.</p></caption>
            <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f24.png"/>

          </fig>

</sec>
<sec id="Ch1.S3.SS6.SSS2">
  <label>3.6.2</label><title>Feature perturbation sensitivity and elasticity analysis</title>
      <p id="d2e7960">The OAT perturbation test was performed by varying each target parameter within <inline-formula><mml:math id="M530" display="inline"><mml:mo>±</mml:mo></mml:math></inline-formula>30 % of its measured range while fixing others, and elasticity coefficients were calculated to quantify the directional response of predicted nitrate per unit parameter change (<inline-formula><mml:math id="M531" display="inline"><mml:mo lspace="0mm">|</mml:mo></mml:math></inline-formula>value<inline-formula><mml:math id="M532" display="inline"><mml:mo>|</mml:mo></mml:math></inline-formula> <inline-formula><mml:math id="M533" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 1 indicates an elastic response). The results corroborated the Sobol analysis. In the normal season, the perturbation sensitivity of TDS and EC remained highest (<inline-formula><mml:math id="M534" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.11) for both models after 10<inline-formula><mml:math id="M535" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation. In the dry season, pH sensitivity rose sharply from 0.3033 to 2.0128 (classical RF) and from 0.6026 to 1.3757 (quantum enhanced RF), confirming the model's strong responsiveness to pH; correspondingly, pH elasticity coefficients reached 18.5446 and 24.2152 after 10<inline-formula><mml:math id="M536" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation, an extremely strong positive elastic response identifying pH as the key control on dry season nitrate accumulation. In the wet season, the perturbation sensitivity of Salt and TDS stabilized at 0.1358 and 0.1616 (classical RF, 10<inline-formula><mml:math id="M537" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula>) as the highest responses. Elasticity analysis further showed that in the normal season <inline-formula><mml:math id="M538" display="inline"><mml:mi>T</mml:mi></mml:math></inline-formula>, pH, and EC had coefficients above 1 (positive elastic response), whereas ORP was inelastic (0.1037); in the wet season, the coefficients of TDS, EC, and Salt were all close to 1, indicating a near proportional linear response of nitrate to salinity indicators, consistent with the dominant dilution effect.</p>
</sec>
<sec id="Ch1.S3.SS6.SSS3">
  <label>3.6.3</label><title>Impact of virtual sample generation on sensitivity estimation</title>
      <p id="d2e8035">The t-SNE-GMM-KNN virtual sample generation markedly improved the stability and reliability of sensitivity estimation. For the dry season with the most severe data sparsity, the 95 % CI of the Sobol index for TDS narrowed by 87.9 % from original to 10<inline-formula><mml:math id="M539" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> generation, and the coefficient of variation of perturbation sensitivity decreased by an average of 62.4 %, indicating that augmentation effectively reduces small sample variance and avoids ranking bias from insufficient representation. As generation multiples increased, the ranking of core parameters (TDS, EC, Salt) remained stable while estimates for secondary parameters (ORP, DO, pH) gradually converged, confirming that virtual sampling improves statistical robustness without altering the inherent hydrochemical driving mechanism. The quantum enhanced RF further showed smaller sensitivity fluctuations across generation gradients than classical RF, particularly on the original dry season data (21.7 % lower coefficient of variation in total order indices), demonstrating that quantum feature encoding enhances robustness under small sample conditions. Overall, the multi-dimensional sensitivity analysis consistently identified TDS, EC, and salinity as the most critical and stable parameters across all seasons, in close agreement with the correlation, Bayesian, and SHAP analyses, while dry season pH and DO and wet season ORP emerged as season specific drivers, revealing differentiated controls on nitrate dynamics under varying hydrological conditions. These results validate the model structure and feature selection and provide a quantitative basis for optimizing monitoring indicators and targeted control of groundwater nitrate pollution.</p>
</sec>
</sec>
</sec>
<sec id="Ch1.S4">
  <label>4</label><title>Discussion</title>
<sec id="Ch1.S4.SS1">
  <label>4.1</label><title>Nitrogen sources, migration, and transformation</title>
      <p id="d2e8062">The Piper and Gibbs diagrams, together with ion ratios, indicate a Ca-Mg-HCO<inline-formula><mml:math id="M540" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> hydrochemical type dominated by carbonate dissolution with weak cation exchange (Liu et al., 2025b). The lower NO<inline-formula><mml:math id="M541" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> concentrations during the rainy season, combined with enhanced cation exchange, suggest rainfall-driven manure leaching accompanied by temporary aquifer retention of NO<inline-formula><mml:math id="M542" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> (Sun et al., 2024; Wang et al., 2025a). Isotopic evidence and MixSIAR apportionment identify domestic sewage and manure (DSM, 74.1 %) and soil organic nitrogen (SON, 20.9 %) as the dominant nitrate sources, with negligible fertilizer and precipitation inputs (Mao et al., 2023). This pattern implies that direct fertilizer leaching is not the dominant pathway; rather, given the thick (<inline-formula><mml:math id="M543" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 10 m) vadose zone of the North China Plain, elevated NO<inline-formula><mml:math id="M544" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> likely derives from historical fertilizer residues and long-term manure infiltration, especially where farmlands adjoin rural residences (Wang et al., 2025b; Wu et al., 2024). The <inline-formula><mml:math id="M545" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">15</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>N-NO<inline-formula><mml:math id="M546" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> values (12.2 ‰–18.2 ‰) fall within the manure/SON range rather than that of chemical fertilizers, indicating microbial mineralization and nitrification, consistent with ammonification and nitrification of SON and DSM-derived ammonium under aerobic conditions (Li et al., 2022; Liu et al., 2023; Ahmed et al., 2013).</p>
      <p id="d2e8144">Seasonal nitrate variation reflects contrasting recharge regimes. During the dry season, scarce precipitation, strong evaporation, and declining groundwater levels create a migration potential gradient that drives nitrate accumulation in discharge areas; concurrently weakened cation exchange (reduced Na<sup>+</sup> adsorption and relative Ca<sup>2+</sup> depletion) lowers aquifer retention capacity, making accumulation the dominant process (Zhang et al., 2023). In the wet season, rapid infiltration raises groundwater levels and velocities and reactivates cation exchange. The oxidizing aquifer conditions, indicated by extremely low nitrite and ammonium concentrations, together with <inline-formula><mml:math id="M549" display="inline"><mml:mrow><mml:msup><mml:mi mathvariant="italic">δ</mml:mi><mml:mn mathvariant="normal">18</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>O values of nitrate (<inline-formula><mml:math id="M550" display="inline"><mml:mo lspace="0mm">-</mml:mo></mml:math></inline-formula>9.58 ‰ to 8.04 ‰) within the typical nitrification range, exclude significant denitrification and confirm nitrification as the dominant transformation process (Zhang et al., 2025).</p>
      <p id="d2e8186">Although the vadose zone of the North China Plain generally exceeds 10 m and should theoretically damp seasonal recharge signals (Liu et al., 2022), the distinct seasonal nitrate variability observed here can be attributed to three regional factors. First, decades of excessive fertilization have built a legacy nitrogen reservoir in the vadose zone, so groundwater nitrate is sustained by readily mobilized historical accumulation rather than current inputs alone (Liu et al., 2025a; Miao et al., 2023). Second, intense monsoonal rainfall can generate preferential flow paths that bypass slow matrix flow, enabling rapid nitrate leaching despite the thick unsaturated zone (Williams et al., 2023). Third, intensive irrigation abstraction alters the flow field and water table depth, enhancing hydraulic connectivity between soil nitrogen and the aquifer (Yang et al., 2022). Together, these factors explain why legacy nitrogen mobilization, dry season evaporative concentration, and wet season dilution override the damping effect of the thick vadose zone (Feng et al., 2024).</p>
</sec>
<sec id="Ch1.S4.SS2">
  <label>4.2</label><title>Virtual sample generation mitigates small sample bias and reveals seasonal sensitivity</title>
      <p id="d2e8197">Model overfitting and poor generalization due to small samples are prevalent challenges in environmental forecasting (Zhu et al., 2023). The virtual samples generated by the t-SNE/GMM/KNN strategy reproduce the statistical characteristics (mean, standard deviation, coefficient of variation) and seasonal hydrochemical differences of the original data, and the substantial performance gains after augmentation indicate that data sparsity, rather than model capacity, is the core bottleneck in seasonal nitrate modeling (Saha et al., 2023). Narrowing bootstrap confidence intervals with increasing sample size further confirms that augmentation reduces variance in performance estimates rather than exploiting specific data splits. The gains diverge seasonally: under the highly right skewed nitrate distribution of the dry season, <inline-formula><mml:math id="M551" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> rises from 0.28 to over 0.85 with tenfold expansion, because evaporation driven concentration intensifies spatial heterogeneity and process nonlinearity, requiring richer samples to capture tail behaviors (Li et al., 2025b); in the wet season, rainfall dilution homogenizes the system, so excellent accuracy is achieved with smaller marginal gains (Bigler et al., 2024).</p>
      <p id="d2e8211">The proposed t-SNE/GMM/KNN strategy outperforms traditional oversampling and deep generative models in preserving multimodal structures and heavy tailed covariance, as the latter either neglect the manifold geometry of high dimensional geochemical spaces or require large training datasets unavailable in this study (Udu et al., 2025). Its advantages over GMM and GAN based methods are threefold: t-SNE captures clustering structures driven by distinct hydrological processes; BIC based cluster optimization avoids subjective parameterization; and KNN inverse mapping reconstructs high dimensional samples without large scale training, suiting small sample scenarios (Silva and Melo-Pinto, 2023; Peng et al., 2025). Bootstrap analysis confirms that confidence intervals narrow with increasing virtual sample size; for the dry season RF model, the 95 % CI width drops from 0.440 to 0.053 under tenfold augmentation, indicating effective mitigation of small sample bias. Although tenfold virtual sample augmentation substantially improves predictive performance, overfitting to synthetic patterns remains a potential risk. The t-SNE/GMM/KNN framework mitigates this risk by adhering to the manifold structure and multimodal distributions of the original data; nevertheless, augmentation beyond tenfold may introduce artificial covariance unrepresentative of true geochemical processes (Tanner et al., 2022). Bootstrap analysis shows that performance gains plateau beyond eightfold generation, indicating that tenfold augmentation represents an empirical upper bound at which information extraction from limited field samples approaches saturation without compromising generalizability (Zhu et al., 2023).</p>
</sec>
<sec id="Ch1.S4.SS3">
  <label>4.3</label><title>Error propagation under virtual sample</title>
      <p id="d2e8222">Assuming a 5.0 % coefficient of variation for input hydrochemical parameters, we propagated uncertainty through 500 perturbed realizations and quantified output variability via standard deviation and 95 % confidence interval width (Fig. 25). Uncertainty propagation shows clear seasonal and model dependent patterns. In the normal season, both RF and quantum enhanced RF models remained stable, with output standard deviations decreasing by 4.5 % and 15.8 % at tenfold augmentation, indicating that data generation constrains uncertainty amplification under relatively symmetric distributions (Dega et al., 2023). The dry season reveals a critical divergence: classical RF uncertainty increased markedly (<inline-formula><mml:math id="M552" display="inline"><mml:mi mathvariant="italic">σ</mml:mi></mml:math></inline-formula>_out <inline-formula><mml:math id="M553" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>48.9 %; CI width <inline-formula><mml:math id="M554" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula>35.9 %), reflecting the dominance of extreme values in highly right skewed nitrate distributions, whereas the quantum enhanced RF reduced output standard deviation by 27.9 % and CI width by 20.6 %. This robustness suggests that quantum feature encoding via Pauli <inline-formula><mml:math id="M555" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> expectation values better disentangles nonlinear relationships in high variability regimes, limiting error amplification. In the wet season, dilution driven homogenization yields low, stable uncertainties for both models (<inline-formula><mml:math id="M556" display="inline"><mml:mi mathvariant="italic">σ</mml:mi></mml:math></inline-formula>_out<inline-formula><mml:math id="M557" display="inline"><mml:mi mathvariant="italic">&lt;</mml:mi></mml:math></inline-formula> 2.1 mg L<sup>−1</sup>), with the quantum enhanced RF again suppressing uncertainty more strongly (<inline-formula><mml:math id="M559" display="inline"><mml:mo lspace="0mm">-</mml:mo></mml:math></inline-formula>28.7 % versus <inline-formula><mml:math id="M560" display="inline"><mml:mo>-</mml:mo></mml:math></inline-formula>7.8 %), though absolute gains remain modest.</p>

      <fig id="F25" specific-use="star"><label>Figure 25</label><caption><p id="d2e8296">Variation of output standard deviation of classical Random Forest and quantum-enhanced Random Forest with virtual sample generation multiples under 5 % input coefficient of variation in normal, dry and wet seasons.</p></caption>
          <graphic xlink:href="https://hess.copernicus.org/articles/30/5647/2026/hess-30-5647-2026-f25.png"/>

        </fig>

      <p id="d2e8305">These results confirm that virtual sample generation enhances not only accuracy but also resilience to measurement uncertainty, provided the generation factor matches the complexity of the underlying distribution: for the high variability dry season, moderate augmentation is optimal, whereas excessive generation (beyond eightfold) may reintroduce variance in classical models. Three broader implications follow. First, error propagation analysis offers a robustness criterion beyond conventional accuracy metrics (<inline-formula><mml:math id="M561" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>, RMSE), which is especially valuable in small sample modeling where overfitting risk is elevated. Second, the superior uncertainty control of the quantum enhanced RF in the dry season supports the theoretical advantage of quantum feature spaces in capturing heavy tailed, nonlinear geochemical processes, highlighting the value of hybrid quantum classical architectures under extreme hydrological conditions. Third, the distinct seasonal uncertainty structures reinforce the necessity of seasonally stratified modeling frameworks, as a single unified model cannot adequately represent error behavior across contrasting hydrological regimes.</p>
</sec>
<sec id="Ch1.S4.SS4">
  <label>4.4</label><title>Performance analysis of hybrid quantum-classical model</title>
      <p id="d2e8327">Quantum machine learning captures complex nonlinear relationships through feature mapping in high dimensional Hilbert spaces, offering gains when data are scarce or highly skewed (Lamichhane and Rawat, 2025). When classical feature representation saturates, the <inline-formula><mml:math id="M562" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> feature mapping of parameterized quantum circuits may expose entangled nonlinear patterns that enhance discriminability, though such gains converge once virtual sample expansion is sufficient (Hong and Lopez, 2025). Here, quantum features were generated by analytically computing Pauli <inline-formula><mml:math id="M563" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> expectation values, circumventing hardware noise and making the quantum enhanced RF feasible for small sample environmental tasks (Gujju et al., 2024). The advantage is conditional rather than universal: gains are marginal in the wet season, where relationships are largely linear, but substantial in the dry season, where sparse data and extreme values make quantum encoding more robust to measurement noise (Ranga et al., 2024). Hybrid quantum classical modeling therefore adds value mainly in complex, information limited scenarios by expanding representational capacity rather than replacing classical logic, supporting the potential of quantum machine learning for small sample problems in earth sciences even when its absolute advantage remains modest.</p>
      <p id="d2e8344">In a comparable intensive agricultural setting on the North China Plain, Zhao et al. (2026) evaluated RF, XGBoost, and LightGBM on 157 groundwater samples from Handan City, with the best-performing LightGBM reaching a test <inline-formula><mml:math id="M564" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> of 0.753 (RMSE <inline-formula><mml:math id="M565" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 3.67 mg L<sup>−1</sup>). Similarly, Alam et al. (2025a) reported XGBoost performance below <inline-formula><mml:math id="M567" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.9 for nitrate prediction in the Yinchuan Region, and a comprehensive review concluded that most well-constrained groundwater nitrate ML studies achieve <inline-formula><mml:math id="M568" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> values between 0.8 and 0.9 (Haggerty et al., 2023). Against these benchmarks, the present framework performs favorably: with in-situ water quality inputs, the virtual-sample-augmented models attained <inline-formula><mml:math id="M569" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> 0.9622, 0.854, and 0.977 (RMSE <inline-formula><mml:math id="M570" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 3.03 mg L<sup>−1</sup>) in the normal, dry, and wet seasons, respectively; even with AEF embeddings alone, inputs carrying no direct hydrochemical information, the <inline-formula><mml:math id="M572" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> values (0.860, 0.674, and 0.784) remain comparable to or exceed purely geospatial models in the literature (Ransom et al., 2022; Karimanzira et al., 2023).</p>
      <p id="d2e8445">A legitimate concern regarding quantum machine learning is whether the derived quantum features bear any correspondence to actual hydrochemical processes, or whether they merely constitute an abstract mathematical augmentation. In the present architecture, this correspondence is direct and traceable. The <inline-formula><mml:math id="M573" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula>-feature map first applies a Hadamard gate followed by a single-qubit rotation <inline-formula><mml:math id="M574" display="inline"><mml:mrow><mml:mi>R</mml:mi><mml:mi>z</mml:mi></mml:mrow></mml:math></inline-formula>(2<inline-formula><mml:math id="M575" display="inline"><mml:mrow><mml:mi mathvariant="italic">ϕ</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>), so that the resulting Pauli-<inline-formula><mml:math id="M576" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> expectation value reduces analytically to <inline-formula><mml:math id="M577" display="inline"><mml:mrow><mml:mo>〈</mml:mo><mml:mi>Z</mml:mi><mml:mi>i</mml:mi><mml:mo>〉</mml:mo><mml:mo>=</mml:mo></mml:mrow></mml:math></inline-formula> cos(2<inline-formula><mml:math id="M578" display="inline"><mml:mrow><mml:mi mathvariant="italic">ϕ</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>), i.e., each quantum feature is a bounded, smooth, one-to-one nonlinear re-expression of an individual measured hydrochemical variable (Havlíček et al., 2019; Schuld and Killoran, 2019). Consequently, every quantum feature inherits the physical identity and units of its parent variable, and its contribution to the prediction can be unambiguously traced back to that variable through SHAP attribution. Physically, the cosine-type response introduced by the quantum encoding is hydrochemically meaningful: whereas classical decision trees partition samples using hard thresholds on raw concentrations, the quantum features provide smooth, saturation-like response curves that resemble the nonlinear equilibria governing nitrate behavior in groundwater, such as carbonate mineral saturation controlling Ca<sup>2+</sup>-HCO<inline-formula><mml:math id="M580" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> buffering, the convex conductivity-concentration relationship under evaporative concentration in the dry season, and the redox thresholds (DO and ORP) that delimit nitrification and denitrification regimes. Moreover, because supervised quantum models of this form are mathematically equivalent to kernel methods, the augmented feature space implicitly measures pairwise similarity among samples in a high-dimensional Hilbert space; in hydrochemical terms, this acts as a facies-similarity measure, consistent with the sample groupings independently observed in the Piper and Gibbs diagrams. The concatenated design [<inline-formula><mml:math id="M581" display="inline"><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mo>〈</mml:mo><mml:mi>Z</mml:mi><mml:mo>〉</mml:mo></mml:mrow></mml:math></inline-formula>] further guarantees that no raw information is discarded: the classical features preserve linear, threshold-based mechanisms, while the quantum features add nonlinear, periodic structure, so the model's mechanistic inferences remain anchored to measurable processes such as evaporative concentration, water-rock interaction, cation exchange, and seasonal dilution (Saberian et al., 2025). We nevertheless acknowledge that quantum features are derived transformations rather than independently measurable physical quantities; their mechanistic interpretation must therefore always proceed through the parent hydrochemical variable, as implemented here via SHAP-based attribution (Pérez-Salinas et al., 2020).</p>
</sec>
<sec id="Ch1.S4.SS5">
  <label>4.5</label><title>Limitations and future directions</title>
      <p id="d2e8570">Although the 66, 65, and 50 sampling points for the normal, dry, and wet seasons adequately covered the 3000 ha farm, discrete points cannot fully resolve fine scale nitrate heterogeneity driven by microtopography, soil texture transitions, and uneven fertilizer application (Li et al., 2025c). The seasonal snapshot design also sacrifices temporal resolution: intervals of four to six months may miss episodic recharge, short term leaching pulses after intense rainfall, and transient biogeochemical shifts during freeze thaw cycles characteristic of the monsoonal North China Plain (Nolte et al., 2025). The single year observation period further precludes assessment of interannual climatic variability and legacy nitrogen depletion, so predictions may be less reliable under extreme hydrological years or future precipitation non stationarity. Three systematic biases warrant acknowledgment: sampling through irrigation wells overrepresents shallow, actively exploited aquifer zones relative to deeper stagnant groundwater (Jia and Qian, 2025); MixSIAR apportionment relies on literature derived endmember signatures that may not capture local fractionation within the thick vadose zone (Jiang et al., 2026); and virtual sample generation may amplify biases inherent in the small original datasets, particularly the overrepresentation of high concentration outliers during the dry season (Tanner et al., 2022).</p>
      <p id="d2e8573">Although the hybrid quantum classical framework performed well in the Xiong'an New Area, its validity is intrinsically tied to the North China Plain context, and direct transfer to contrasting settings, such as karst aquifers, humid tropical climates, or fertilizer dominated nitrate regimes, would likely suffer domain shift. The single site design also precludes evaluation of transfer learning benefits that multi site pooling could provide (Clark and Jaffres, 2025; Farahani et al., 2025). The t-SNE/GMM/KNN methodology itself is portable, but model parameters require recalibration with local hydrochemical data. Future work should therefore combine leave one region out cross validation, domain adaptation techniques (adversarial training or feature alignment), and meta learning over multi site hydrochemical databases to separate universal nitrate transport principles from site specific behavior, enabling cross regional application without extensive new sampling (Zheng et al., 2025).</p>
      <p id="d2e8576">Models using only AEF embeddings achieve <inline-formula><mml:math id="M582" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> values roughly 10 % to 20 % lower than those using measured water quality parameters, because field parameters directly reflect the groundwater chemical state and nitrate transformation processes, whereas remote sensing semantics offer only indirect characterization (Alam et al., 2025b). Admittedly, predicting nitrate from field parameters has limited value if every parameter must be measured anyway; however, these parameters can be acquired on site within minutes using portable meters, whereas nitrate requires laboratory analysis, making the field parameter model practical for preliminary screening and rapid decision making when laboratory results are unavailable. After tenfold virtual expansion, the AEF model still attains <inline-formula><mml:math id="M583" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> above 0.67 (dry), 0.85 (normal), and 0.78 (wet season), supporting its use as a rapid, large scale screening tool, especially in unmonitored areas. Seasonal shifts in dominant embeddings (A05/A00, A08/A06, A02/A03) plausibly encode crop residue or soil organic matter, bare soil exposure, and vegetation or runoff potential, respectively, consistent with MixSIAR apportionment and Bayesian driving factors (Alvarez et al., 2025). Although causal inference remains indirect, the global coverage and annual updates of AEF make it a powerful supplement to, rather than substitute for, monitoring networks, and ground truth validation remains essential for linking remote sensing signals to subsurface water quality (Cai et al., 2025; Tollefson, 2025).</p>
</sec>
</sec>
<sec id="Ch1.S5" sec-type="conclusions">
  <label>5</label><title>Conclusion</title>
      <p id="d2e8611">With the clear goal of solving the practical problems of difficult seasonal prediction, high monitoring cost, and lack of targeted control basis for groundwater nitrate in intensive agricultural areas, this study develops an integrated prediction framework combining hybrid quantum-classical machine learning, advanced virtual sample generation (t-SNE–GMM–KNN), and remote sensing foundation model embedding (AlphaEarth Foundation, AEF). The framework is designed to systematically address three core challenges in predicting groundwater nitrate concentrations in agricultural areas across different hydrological seasons: small sample bias, seasonal heterogeneity, and input data scarcity.</p>
      <p id="d2e8614">Hydrological seasonality acts as the dominant controlling factor for the spatiotemporal variability of nitrates. Nitrate concentrations peak during the dry season (mean: 42.93 mg L<sup>−1</sup>), driven primarily by evaporative concentration and pollutant accumulation effects. In contrast, concentrations reach a minimum in the wet season (mean: 27.14 mg L<sup>−1</sup>) due to dilution by precipitation. The groundwater hydrochemical type is consistently Ca-Mg-HCO<inline-formula><mml:math id="M586" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">3</mml:mn><mml:mo>-</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> across all seasons, controlled predominantly by carbonate mineral dissolution. TDS, EC, and salinity remain consistently top-ranked across all seasons, with additional season-specific drivers including Mg<sup>2+</sup> and Na<sup>+</sup> (normal season), SO<inline-formula><mml:math id="M589" display="inline"><mml:mrow><mml:msubsup><mml:mi/><mml:mn mathvariant="normal">4</mml:mn><mml:mrow><mml:mn mathvariant="normal">2</mml:mn><mml:mo>-</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> (dry season), and Cl<sup>−</sup> (wet season). Stable hydrogen and oxygen isotope analysis reveals strong evaporative fractionation of groundwater. MixSIAR analysis quantitatively apportioned nitrate sources: domestic sewage and manure (DSM) contribute 74.1 %, soil organic nitrogen (SON) 20.9 %, while synthetic fertilizers (NHF <inline-formula><mml:math id="M591" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> NOF <inline-formula><mml:math id="M592" display="inline"><mml:mo>=</mml:mo></mml:math></inline-formula> 4.8 %) and atmospheric deposition (0.2 %) are negligible, strongly indicating that legacy nitrogen stored in the thick vadose zone, rather than in-season fertilizer leaching, sustains current pollution.</p>
      <p id="d2e8713">The proposed t-SNE-GMM-KNN virtual sample strategy effectively alleviates the bottleneck associated with small-sample modeling. By preserving the nonlinear manifold structure and multimodal distribution characteristics of the high-dimensional hydrochemical space, this method significantly enhances the model's ability to fit heavy-tailed distributions. Model performance improves significantly with virtual sample expansion. Using measured parameters as inputs, a 10-fold generation increased the coefficient of determination (<inline-formula><mml:math id="M593" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula>) for the dry season from 0.284 to <inline-formula><mml:math id="M594" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.85, while stabilizing it at <inline-formula><mml:math id="M595" display="inline"><mml:mi mathvariant="italic">&gt;</mml:mi></mml:math></inline-formula> 0.95 for the normal and wet seasons. This confirms that data sparsity is the fundamental constraint limiting performance. Although performance gains are limited with high-quality data, the quantum-enhanced Random Forest demonstrates superior stability compared to classical models in small-sample, highly skewed scenarios, validating the feasibility and value of quantum feature enhancement strategies in environmental small-sample learning. The overall prediction performance using measured hydrochemical parameters surpasses that of AEF remote sensing semantic embeddings (<inline-formula><mml:math id="M596" display="inline"><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:math></inline-formula> is approximately 10 %–20 % higher), as the former directly reflects subsurface nitrogen migration and transformation processes. Given that field-measurable water quality parameters can be rapidly acquired using portable instruments while nitrate determination requires laboratory analysis, the field-based model provides a practical tool for both diagnosing pollution drivers and enabling rapid on-site nitrate estimation. Following 10-fold virtual sample generation, the AEF model also achieves usable accuracy, with feature importance exhibiting seasonal shifts, validating its distinct role as a cost-effective solution for regional risk assessment in unmonitored areas.</p>
</sec>

      
      </body>
    <back><notes notes-type="codedataavailability"><title>Code and data availability</title>

      <p id="d2e8756">All code and data used in this study are available at <ext-link xlink:href="https://doi.org/10.5281/zenodo.22329075" ext-link-type="DOI">10.5281/zenodo.22329075</ext-link> (Xu, 2026).</p>
  </notes><app-group>
        <supplementary-material position="anchor"><p id="d2e8762">The supplement related to this article is available online at <inline-supplementary-material xlink:href="https://doi.org/10.5194/hess-30-5647-2026-supplement" xlink:title="pdf">https://doi.org/10.5194/hess-30-5647-2026-supplement</inline-supplementary-material>.</p></supplementary-material>
        </app-group><notes notes-type="authorcontribution"><title>Author contributions</title>

      <p id="d2e8771">JX: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing (original draft preparation), Writing (review and editing). XW: Validation, Formal analysis, Investigation, Resources, Visualization, Writing (original draft preparation). YY: Funding acquisition, Project administration, Supervision, Writing-review and editing. LY and XS: Resources, Supervision. YZ: Funding acquisition. CL: Investigation, Supervision.</p>
  </notes><notes notes-type="competinginterests"><title>Competing interests</title>

      <p id="d2e8777">The contact author has declared that none of the authors has any competing interests.</p>
  </notes><notes notes-type="disclaimer"><title>Disclaimer</title>

      <p id="d2e8783">Publisher's note: Copernicus Publications remains neutral with regard to jurisdictional claims made in the text, published maps, institutional affiliations, or any other geographical representation in this paper. The authors bear the ultimate responsibility for providing appropriate place names. Views expressed in the text are those of the authors and do not necessarily reflect the views of the publisher.</p>
  </notes><ack><title>Acknowledgements</title><p id="d2e8789">We sincerely thank the four anonymous reviewers and the handling Editor, Dr. Heng Dai, for their careful reading of our manuscript and for their insightful comments and constructive suggestions, which have substantially improved the quality and clarity of this work.</p></ack><notes notes-type="reviewstatement"><title>Review statement</title>

      <p id="d2e8794">This paper was edited by Heng Dai and reviewed by four anonymous referees.</p>
  </notes><notes notes-type="financialsupport"><title>Financial support</title>

      <p id="d2e8800">This work is supported by the National Natural Science Foundation of China (grant no. 41601037) and the Open Project Program of Engineering Research Center of Groundwater Pollution Control and Remediation, Ministry of Education of China (grant no. GW202212).</p>
  </notes><ref-list>
    <title>References</title>

      <ref id="bib1.bib1"><label>1</label><mixed-citation>Abderzak, M., Zeghmar, A., Leila, B., Aziz, M., Velibor, S., Lizny, J., Mohamed, K., Fernanda, H., and Shuraik, K.: Ensemble learning-driven optimization of coagulant dosing for drinking water treatment plants using a scalable framework for smart and sustainable process control, Environ. Res., 288, 123229, <ext-link xlink:href="https://doi.org/10.1016/j.envres.2025.123229" ext-link-type="DOI">10.1016/j.envres.2025.123229</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib2"><label>2</label><mixed-citation>Addy, J. W. G., MacLaren, C., and Lang, R.: A Bayesian approach to analyzing long-term agricultural experiments, Eur. J. Agron., 159, 127227, <ext-link xlink:href="https://doi.org/10.1016/j.eja.2024.127227" ext-link-type="DOI">10.1016/j.eja.2024.127227</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib3"><label>3</label><mixed-citation>Ahmed, M. A., Abdel Samie, S. G., and Badawy, H. A.: Factors controlling mechanisms of groundwater salinization and hydrogeochemical processes in the Quaternary aquifer of the Eastern Nile Delta, Egypt, Environ. Earth Sci., 68, 369–394, <ext-link xlink:href="https://doi.org/10.1007/s12665-012-1744-6" ext-link-type="DOI">10.1007/s12665-012-1744-6</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bib4"><label>4</label><mixed-citation>Alam, G. M. I., Arfin Tanim, S., Sarker, S. K., Watanobe, Y., Islam, R., Mridha, M. F., and Nur, K.: Deep learning model based prediction of vehicle CO<sub>2</sub> emissions with eXplainable AI integration for sustainable environment, Sci. Rep., 15, 3655, <ext-link xlink:href="https://doi.org/10.1038/s41598-025-87233-y" ext-link-type="DOI">10.1038/s41598-025-87233-y</ext-link>, 2025a.</mixed-citation></ref>
      <ref id="bib1.bib5"><label>5</label><mixed-citation>Alam, S. M. K., Li, P., Rahman, M., Fida, M., and Elumalai, V.: Key factors affecting groundwater nitrate levels in the Yinchuan Region, Northwest China: Research using the eXtreme Gradient Boosting (XGBoost) model with the SHapley Additive exPlanations (SHAP) method, Environ. Pollut., 364, 125336, <ext-link xlink:href="https://doi.org/10.1016/j.envpol.2024.125336" ext-link-type="DOI">10.1016/j.envpol.2024.125336</ext-link>, 2025b.</mixed-citation></ref>
      <ref id="bib1.bib6"><label>6</label><mixed-citation>Alvarez, C. I., Ulloa Vaca, C. A., and Echeverria Llumipanta, N. A.: Machine learning for urban air quality prediction using Google AlphaEarth Foundations satellite embeddings: A case study of Quito, Ecuador, Remote Sens., 17, 3472, <ext-link xlink:href="https://doi.org/10.3390/rs17203472" ext-link-type="DOI">10.3390/rs17203472</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib7"><label>7</label><mixed-citation>An, B., Zhang, Z., Ren, J., and Zhang, W.: Recurrent adversarial learning for geo-technical time-series augmentation: application to slope instability forecasting in open-pit mines, Environ. Earth Sci., 84, 559, <ext-link xlink:href="https://doi.org/10.1007/s12665-025-12566-w" ext-link-type="DOI">10.1007/s12665-025-12566-w</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib8"><label>8</label><mixed-citation>Anderson, G. J. and Lucas, D. D.: Machine learning predictions of a multiresolution climate model ensemble, Geophys. Res. Lett., 45, 4273–4280, <ext-link xlink:href="https://doi.org/10.1029/2018GL077049" ext-link-type="DOI">10.1029/2018GL077049</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib9"><label>9</label><mixed-citation>Balogun, E., Rajagopal, R., and Majumdar, A.: TemperatureGAN: generative modeling of regional atmospheric temperatures, Environ. Data Sci., 3, e21, <ext-link xlink:href="https://doi.org/10.1017/eds.2024.21" ext-link-type="DOI">10.1017/eds.2024.21</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib10"><label>10</label><mixed-citation>Bigler, M. C., Brusseau, M. L., Guo, B., Jones, S. L., Pritchard, J. C., Higgins, C. P., and Hatton, J.: High-resolution depth-discrete analysis of PFAS distribution and leaching for a vadose-zone source at an AFFF-Impacted site, Environ. Sci. Technol., 58, 9863–9874, <ext-link xlink:href="https://doi.org/10.1021/acs.est.4c01615" ext-link-type="DOI">10.1021/acs.est.4c01615</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib11"><label>11</label><mixed-citation>Cai, C., Zhao, H., Zhang, H., Wang, C., Wang, Z., Liu, M., Chen, J., and Zhang, H.: Timely assessment of maize lodging severity with limited samples using multi-temporal Sentinel-1 and Sentinel-2 data across large spatial extents, Comput. Electron. Agr., 237, 110671, <ext-link xlink:href="https://doi.org/10.1016/j.compag.2025.110671" ext-link-type="DOI">10.1016/j.compag.2025.110671</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib12"><label>12</label><mixed-citation>Charizanos, G. and Demirhan, H.: Bayesian prediction of wildfire event probability using normalized difference vegetation index data from an Australian forest, Ecol. Inform., 73, 101899, <ext-link xlink:href="https://doi.org/10.1016/j.ecoinf.2022.101899" ext-link-type="DOI">10.1016/j.ecoinf.2022.101899</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib13"><label>13</label><mixed-citation>Chen, Q., Yang, H., Cui, R., Hu, W., Wang, C., Chen, A., and Zhang, D.: Shallow groundwater table fluctuations: A driving force for accelerating the migration and transformation of phosphorus in cropland soil, Water Res., 275, 123209, <ext-link xlink:href="https://doi.org/10.1016/j.watres.2025.123209" ext-link-type="DOI">10.1016/j.watres.2025.123209</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib14"><label>14</label><mixed-citation>Clark, S. R. and Jaffres, J. B. D.: Associations between deep learning runoff predictions and hydrogeological conditions in Australia, J. Hydrol., 651, 132569, <ext-link xlink:href="https://doi.org/10.1016/j.jhydrol.2024.132569" ext-link-type="DOI">10.1016/j.jhydrol.2024.132569</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib15"><label>15</label><mixed-citation>Cowlessur, H., Alpcan, T., Thapa, C., Camtepe, S., and Kundu, N. K.: A Qubit-Efficient Hybrid Quantum Encoding Mechanism for Quantum Machine Learning, arXiv [preprint], <ext-link xlink:href="https://doi.org/10.48550/arXiv.2506.19275" ext-link-type="DOI">10.48550/arXiv.2506.19275</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib16"><label>16</label><mixed-citation>Dega, S., Dietrich, P., Schroen, M., and Paasche, H.: Probabilistic prediction by means of the propagation of response variable uncertainty through a Monte Carlo approach in regression random forest: Application to soil moisture regionalization, Front. Environ. Sci., 11, 1009191, <ext-link xlink:href="https://doi.org/10.3389/fenvs.2023.1009191" ext-link-type="DOI">10.3389/fenvs.2023.1009191</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib17"><label>17</label><mixed-citation>Deng, Y., Ye, X., and Du, X.: Predictive modeling and analysis of key drivers of groundwater nitrate pollution based on machine learning, J. Hydrol., 624, 129934, <ext-link xlink:href="https://doi.org/10.1016/j.jhydrol.2023.129934" ext-link-type="DOI">10.1016/j.jhydrol.2023.129934</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib18"><label>18</label><mixed-citation>Di Santo, D., He, C., Chen, F., and Giovannini, L.: ML-AMPSIT: Machine Learning-based Automated Multi-method Parameter Sensitivity and Importance analysis Tool, Geosci. Model Dev., 18, 433–459, <ext-link xlink:href="https://doi.org/10.5194/gmd-18-433-2025" ext-link-type="DOI">10.5194/gmd-18-433-2025</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib19"><label>19</label><mixed-citation>Farahani, M. A., Wood, A. W., Tang, G., and Mizukami, N.: Calibrating a large-domain land/hydrology process model in the age of AI: the SUMMA CAMELS emulator experiments, Hydrol. Earth Syst. Sci., 29, 4515–4537, <ext-link xlink:href="https://doi.org/10.5194/hess-29-4515-2025" ext-link-type="DOI">10.5194/hess-29-4515-2025</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib20"><label>20</label><mixed-citation>Farnia, F., Wang, W. W., Das, S., and Jadbabaie, A.: Gat–gmm: Generative adversarial training for gaussian mixture models, SIAM J. Math. Data Sci., 5, 122–146, <ext-link xlink:href="https://doi.org/10.1137/21M1445831" ext-link-type="DOI">10.1137/21M1445831</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib21"><label>21</label><mixed-citation>Feng, D., Liu, J., Lawson, K., and Shen, C.: Differentiable, learnable, regionalized process-based models with multiphysical outputs can approach state-of-the-art hydrologic prediction accuracy, Water Resour. Res., 58, e2022WR032404, <ext-link xlink:href="https://doi.org/10.1029/2022WR032404" ext-link-type="DOI">10.1029/2022WR032404</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib22"><label>22</label><mixed-citation>Feng, W., Wang, S., Tan, K., Ma, L., and Hu, C.: Simulation of spatial and temporal variation of nitrate leaching in the vadose zone of alluvial regions on a large regional scale, Sci. Total Environ., 916, 170114, <ext-link xlink:href="https://doi.org/10.1016/j.scitotenv.2024.170114" ext-link-type="DOI">10.1016/j.scitotenv.2024.170114</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib23"><label>23</label><mixed-citation>Gao, H., Yang, L., Song, X., Guo, M., Li, B., and Cui, X.: Sources and hydrogeochemical processes of groundwater under multiple water source recharge condition, Sci. Total Environ., 903, 166660, <ext-link xlink:href="https://doi.org/10.1016/j.scitotenv.2023.166660" ext-link-type="DOI">10.1016/j.scitotenv.2023.166660</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib24"><label>24</label><mixed-citation>Ghodba, A., Richelle, A., McCready, C., Ricardez-Sandoval, L., and Budman, H.: A novel dynamic flux balance analysis for modeling CHO cell fed-batch cultures with pH and temperature shifts, J. Biotechnol., 408, 61–71, <ext-link xlink:href="https://doi.org/10.1016/j.jbiotec.2025.08.010" ext-link-type="DOI">10.1016/j.jbiotec.2025.08.010</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib25"><label>25</label><mixed-citation>Gu, B. J., Ge, Y., Chang, S. X., Luo, W. D., and Chang, J.: Nitrate in groundwater of China: Sources and driving forces, Global Environ. Chang., 23, 1112–1121, <ext-link xlink:href="https://doi.org/10.1016/j.gloenvcha.2013.05.004" ext-link-type="DOI">10.1016/j.gloenvcha.2013.05.004</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bib26"><label>26</label><mixed-citation>Gujju, Y., Matsuo, A., and Raymond, R.: Quantum machine learning on near-term quantum devices: Current state of supervised and unsupervised techniques for real-world applications, Phys. Rev. Appl., 21, 067001, <ext-link xlink:href="https://doi.org/10.1103/PhysRevApplied.21.067001" ext-link-type="DOI">10.1103/PhysRevApplied.21.067001</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib27"><label>27</label><mixed-citation>Gul, N., Khan, A. U., Ullah, B., Khan, B. N., Almalki, H. M., Banga, A. S., and Kumar, K.: Optimizing LSTM for sediment load prediction in the Swat River Basin, Pakistan: Evaluation of optimizers and activation functions, Phys. Chem. Earth, 140, 104019, <ext-link xlink:href="https://doi.org/10.1016/j.pce.2025.104019" ext-link-type="DOI">10.1016/j.pce.2025.104019</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib28"><label>28</label><mixed-citation>Haggerty, R., Sun, J., Yu, H., and Li, Y.: Application of machine learning in groundwater quality modeling – A comprehensive review, Water Res., 233, 119745, <ext-link xlink:href="https://doi.org/10.1016/j.watres.2023.119745" ext-link-type="DOI">10.1016/j.watres.2023.119745</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib29"><label>29</label><mixed-citation>Han, D. M., Currell, M. J., and Cao, G. L.: Deep challenges for China's war on water pollution, Environ. Pollut., 218, 1222–1233, <ext-link xlink:href="https://doi.org/10.1016/j.envpol.2016.08.078" ext-link-type="DOI">10.1016/j.envpol.2016.08.078</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bib30"><label>30</label><mixed-citation>Havlíček, V., Córcoles, A. D., Temme, K., Harrow, A., Kandala, A., Chow, J. M., and Gambetta, J. M.: Supervised learning with quantum-enhanced feature spaces, Nature, 567, 209–212, <ext-link xlink:href="https://doi.org/10.1038/s41586-019-0980-2" ext-link-type="DOI">10.1038/s41586-019-0980-2</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bib31"><label>31</label><mixed-citation>Hollmann, N., Müller, S., Purucker, L., Krishnakumar, A., Körfer, M., Hoo, S. B., Schirrmeister, R. T., and Hutter, F.: Accurate predictions on small data with a tabular foundation model, Nature, 637, 319–326, <ext-link xlink:href="https://doi.org/10.1038/s41586-024-08328-6" ext-link-type="DOI">10.1038/s41586-024-08328-6</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib32"><label>32</label><mixed-citation>Hong, Y. Y. and Lopez, D. J. D.: A Review on Quantum Machine Learning in Applied Systems and Engineering, IEEE Access, 13, 144607–144631, <ext-link xlink:href="https://doi.org/10.1109/ACCESS.2025.3599147" ext-link-type="DOI">10.1109/ACCESS.2025.3599147</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib33"><label>33</label><mixed-citation>Hou, X., Peng, L., Zhang, Y., Zhang, Y., Wang, Y., Feng, W., and Yang, H.: A Data-Driven Method for Determining DRASTIC Weights to Assess Groundwater Vulnerability to Nitrate: Application in the Lake Baiyangdian Watershed, North China Plain, Appl. Sci., 15, 2866, <ext-link xlink:href="https://doi.org/10.3390/app15052866" ext-link-type="DOI">10.3390/app15052866</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib34"><label>34</label><mixed-citation>Huang, P. and Chen, J.: Recharge sources and hydrogeochemical evolution of groundwater in the coal-mining district of Jiaozuo, China, Hydrogeol. J., 20, 739–754, <ext-link xlink:href="https://doi.org/10.1007/s10040-012-0836-4" ext-link-type="DOI">10.1007/s10040-012-0836-4</ext-link>, 2012.</mixed-citation></ref>
      <ref id="bib1.bib35"><label>35</label><mixed-citation>Huang, C., Tong, J., and Ye, M.: Global sensitivity analysis for a prediction model of soil solute transfer into surface runoff, J. Hydrol., 599, 126342, <ext-link xlink:href="https://doi.org/10.1016/j.jhydrol.2021.126342" ext-link-type="DOI">10.1016/j.jhydrol.2021.126342</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bib36"><label>36</label><mixed-citation>Islam, M. T., Zhou, Z., Ren, H., Khuzani, M. B., Kapp, D., Zou, J., Tian, L., Liao, J. C., and Xing, L.: Revealing hidden patterns in deep neural network feature space continuum via manifold learning, Nat. Commun., 14, 8506, <ext-link xlink:href="https://doi.org/10.1038/s41467-023-43958-w" ext-link-type="DOI">10.1038/s41467-023-43958-w</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib37"><label>37</label><mixed-citation>Jamshidi, E. J., Yusup, Y., Kayode, J. S., and Kamaruddin, M. A.: Detecting outliers in a univariate time series dataset using unsupervised combined statistical methods: A case study on surface water temperature, Ecol. Inform., 69, 101672, <ext-link xlink:href="https://doi.org/10.1016/j.ecoinf.2022.101672" ext-link-type="DOI">10.1016/j.ecoinf.2022.101672</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib38"><label>38</label><mixed-citation>Jia, H. and Qian, H.: Groundwater nitrate response to hydrogeological conditions and socioeconomic load in an agriculture dominated area, Sci. Rep., 15, 1315, <ext-link xlink:href="https://doi.org/10.1038/s41598-024-84318-y" ext-link-type="DOI">10.1038/s41598-024-84318-y</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib39"><label>39</label><mixed-citation>Jia, B., Zhou, J., Tang, Z., Xu, Z., Chen, X., and Fang, W.: Effective stochastic streamflow simulation method based on Gaussian mixture model, J. Hydrol., 605, 127366, <ext-link xlink:href="https://doi.org/10.1016/j.jhydrol.2021.127366" ext-link-type="DOI">10.1016/j.jhydrol.2021.127366</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib40"><label>40</label><mixed-citation>Jiang, H., Geng, D., Fu, J., Liu, M., Qi, Y., Min, L., Wang, S., and Shen, Y.: Quantifying nitrate dynamics in deep vadose zone under irrigated farmland in the North China Plain: Insights from continuous in-situ monitoring, Agr. Ecosyst. Environ., 396, 109956, <ext-link xlink:href="https://doi.org/10.1016/j.agee.2025.109956" ext-link-type="DOI">10.1016/j.agee.2025.109956</ext-link>, 2026.</mixed-citation></ref>
      <ref id="bib1.bib41"><label>41</label><mixed-citation>Kalinin, A. A., Arevalo, J., Serrano, E., Vulliard, L., Tsang, H., Bornholdt, M., Muñoz, A. F., Sivagurunathan, S., Rajwa, B., Carpenter, A. E., Way, G. P., and Singh, S.: A versatile information retrieval framework for evaluating profile strength and similarity, Nat. Commun., 16, 5181, <ext-link xlink:href="https://doi.org/10.1038/s41467-025-60306-2" ext-link-type="DOI">10.1038/s41467-025-60306-2</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib42"><label>42</label><mixed-citation>Karimanzira, D., Weis, J., Wunsch, A., Ritzau, L., Liesch, T., and Ohmer, M.: Application of machine learning and deep neural networks for spatial prediction of groundwater nitrate concentration to improve land use management practices, Front. Water, 5, 1193142, <ext-link xlink:href="https://doi.org/10.3389/frwa.2023.1193142" ext-link-type="DOI">10.3389/frwa.2023.1193142</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib43"><label>43</label><mixed-citation>Kaur, H., Bansod, B. S., Khungar, P., and Dhawan, C.: Combining clustering and ensemble learning for groundwater quality monitoring: a data-driven framework for sustainable water management, Environ. Sci. Pollut. R., 32, 13862–13903, <ext-link xlink:href="https://doi.org/10.1007/s11356-025-36477-2" ext-link-type="DOI">10.1007/s11356-025-36477-2</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib44"><label>44</label><mixed-citation>Khalil, M., Zhang, C., Ye, Z., and Zhang, P.: PegasosQSVM: A Quantum Machine Learning Approach for Accurate Fake News Detection, Appl. Artif. Intell., 39, 2457207, <ext-link xlink:href="https://doi.org/10.1080/08839514.2025.2457207" ext-link-type="DOI">10.1080/08839514.2025.2457207</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib45"><label>45</label><mixed-citation>Kobak, D. and Berens, P.: The art of using t-SNE for single-cell transcriptomics, Nat. Commun., 10, 5416, <ext-link xlink:href="https://doi.org/10.1038/s41467-019-13056-x" ext-link-type="DOI">10.1038/s41467-019-13056-x</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bib46"><label>46</label><mixed-citation>Lamichhane, P. and Rawat, D. B.: Quantum Machine Learning: Recent Advances, Challenges and Perspectives, IEEE Access, 19, 94057–94105, <ext-link xlink:href="https://doi.org/10.1109/ACCESS.2025.3573244" ext-link-type="DOI">10.1109/ACCESS.2025.3573244</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib47"><label>47</label><mixed-citation>Li, X. and Yu, L.: Mapping complex cropping patterns in China (2018–2021) at 10 m resolution: a data-driven framework based on multi-product integration and Google satellite embedding, Earth Syst. Sci. Data, 18, 5601–5626, <ext-link xlink:href="https://doi.org/10.5194/essd-18-5601-2026" ext-link-type="DOI">10.5194/essd-18-5601-2026</ext-link>, 2026.</mixed-citation></ref>
      <ref id="bib1.bib48"><label>48</label><mixed-citation>Li, J., Zhu, D., Zhang, S., Yang, G., Zhao, Y., Zhou, C., Lin, Y., and Zou, S.: Application of the hydrochemistry, stable isotopes and MixSIAR model to identify nitrate sources and transformations in surface water and groundwater of an intensive agricultural karst wetland in Guilin, China, Ecotox. Environ. Safe, 231, 113205, <ext-link xlink:href="https://doi.org/10.1016/j.ecoenv.2022.113205" ext-link-type="DOI">10.1016/j.ecoenv.2022.113205</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib49"><label>49</label><mixed-citation>Li, X., Wang, Y., Xue, B., A, Y., Zhang, X., and Wang, G.: Attribution of runoff and hydrological drought changes in an ecologically vulnerable basin in semi-arid regions of China, Hydrol. Process., 37, e15003, <ext-link xlink:href="https://doi.org/10.1002/hyp.15003" ext-link-type="DOI">10.1002/hyp.15003</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib50"><label>50</label><mixed-citation>Li, R., Feng, K., An, T., Cheng, P., Wei, L., Zhao, Z., Xu, X., and Zhu, L.: Enhanced insights into effluent prediction in wastewater treatment plants: Comprehensive deep learning model explanation based on shap, ACS ES&amp;T Water, 4, 1904–1915, <ext-link xlink:href="https://doi.org/10.1021/acsestwater.4c00040" ext-link-type="DOI">10.1021/acsestwater.4c00040</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib51"><label>51</label><mixed-citation>Li, J., Liu, G., and Shen, Z.: Integrating spatiotemporal variability of non-point source pollution and best management practice efficiency to improve adaptive watershed management, Water Res., 284, 124042, <ext-link xlink:href="https://doi.org/10.1016/j.watres.2025.124042" ext-link-type="DOI">10.1016/j.watres.2025.124042</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib52"><label>52</label><mixed-citation>Li, S., Zheng, T., Farchi, A., Bocquet, M., and Gentine, P.: Probabilistic data assimilation for ensemble distribution projections with generative machine learning: A Lorenz'96 proof-of-concept, Geophys. Res. Lett., 52, e2024GL112523, <ext-link xlink:href="https://doi.org/10.1029/2024GL112523" ext-link-type="DOI">10.1029/2024GL112523</ext-link>, 2025a.</mixed-citation></ref>
      <ref id="bib1.bib53"><label>53</label><mixed-citation>Li, X., Liu, M., Min, L., and Shen, Y.: Nitrogen transport and transformation processes in the typical deep vadose zone in the central North China Plain, Chinese Journal of Eco-Agriculture, 33, 2359–2370, <ext-link xlink:href="https://doi.org/10.12357/cjea.20250148" ext-link-type="DOI">10.12357/cjea.20250148</ext-link>, 2025b.</mixed-citation></ref>
      <ref id="bib1.bib54"><label>54</label><mixed-citation>Li, X., Zhou, J., Zhou, W., Mao, L., Wang, C., Hao, Y., and Bian, P.: Hydrochemical Characteristics of Shallow Groundwater and Analysis of Vegetation Water Sources in the Ulan Buh Desert, Water, 17, 3058, <ext-link xlink:href="https://doi.org/10.3390/w17213058" ext-link-type="DOI">10.3390/w17213058</ext-link>, 2025c.</mixed-citation></ref>
      <ref id="bib1.bib55"><label>55</label><mixed-citation>Li, Y., Jiang, Z., Wu, S., Gao, J., and Sharma, A.: Systematic evaluation and correction of extreme water level in global storm surge simulation, Geophys. Res. Lett., 53, e2025GL119793, <ext-link xlink:href="https://doi.org/10.1029/2025GL119793" ext-link-type="DOI">10.1029/2025GL119793</ext-link>, 2026.</mixed-citation></ref>
      <ref id="bib1.bib56"><label>56</label><mixed-citation>Liao, H., Wang, D. S., Sitdikov, I., Salcedo, C., Seif, A., and Minev, Z. K.: Machine learning for practical quantum error mitigation, Nat. Mach. Intell., 6, 1478–1486, <ext-link xlink:href="https://doi.org/10.1038/s42256-024-00927-2" ext-link-type="DOI">10.1038/s42256-024-00927-2</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib57"><label>57</label><mixed-citation>Liu, H., Yang, J., Ye, M., James, S. C., Tang, Z., Dong, J., and Xing, T.: Using t-distributed Stochastic Neighbor Embedding (t-SNE) for cluster analysis and spatial zone delineation of groundwater geochemistry data, J. Hydrol., 597, 126146, <ext-link xlink:href="https://doi.org/10.1016/j.jhydrol.2021.126146" ext-link-type="DOI">10.1016/j.jhydrol.2021.126146</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bib58"><label>58</label><mixed-citation>Liu, M., Min, L., Wu, L., Pei, H., and Shen, Y.: Evaluating nitrate transport and accumulation in the deep vadose zone of the intensive agricultural region, North China Plain, Sci. Total Environ., 825, 153894, <ext-link xlink:href="https://doi.org/10.1016/j.scitotenv.2022.153894" ext-link-type="DOI">10.1016/j.scitotenv.2022.153894</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib59"><label>59</label><mixed-citation>Liu, S., Hao, Y., Wang, H., Zheng, X., Yu, X., Meng, X., Qiu, Y., Li, S., and Zheng, T.: Bidirectional potential effects of DON transformation in vadose zones on groundwater nitrate contamination: Different contributions to nitrification and denitrification, J. Hazard. Mater., 448, 130976, <ext-link xlink:href="https://doi.org/10.1016/j.jhazmat.2023.130976" ext-link-type="DOI">10.1016/j.jhazmat.2023.130976</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib60"><label>60</label><mixed-citation>Liu, M., Geng, D., Wu, L., Min, L., Wang, S., and Shen, Y.: The impact of agricultural land use change on water and nitrate fluxes in the deep vadose zone, the North China Plain, J. Hydrol.-Reg. Stud., 62, 102914, <ext-link xlink:href="https://doi.org/10.1016/j.ejrh.2025.102914" ext-link-type="DOI">10.1016/j.ejrh.2025.102914</ext-link>, 2025a.</mixed-citation></ref>
      <ref id="bib1.bib61"><label>61</label><mixed-citation>Liu, N., Chen, M., Gao, D., Wu, Y., and Wang, X.: Identification of hydrogeochemical processes in shallow groundwater using multivariate statistical analysis and inverse geochemical modeling, Environ. Monit. Assess., 197, 135, <ext-link xlink:href="https://doi.org/10.1007/s10661-024-13528-8" ext-link-type="DOI">10.1007/s10661-024-13528-8</ext-link>, 2025b.</mixed-citation></ref>
      <ref id="bib1.bib62"><label>62</label><mixed-citation>Liu, X., Yue, F. J., Li, L., Zhou, F., Wen, H., Yan, Z., Wang, L., Wong, W. W., Liu, C. Q., and Li, S. L.: Chronic nitrogen legacy in the aquifers of China, Commun. Earth Environ., 6, 58, <ext-link xlink:href="https://doi.org/10.1038/s43247-025-02016-7" ext-link-type="DOI">10.1038/s43247-025-02016-7</ext-link>, 2025c.</mixed-citation></ref>
      <ref id="bib1.bib63"><label>63</label><mixed-citation>Luo, Y., Yan, J., McClure, S. C., and Li, F.: Socioeconomic and environmental factors of poverty in China using geographically weighted random forest regression model, Environ. Sci. Pollut. R., 29, 33205–33217, <ext-link xlink:href="https://doi.org/10.1007/s11356-021-17513-3" ext-link-type="DOI">10.1007/s11356-021-17513-3</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib64"><label>64</label><mixed-citation>Ma, Z., Ye, C., Lu, C., Wei, Q., Zhu, R., Xie, X., Zhong, S., Chu, W., and Xu, Z.: A Robust Gaussian Process Paradigm for Predictive Modeling on Small Data sets in Environmental Science: A Case Study in Ballasted Flocculation, Environ. Sci. Technol., 60, 748–759, <ext-link xlink:href="https://doi.org/10.1021/acs.est.5c12617" ext-link-type="DOI">10.1021/acs.est.5c12617</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib65"><label>65</label><mixed-citation>Mao, H., Wang, G., Liao, F., Shi, Z., Zhang, H., Chen, X., Qiao, Z., Li, B., and Bai, Y.: Spatial variability of source contributions to nitrate in regional groundwater based on the positive matrix factorization and Bayesian model, J. Hazard. Mater., 445, 130569, <ext-link xlink:href="https://doi.org/10.1016/j.jhazmat.2022.130569" ext-link-type="DOI">10.1016/j.jhazmat.2022.130569</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib66"><label>66</label><mixed-citation>Merabet, K., Di Nunno, F., Granata, F., Kim, S., Adnan, R. M., Heddam, S., Kisi, O., and Zounemat-Kermani, M.: Predicting water quality variables using gradient boosting machine: global versus local explainability using SHapley Additive Explanations (SHAP), Earth Sci. Inform., 18, 298, <ext-link xlink:href="https://doi.org/10.1007/s12145-025-01796-y" ext-link-type="DOI">10.1007/s12145-025-01796-y</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib67"><label>67</label><mixed-citation>Miao, P., Zhu, X., Zhang, S., Li, W., Zhou, J., and Chen, Z.: Long-term high nitrogen surplus in intensive apple-planting regions results in huge legacy nitrogen in the vadose zone: Potential or real risk to the groundwater, Agr. Ecosyst. Environ., 357, 108682, <ext-link xlink:href="https://doi.org/10.1016/j.agee.2023.108682" ext-link-type="DOI">10.1016/j.agee.2023.108682</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib68"><label>68</label><mixed-citation>Michalek, A. T., Villarini, G., Kim, T., Quintero, F., Krajewski, W., and Scoccimarro, E.: Evaluation of CMIP6 HighResMIP for hydrologic modeling of annual maximum discharge in Iowa, Water Resour. Res., 59, e2022WR034166, <ext-link xlink:href="https://doi.org/10.1029/2022WR034166" ext-link-type="DOI">10.1029/2022WR034166</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib69"><label>69</label><mixed-citation>Naresh, V. S. and Reddi, S.: Quantum-enhanced predictive analytics in healthcare: benchmarking QSVM and QNN on medical datasets, Measurement, 258, 119099, <ext-link xlink:href="https://doi.org/10.1016/j.measurement.2025.119099" ext-link-type="DOI">10.1016/j.measurement.2025.119099</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib70"><label>70</label><mixed-citation>Niu, H., McCallum, G. B., Chang, A. B., Khan, K., and Azam, S.: Exploring unsupervised feature extraction algorithms: tackling high dimensionality in small datasets, Sci. Rep., 15, 21973, <ext-link xlink:href="https://doi.org/10.1038/s41598-025-07725-9" ext-link-type="DOI">10.1038/s41598-025-07725-9</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib71"><label>71</label><mixed-citation>Nolte, A., Heudorfer, B., Bender, S., and Hartmann, J.: Multi-site deep learning for groundwater level prediction across global datasets: toward scalable applications under data scarcity, J. Hydroinform., 27, 1632–1651, <ext-link xlink:href="https://doi.org/10.2166/hydro.2025.095" ext-link-type="DOI">10.2166/hydro.2025.095</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib72"><label>72</label><mixed-citation>Oliveira Santos, V., Costa Rocha, P. A., Thé, J. V. G., and Gharabaghi, B.: Optimizing the Architecture of a Quantum–Classical Hybrid Machine Learning Model for Forecasting Ozone Concentrations: Air Quality Management Tool for Houston, Texas, Atmosphere, 16, 255, <ext-link xlink:href="https://doi.org/10.3390/atmos16030255" ext-link-type="DOI">10.3390/atmos16030255</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib73"><label>73</label><mixed-citation>Peng, D., Gui, Z., Wei, W., Li, F., Gui, J., Wu, H., and Gong, J.: Sampling-enabled scalable manifold learning unveils the discriminative cluster structure of high-dimensional data, Nat. Mach. Intell., 7, 1669–1684, <ext-link xlink:href="https://doi.org/10.1038/s42256-025-01112-9" ext-link-type="DOI">10.1038/s42256-025-01112-9</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib74"><label>74</label><mixed-citation>Pérez-Salinas, A., Cervera-Lierta, A., Gil-Fuster, E., and Latorre, J.: Data re-uploading for a universal quantum classifier, Quantum, 4, 226, <ext-link xlink:href="https://doi.org/10.22331/q-2020-02-06-226" ext-link-type="DOI">10.22331/q-2020-02-06-226</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bib75"><label>75</label><mixed-citation>Ranga, D., Rana, A., Prajapat, S., Kumar, P., Kumar, K., and Vasilakos, A. V.: Quantum machine learning: Exploring the role of data encoding techniques, challenges, and future directions, Mathematics, 12, 3318, <ext-link xlink:href="https://doi.org/10.3390/math12213318" ext-link-type="DOI">10.3390/math12213318</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib76"><label>76</label><mixed-citation>Ransom, K. M., Nolan, B. T., Stackelberg, P. E., Belitz, K., and Fram, M. S.: Machine learning predictions of nitrate in groundwater used for drinking supply in the conterminous United States, Sci. Total Environ., 807, 151065, <ext-link xlink:href="https://doi.org/10.1016/j.scitotenv.2021.151065" ext-link-type="DOI">10.1016/j.scitotenv.2021.151065</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib77"><label>77</label><mixed-citation>Ren, M., Sun, W., and Chen, S.: Combining machine learning models through multiple data division methods for PM2.5 forecasting in Northern Xinjiang, China, Environ. Monit. Assess., 193, 476, <ext-link xlink:href="https://doi.org/10.1007/s10661-021-09233-5" ext-link-type="DOI">10.1007/s10661-021-09233-5</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bib78"><label>78</label><mixed-citation>Saberian, M., Zafarmomen, N., Neupane, A., Panthi, K., and Samadi, V.: HydroQuantum: A new quantum-driven Python package for hydrological simulation, Environ. Modell. Softw., 195, 106736, <ext-link xlink:href="https://doi.org/10.1016/j.envsoft.2025.106736" ext-link-type="DOI">10.1016/j.envsoft.2025.106736</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib79"><label>79</label><mixed-citation>Saha, G. K., Rahmani, F., Shen, C., Li, L., and Cibin, R.: A deep learning-based novel approach to generate continuous daily stream nitrate concentration for nitrate data-sparse watersheds, Sci. Total Environ., 878, 162930, <ext-link xlink:href="https://doi.org/10.1016/j.scitotenv.2023.162930" ext-link-type="DOI">10.1016/j.scitotenv.2023.162930</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib80"><label>80</label><mixed-citation>Schuld, M. and Killoran, N.: Quantum machine learning in feature Hilbert spaces, Phys. Rev. Lett., 122, 040504, <ext-link xlink:href="https://doi.org/10.1103/PhysRevLett.122.040504" ext-link-type="DOI">10.1103/PhysRevLett.122.040504</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bib81"><label>81</label><mixed-citation>Sebestyen, S. D., Shanley, J. B., Boyer, E. W., Kendall, C., and Doctor, D. H.: Coupled hydrological and biogeochemical processes controlling variability of nitrogen species in streamflow during autumn in an upland forest, Water Resour. Res., 50, 1569–1591, <ext-link xlink:href="https://doi.org/10.1002/2013WR013670" ext-link-type="DOI">10.1002/2013WR013670</ext-link>, 2014.</mixed-citation></ref>
      <ref id="bib1.bib82"><label>82</label><mixed-citation>Silva, R. and Melo-Pinto, P.: t-SNE: A study on reducing the dimensionality of hyperspectral data for the regression problem of estimating oenological parameters, Artif. Intell. Agric., 7, 58–68, <ext-link xlink:href="https://doi.org/10.1016/j.aiia.2023.02.003" ext-link-type="DOI">10.1016/j.aiia.2023.02.003</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib83"><label>83</label><mixed-citation>Stock, B. C., Jackson, A. L., Ward, E. J., Parnell, A. C., Phillips, D. L., and Semmens, B. X.: Analyzing mixing systems using a new generation of Bayesian tracer mixing models, PeerJ, 6, e5096, <ext-link xlink:href="https://doi.org/10.7717/peerj.5096" ext-link-type="DOI">10.7717/peerj.5096</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib84"><label>84</label><mixed-citation>Su, Y., Wu, Z., Zheng, X., Qiu, Y., Ma, Z., Ren, Y., and Bai, Y.: Harmonizing remote sensing and ground data for forest aboveground biomass estimation, Ecol. Inform., 86, 103002, <ext-link xlink:href="https://doi.org/10.1016/j.ecoinf.2025.103002" ext-link-type="DOI">10.1016/j.ecoinf.2025.103002</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib85"><label>85</label><mixed-citation>Sun, H., Zheng, W., Wang, S., Ma, L., Min, L., and Shen, Y.: Variation of nitrate sources affected by precipitation with different intensities in groundwater in the piedmont plain area of alluvial-pluvial fan, J. Environ. Manage., 367, 121885, <ext-link xlink:href="https://doi.org/10.1016/j.jenvman.2024.121885" ext-link-type="DOI">10.1016/j.jenvman.2024.121885</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib86"><label>86</label><mixed-citation>Tang, W. and Carey, S. K.: Classifying annual daily hydrographs in Western North America using t-distributed stochastic neighbour embedding, Hydrol. Process., 36, e14473, <ext-link xlink:href="https://doi.org/10.1002/hyp.14473" ext-link-type="DOI">10.1002/hyp.14473</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib87"><label>87</label><mixed-citation>Tanner, K. B., Cardall, A. C., and Williams, G. P.: A spatial long-term trend analysis of estimated chlorophyll-a concentrations in Utah Lake using Earth observation data, Remote Sens., 14, 3664, <ext-link xlink:href="https://doi.org/10.3390/rs14153664" ext-link-type="DOI">10.3390/rs14153664</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib88"><label>88</label><mixed-citation>Thunyawatcharakul, P., Cho, K. H., and Chotpantarat, S.: Predicting Arsenic Speciation in Coastal Aquifers Using Machine Learning: A Case Study of the Chonburi and Rayong Groundwater Basins, Thailand, ACS ES&amp;T Water, 5, 5011–5024, <ext-link xlink:href="https://doi.org/10.1021/acsestwater.4c01082" ext-link-type="DOI">10.1021/acsestwater.4c01082</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib89"><label>89</label><mixed-citation>Tian, D., Zhao, X., Gao, L., Jiang, T., Liang, Z., Yang, Z., Zhang, P., Wu, Q., Ren, K., Yang, C., Li, R., Li, S., Cao, Y., Xuan, Y., Chen, J., and Zhu, A.: A framework for tracing the sources of nitrate in surface water through remote sensing data coupled with machine learning, npj Clean Water, 8, 43, <ext-link xlink:href="https://doi.org/10.1038/s41545-025-00473-3" ext-link-type="DOI">10.1038/s41545-025-00473-3</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib90"><label>90</label><mixed-citation>Tollefson, J.: Google AI model creates maps of Earth `at any place and time', Nature, 644, 313, <ext-link xlink:href="https://doi.org/10.1038/d41586-025-02412-1" ext-link-type="DOI">10.1038/d41586-025-02412-1</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib91"><label>91</label><mixed-citation>Torres-Martínez, J. A., Mora, A., Mahlknecht, J., Daessle, L. W., Cervantes-Avilés, P. A., and Ledesma-Ruiz, R. L.: Estimation of nitrate pollution sources and transformations in groundwater of an intensive livestock-agricultural area (Comarca Lagunera), combining major ions, stable isotopes and MixSIAR model, Environ. Pollut., 269, 115445, <ext-link xlink:href="https://doi.org/10.1016/j.envpol.2020.115445" ext-link-type="DOI">10.1016/j.envpol.2020.115445</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bib92"><label>92</label><mixed-citation>Tung, P. Y., Sheikh, H. A., Ball, M., Nabiei, F., and Harrison, R.: SIGMA: Spectral interpretation using gaussian mixtures and autoencoder, Geochem. Geophy. Geosy., 24, e2022GC010530, <ext-link xlink:href="https://doi.org/10.1029/2022GC010530" ext-link-type="DOI">10.1029/2022GC010530</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib93"><label>93</label><mixed-citation>Udu, A. G., Salman, M. T., Ghalati, M. K., Lecchini-Visintini, A., Siddle, D., and Dong, H.: Emerging SMOTE and GAN-variants for Data Augmentation in Imbalance Machine Learning Tasks: A Review, IEEE Access, 13, 113838–113853, <ext-link xlink:href="https://doi.org/10.1109/ACCESS.2025.3584532" ext-link-type="DOI">10.1109/ACCESS.2025.3584532</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib94"><label>94</label><mixed-citation>Van Katwyk, P., Fox-Kemper, B., Seroussi, H., Nowicki, S., and Bergen, K.: A variational LSTM emulator of sea level contribution from the Antarctic ice sheet, J. Adv. Model. Earth Sy., 15, e2023MS003899, <ext-link xlink:href="https://doi.org/10.1029/2023MS003899" ext-link-type="DOI">10.1029/2023MS003899</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib95"><label>95</label><mixed-citation>Vedavyasa, K. V. and Kumar, A.: Classification Analysis of Transition Metal Compounds Using Quantum Machine Learning, Adv. Quantum Technol., 8, 2400081, <ext-link xlink:href="https://doi.org/10.1002/qute.202400081" ext-link-type="DOI">10.1002/qute.202400081</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib96"><label>96</label><mixed-citation>Wang, S. Q., Zheng, W. B., and Kong, X. L.: Spatial distribution characteristics of nitrate in shallow groundwater of the agricultural area of the North China Plain, Chinese Journal of Eco-Agriculture, 26, 1476–1482, <ext-link xlink:href="https://doi.org/10.13930/j.cnki.cjea.180639" ext-link-type="DOI">10.13930/j.cnki.cjea.180639</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib97"><label>97</label><mixed-citation>Wang, S., Chen, J., Zhang, S., Zhang, X., Chen, D., and Zhou, J.: Hydrochemical evolution characteristics, controlling factors, and high nitrate hazards of shallow groundwater in a typical agricultural area of Nansi Lake Basin, North China, Environ. Res., 223, 115430, <ext-link xlink:href="https://doi.org/10.1016/j.envres.2023.115430" ext-link-type="DOI">10.1016/j.envres.2023.115430</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib98"><label>98</label><mixed-citation>Wang, C., Yang, J., and Zhang, B.: A fault diagnosis method using improved prototypical network and weighting similarity-Manhattan distance with insufficient noisy data, Measurement, 226, 114171, <ext-link xlink:href="https://doi.org/10.1016/j.measurement.2024.114171" ext-link-type="DOI">10.1016/j.measurement.2024.114171</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib99"><label>99</label><mixed-citation>Wang, J., Hao, X., Liu, X., Ouyang, W., Li, T., Cui, X., Pei, J., Zhang, S., Zhu, W., and Jin, R.: Groundwater–surface water exchange affects nitrate fate in a seasonal freeze–thaw watershed: Sources, migration and removal, J. Hydrol., 654, 132803, <ext-link xlink:href="https://doi.org/10.1016/j.jhydrol.2025.132803" ext-link-type="DOI">10.1016/j.jhydrol.2025.132803</ext-link>, 2025a.</mixed-citation></ref>
      <ref id="bib1.bib100"><label>100</label><mixed-citation>Wang, J., Wang, M., Zhang, C., Fan, J., and Lv, F.: Vertical Migration Characteristics and Effect Factors of BaP in Contaminated Soil Under Rainwater Infiltration, Water Air Soil Poll., 236, 915, <ext-link xlink:href="https://doi.org/10.1007/s11270-025-08578-8" ext-link-type="DOI">10.1007/s11270-025-08578-8</ext-link>, 2025b.</mixed-citation></ref>
      <ref id="bib1.bib101"><label>101</label><mixed-citation>Wang, N., Zhou, Q., Gao, J., and Wang, Z.: Evaluating the efficacy of PCA and t-SNE in optimizing input features for groundwater level simulation using machine learning models, Environ. Earth Sci., 84, 336, <ext-link xlink:href="https://doi.org/10.1007/s12665-025-12357-3" ext-link-type="DOI">10.1007/s12665-025-12357-3</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib102"><label>102</label><mixed-citation>Weller, D. L., Murphy, C. M., Johnson, S., Green, H., Michalenko, E., Love, T. M., and Strawn, L. K.: Land use, weather, and water quality factors associated with fecal contamination of northeastern streams that span an urban-rural gradient, Front. Water, 3, 741676, <ext-link xlink:href="https://doi.org/10.3389/frwa.2021.741676" ext-link-type="DOI">10.3389/frwa.2021.741676</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib103"><label>103</label><mixed-citation>Williams, M. R., Ford, W. I., and Mumbi, R. C. K.: Preferential flow in the shallow vadose zone: Effect of rainfall intensity, soil moisture, connectivity, and agricultural management, Hydrol. Process., 37, 15057, <ext-link xlink:href="https://doi.org/10.1002/hyp.15057" ext-link-type="DOI">10.1002/hyp.15057</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib104"><label>104</label><mixed-citation>Wu, H., Song, F., Min, L., Li, J., Shen, Y., Huang, Y., Fan, H., Liu, J., and Fu, S.: Exploring recharge mechanisms of soil water in the thick unsaturated zone using water isotopes in the North China Plain, Catena, 234, 107615, <ext-link xlink:href="https://doi.org/10.1016/j.catena.2023.107615" ext-link-type="DOI">10.1016/j.catena.2023.107615</ext-link>, 2024.</mixed-citation></ref>
      <ref id="bib1.bib105"><label>105</label><mixed-citation>Wu, Y., Wang, J., Liu, Z., Li, C., Niu, Y., and Jiang, X.: Seasonal nitrate input drives the spatiotemporal variability of regional surface water-groundwater interactions, nitrate sources and transformations, J. Hydrol., 655, 132973, <ext-link xlink:href="https://doi.org/10.1016/j.jhydrol.2025.132973" ext-link-type="DOI">10.1016/j.jhydrol.2025.132973</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib106"><label>106</label><mixed-citation>Xiong, X., Hao, G., Xu, L., Li, Z., Zhang, X., Lu, J., and Zhang, Z.: Characteristics of spatial distributions for nitrate and traceability of pollution in shallow groundwater of the Qingshui River Basin, China, Hydrol. Res., 56, 920–936, <ext-link xlink:href="https://doi.org/10.2166/nh.2025.047" ext-link-type="DOI">10.2166/nh.2025.047</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib107"><label>107</label><mixed-citation>Xu, G., Su, X., Yuan, Z., Ji, L., Li, N., and Liang, H.: Nitrogen behavior during artificial groundwater recharge through ponds: A case study in Xiong'an New Area, Environ. Geochem. Hlth., 44, 2545–2561, <ext-link xlink:href="https://doi.org/10.1007/s10653-021-01041-7" ext-link-type="DOI">10.1007/s10653-021-01041-7</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bib108"><label>108</label><mixed-citation>Xu, J.: sherlockjjobs/Quantum-RF-framework: Quantum-RF-framework (Version groundwater-nitrate-prediction), Zenodo [data set] and [code], <ext-link xlink:href="https://doi.org/10.5281/zenodo.22329075" ext-link-type="DOI">10.5281/zenodo.22329075</ext-link>,  2026.</mixed-citation></ref>
      <ref id="bib1.bib109"><label>109</label><mixed-citation>Yan, R. and Huang, J. J.: Confident learning-based Gaussian mixture model for leakage detection in water distribution networks, Water Res., 247, 120773, <ext-link xlink:href="https://doi.org/10.1016/j.watres.2023.120773" ext-link-type="DOI">10.1016/j.watres.2023.120773</ext-link>, 2023.</mixed-citation></ref>
      <ref id="bib1.bib110"><label>110</label><mixed-citation>Yang, W., Long, D., Scanlon, B. R., Burek, P., Zhang, C., Han, Z., Butler, J., Pan, Y., Lei, X., and Wada, Y.: Human intervention will stabilize groundwater storage across the North China Plain, Water Resour. Res., 58, e2021WR030884, <ext-link xlink:href="https://doi.org/10.1029/2021WR030884" ext-link-type="DOI">10.1029/2021WR030884</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bib111"><label>111</label><mixed-citation>Zhang, M., Zhi, Y. Y., Shi, J. C., and Wu, L. S.: Apportionment and uncertainty analysis of nitrate sources based on the dual isotope approach and a Bayesian isotope mixing model at the watershed scale, Sci. Total Environ., 639, 1175–1187, <ext-link xlink:href="https://doi.org/10.1016/j.scitotenv.2018.05.239" ext-link-type="DOI">10.1016/j.scitotenv.2018.05.239</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bib112"><label>112</label><mixed-citation>Zhang, W., Xin, C., and Yu, S.: A review of heavy metal migration and its influencing factors in karst groundwater, Northern and Southern China, Water, 15, 3690, <ext-link xlink:href="https://doi.org/10.3390/w15203690" ext-link-type="DOI">10.3390/w15203690</ext-link>, 2023. </mixed-citation></ref>
      <ref id="bib1.bib113"><label>113</label><mixed-citation>Zhang, J., Zhang, L., Zheng, T., Jin, M., Kang, F., Jiang, J., Yuan, Z., and Luo, J.: Tracing Nitrate Contamination Sources and Transformations in a Rural-Urban Karst Groundwater System in North China Using Multiple Isotopes and Simmr Modeling, Water Resour. Res., 61, e2025WR040156, <ext-link xlink:href="https://doi.org/10.1029/2025WR040156" ext-link-type="DOI">10.1029/2025WR040156</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib114"><label>114</label><mixed-citation>Zhao, Y., Liu, J., Zhang, X., Li, Q., and Wu, J.: Integrated Machine Learning and Health Risk Assessment for Groundwater Nitrate Contamination in Handan City, China, Water, 18, 1174, <ext-link xlink:href="https://doi.org/10.3390/w18101174" ext-link-type="DOI">10.3390/w18101174</ext-link>, 2026.</mixed-citation></ref>
      <ref id="bib1.bib115"><label>115</label><mixed-citation>Zheng, Y., Zhang, X., Zhou, Y., Zhang, Y., Zhang, T., and Farmani, R.: Deep representation learning enables cross-basin water quality prediction under data-scarce conditions, npj Clean Water, 8, 33, <ext-link xlink:href="https://doi.org/10.1038/s41545-025-00466-2" ext-link-type="DOI">10.1038/s41545-025-00466-2</ext-link>, 2025.</mixed-citation></ref>
      <ref id="bib1.bib116"><label>116</label><mixed-citation>Zhu, J. J., Yang, M., and Ren, Z. J.: Machine learning in environmental research: common pitfalls and best practices, Environ. Sci. Technol., 57, 17671–17689, <ext-link xlink:href="https://doi.org/10.1021/acs.est.3c00026" ext-link-type="DOI">10.1021/acs.est.3c00026</ext-link>, 2023.</mixed-citation></ref>

  </ref-list></back>
    <!--<article-title-html>Hydrochemistry and modeling nitrate concentration in farmland groundwater under different hydrological seasons by integrating hybrid quantum-classical ML, virtual sample generation and AlphaEarth Foundation</article-title-html>
<abstract-html/>
<ref-html id="bib1.bib1"><label>1</label><mixed-citation>
      
Abderzak, M., Zeghmar, A., Leila, B., Aziz, M., Velibor, S., Lizny, J.,
Mohamed, K., Fernanda, H., and Shuraik, K.: Ensemble learning-driven
optimization of coagulant dosing for drinking water treatment plants using a
scalable framework for smart and sustainable process control, Environ. Res.,
288, 123229, <a href="https://doi.org/10.1016/j.envres.2025.123229" target="_blank">https://doi.org/10.1016/j.envres.2025.123229</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib2"><label>2</label><mixed-citation>
      
Addy, J. W. G., MacLaren, C., and Lang, R.: A Bayesian approach to analyzing
long-term agricultural experiments, Eur. J. Agron., 159, 127227,
<a href="https://doi.org/10.1016/j.eja.2024.127227" target="_blank">https://doi.org/10.1016/j.eja.2024.127227</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib3"><label>3</label><mixed-citation>
      
Ahmed, M. A., Abdel Samie, S. G., and Badawy, H. A.: Factors controlling
mechanisms of groundwater salinization and hydrogeochemical processes in the
Quaternary aquifer of the Eastern Nile Delta, Egypt, Environ. Earth Sci.,
68, 369–394, <a href="https://doi.org/10.1007/s12665-012-1744-6" target="_blank">https://doi.org/10.1007/s12665-012-1744-6</a>, 2013.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib4"><label>4</label><mixed-citation>
      
Alam, G. M. I., Arfin Tanim, S., Sarker, S. K., Watanobe, Y., Islam, R.,
Mridha, M. F., and Nur, K.: Deep learning model based prediction of vehicle
CO<sub>2</sub> emissions with eXplainable AI integration for sustainable environment,
Sci. Rep., 15, 3655, <a href="https://doi.org/10.1038/s41598-025-87233-y" target="_blank">https://doi.org/10.1038/s41598-025-87233-y</a>, 2025a.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib5"><label>5</label><mixed-citation>
      
Alam, S. M. K., Li, P., Rahman, M., Fida, M., and Elumalai, V.: Key factors
affecting groundwater nitrate levels in the Yinchuan Region, Northwest
China: Research using the eXtreme Gradient Boosting (XGBoost) model with the
SHapley Additive exPlanations (SHAP) method, Environ. Pollut., 364, 125336,
<a href="https://doi.org/10.1016/j.envpol.2024.125336" target="_blank">https://doi.org/10.1016/j.envpol.2024.125336</a>, 2025b.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib6"><label>6</label><mixed-citation>
      
Alvarez, C. I., Ulloa Vaca, C. A., and Echeverria Llumipanta, N. A.: Machine
learning for urban air quality prediction using Google AlphaEarth
Foundations satellite embeddings: A case study of Quito, Ecuador, Remote
Sens., 17, 3472, <a href="https://doi.org/10.3390/rs17203472" target="_blank">https://doi.org/10.3390/rs17203472</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib7"><label>7</label><mixed-citation>
      
An, B., Zhang, Z., Ren, J., and Zhang, W.: Recurrent adversarial learning
for geo-technical time-series augmentation: application to slope instability
forecasting in open-pit mines, Environ. Earth Sci., 84, 559,
<a href="https://doi.org/10.1007/s12665-025-12566-w" target="_blank">https://doi.org/10.1007/s12665-025-12566-w</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib8"><label>8</label><mixed-citation>
      
Anderson, G. J. and Lucas, D. D.: Machine learning predictions of a
multiresolution climate model ensemble, Geophys. Res. Lett., 45, 4273–4280,
<a href="https://doi.org/10.1029/2018GL077049" target="_blank">https://doi.org/10.1029/2018GL077049</a>, 2018.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib9"><label>9</label><mixed-citation>
      
Balogun, E., Rajagopal, R., and Majumdar, A.: TemperatureGAN: generative
modeling of regional atmospheric temperatures, Environ. Data Sci., 3, e21,
<a href="https://doi.org/10.1017/eds.2024.21" target="_blank">https://doi.org/10.1017/eds.2024.21</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib10"><label>10</label><mixed-citation>
      
Bigler, M. C., Brusseau, M. L., Guo, B., Jones, S. L., Pritchard, J. C.,
Higgins, C. P., and Hatton, J.: High-resolution depth-discrete analysis of
PFAS distribution and leaching for a vadose-zone source at an AFFF-Impacted
site, Environ. Sci. Technol., 58, 9863–9874,
<a href="https://doi.org/10.1021/acs.est.4c01615" target="_blank">https://doi.org/10.1021/acs.est.4c01615</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib11"><label>11</label><mixed-citation>
      
Cai, C., Zhao, H., Zhang, H., Wang, C., Wang, Z., Liu, M., Chen, J., and
Zhang, H.: Timely assessment of maize lodging severity with limited samples
using multi-temporal Sentinel-1 and Sentinel-2 data across large spatial
extents, Comput. Electron. Agr., 237, 110671,
<a href="https://doi.org/10.1016/j.compag.2025.110671" target="_blank">https://doi.org/10.1016/j.compag.2025.110671</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib12"><label>12</label><mixed-citation>
      
Charizanos, G. and Demirhan, H.: Bayesian prediction of wildfire event
probability using normalized difference vegetation index data from an
Australian forest, Ecol. Inform., 73, 101899,
<a href="https://doi.org/10.1016/j.ecoinf.2022.101899" target="_blank">https://doi.org/10.1016/j.ecoinf.2022.101899</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib13"><label>13</label><mixed-citation>
      
Chen, Q., Yang, H., Cui, R., Hu, W., Wang, C., Chen, A., and Zhang, D.:
Shallow groundwater table fluctuations: A driving force for accelerating the
migration and transformation of phosphorus in cropland soil, Water Res.,
275, 123209, <a href="https://doi.org/10.1016/j.watres.2025.123209" target="_blank">https://doi.org/10.1016/j.watres.2025.123209</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib14"><label>14</label><mixed-citation>
      
Clark, S. R. and Jaffres, J. B. D.: Associations between deep learning
runoff predictions and hydrogeological conditions in Australia, J. Hydrol.,
651, 132569, <a href="https://doi.org/10.1016/j.jhydrol.2024.132569" target="_blank">https://doi.org/10.1016/j.jhydrol.2024.132569</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib15"><label>15</label><mixed-citation>
      
Cowlessur, H., Alpcan, T., Thapa, C., Camtepe, S., and Kundu, N. K.: A
Qubit-Efficient Hybrid Quantum Encoding Mechanism for Quantum Machine
Learning, arXiv [preprint],
<a href="https://doi.org/10.48550/arXiv.2506.19275" target="_blank">https://doi.org/10.48550/arXiv.2506.19275</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib16"><label>16</label><mixed-citation>
      
Dega, S., Dietrich, P., Schroen, M., and Paasche, H.: Probabilistic
prediction by means of the propagation of response variable uncertainty
through a Monte Carlo approach in regression random forest: Application to
soil moisture regionalization, Front. Environ. Sci., 11, 1009191,
<a href="https://doi.org/10.3389/fenvs.2023.1009191" target="_blank">https://doi.org/10.3389/fenvs.2023.1009191</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib17"><label>17</label><mixed-citation>
      
Deng, Y., Ye, X., and Du, X.: Predictive modeling and analysis of key
drivers of groundwater nitrate pollution based on machine learning, J.
Hydrol., 624, 129934, <a href="https://doi.org/10.1016/j.jhydrol.2023.129934" target="_blank">https://doi.org/10.1016/j.jhydrol.2023.129934</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib18"><label>18</label><mixed-citation>
      
Di Santo, D., He, C., Chen, F., and Giovannini, L.: ML-AMPSIT: Machine Learning-based Automated Multi-method Parameter Sensitivity and Importance analysis Tool, Geosci. Model Dev., 18, 433–459, <a href="https://doi.org/10.5194/gmd-18-433-2025" target="_blank">https://doi.org/10.5194/gmd-18-433-2025</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib19"><label>19</label><mixed-citation>
      
Farahani, M. A., Wood, A. W., Tang, G., and Mizukami, N.: Calibrating a large-domain land/hydrology process model in the age of AI: the SUMMA CAMELS emulator experiments, Hydrol. Earth Syst. Sci., 29, 4515–4537, <a href="https://doi.org/10.5194/hess-29-4515-2025" target="_blank">https://doi.org/10.5194/hess-29-4515-2025</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib20"><label>20</label><mixed-citation>
      
Farnia, F., Wang, W. W., Das, S., and Jadbabaie, A.: Gat–gmm: Generative
adversarial training for gaussian mixture models, SIAM J. Math. Data Sci.,
5, 122–146, <a href="https://doi.org/10.1137/21M1445831" target="_blank">https://doi.org/10.1137/21M1445831</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib21"><label>21</label><mixed-citation>
      
Feng, D., Liu, J., Lawson, K., and Shen, C.: Differentiable, learnable,
regionalized process-based models with multiphysical outputs can approach
state-of-the-art hydrologic prediction accuracy, Water Resour. Res., 58,
e2022WR032404, <a href="https://doi.org/10.1029/2022WR032404" target="_blank">https://doi.org/10.1029/2022WR032404</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib22"><label>22</label><mixed-citation>
      
Feng, W., Wang, S., Tan, K., Ma, L., and Hu, C.: Simulation of spatial and
temporal variation of nitrate leaching in the vadose zone of alluvial
regions on a large regional scale, Sci. Total Environ., 916, 170114,
<a href="https://doi.org/10.1016/j.scitotenv.2024.170114" target="_blank">https://doi.org/10.1016/j.scitotenv.2024.170114</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib23"><label>23</label><mixed-citation>
      
Gao, H., Yang, L., Song, X., Guo, M., Li, B., and Cui, X.: Sources and
hydrogeochemical processes of groundwater under multiple water source
recharge condition, Sci. Total Environ., 903, 166660,
<a href="https://doi.org/10.1016/j.scitotenv.2023.166660" target="_blank">https://doi.org/10.1016/j.scitotenv.2023.166660</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib24"><label>24</label><mixed-citation>
      
Ghodba, A., Richelle, A., McCready, C., Ricardez-Sandoval, L., and Budman,
H.: A novel dynamic flux balance analysis for modeling CHO cell fed-batch
cultures with pH and temperature shifts, J. Biotechnol., 408, 61–71,
<a href="https://doi.org/10.1016/j.jbiotec.2025.08.010" target="_blank">https://doi.org/10.1016/j.jbiotec.2025.08.010</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib25"><label>25</label><mixed-citation>
      
Gu, B. J., Ge, Y., Chang, S. X., Luo, W. D., and Chang, J.: Nitrate in
groundwater of China: Sources and driving forces, Global Environ. Chang., 23,
1112–1121, <a href="https://doi.org/10.1016/j.gloenvcha.2013.05.004" target="_blank">https://doi.org/10.1016/j.gloenvcha.2013.05.004</a>, 2013.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib26"><label>26</label><mixed-citation>
      
Gujju, Y., Matsuo, A., and Raymond, R.: Quantum machine learning on
near-term quantum devices: Current state of supervised and unsupervised
techniques for real-world applications, Phys. Rev. Appl., 21, 067001,
<a href="https://doi.org/10.1103/PhysRevApplied.21.067001" target="_blank">https://doi.org/10.1103/PhysRevApplied.21.067001</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib27"><label>27</label><mixed-citation>
      
Gul, N., Khan, A. U., Ullah, B., Khan, B. N., Almalki, H. M., Banga, A. S.,
and Kumar, K.: Optimizing LSTM for sediment load prediction in the Swat
River Basin, Pakistan: Evaluation of optimizers and activation functions,
Phys. Chem. Earth, 140, 104019, <a href="https://doi.org/10.1016/j.pce.2025.104019" target="_blank">https://doi.org/10.1016/j.pce.2025.104019</a>,
2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib28"><label>28</label><mixed-citation>
      
Haggerty, R., Sun, J., Yu, H., and Li, Y.: Application of machine learning
in groundwater quality modeling – A comprehensive review, Water Res., 233,
119745, <a href="https://doi.org/10.1016/j.watres.2023.119745" target="_blank">https://doi.org/10.1016/j.watres.2023.119745</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib29"><label>29</label><mixed-citation>
      
Han, D. M., Currell, M. J., and Cao, G. L.: Deep challenges for China's war
on water pollution, Environ. Pollut., 218, 1222–1233,
<a href="https://doi.org/10.1016/j.envpol.2016.08.078" target="_blank">https://doi.org/10.1016/j.envpol.2016.08.078</a>, 2016.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib30"><label>30</label><mixed-citation>
      
Havlíček, V., Córcoles, A. D., Temme, K., Harrow, A., Kandala,
A., Chow, J. M., and Gambetta, J. M.: Supervised learning with
quantum-enhanced feature spaces, Nature, 567, 209–212,
<a href="https://doi.org/10.1038/s41586-019-0980-2" target="_blank">https://doi.org/10.1038/s41586-019-0980-2</a>, 2019.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib31"><label>31</label><mixed-citation>
      
Hollmann, N., Müller, S., Purucker, L., Krishnakumar, A., Körfer,
M., Hoo, S. B., Schirrmeister, R. T., and Hutter, F.: Accurate predictions
on small data with a tabular foundation model, Nature, 637, 319–326,
<a href="https://doi.org/10.1038/s41586-024-08328-6" target="_blank">https://doi.org/10.1038/s41586-024-08328-6</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib32"><label>32</label><mixed-citation>
      
Hong, Y. Y. and Lopez, D. J. D.: A Review on Quantum Machine Learning in
Applied Systems and Engineering, IEEE Access, 13, 144607–144631,
<a href="https://doi.org/10.1109/ACCESS.2025.3599147" target="_blank">https://doi.org/10.1109/ACCESS.2025.3599147</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib33"><label>33</label><mixed-citation>
      
Hou, X., Peng, L., Zhang, Y., Zhang, Y., Wang, Y., Feng, W., and Yang, H.: A
Data-Driven Method for Determining DRASTIC Weights to Assess Groundwater
Vulnerability to Nitrate: Application in the Lake Baiyangdian Watershed,
North China Plain, Appl. Sci., 15, 2866,
<a href="https://doi.org/10.3390/app15052866" target="_blank">https://doi.org/10.3390/app15052866</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib34"><label>34</label><mixed-citation>
      
Huang, P. and Chen, J.: Recharge sources and hydrogeochemical evolution of
groundwater in the coal-mining district of Jiaozuo, China, Hydrogeol. J.,
20, 739–754, <a href="https://doi.org/10.1007/s10040-012-0836-4" target="_blank">https://doi.org/10.1007/s10040-012-0836-4</a>, 2012.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib35"><label>35</label><mixed-citation>
      
Huang, C., Tong, J., and Ye, M.: Global sensitivity analysis for a
prediction model of soil solute transfer into surface runoff, J. Hydrol.,
599, 126342, <a href="https://doi.org/10.1016/j.jhydrol.2021.126342" target="_blank">https://doi.org/10.1016/j.jhydrol.2021.126342</a>, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib36"><label>36</label><mixed-citation>
      
Islam, M. T., Zhou, Z., Ren, H., Khuzani, M. B., Kapp, D., Zou, J., Tian,
L., Liao, J. C., and Xing, L.: Revealing hidden patterns in deep neural
network feature space continuum via manifold learning, Nat. Commun., 14,
8506, <a href="https://doi.org/10.1038/s41467-023-43958-w" target="_blank">https://doi.org/10.1038/s41467-023-43958-w</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib37"><label>37</label><mixed-citation>
      
Jamshidi, E. J., Yusup, Y., Kayode, J. S., and Kamaruddin, M. A.: Detecting
outliers in a univariate time series dataset using unsupervised combined
statistical methods: A case study on surface water temperature, Ecol.
Inform., 69, 101672, <a href="https://doi.org/10.1016/j.ecoinf.2022.101672" target="_blank">https://doi.org/10.1016/j.ecoinf.2022.101672</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib38"><label>38</label><mixed-citation>
      
Jia, H. and Qian, H.: Groundwater nitrate response to hydrogeological
conditions and socioeconomic load in an agriculture dominated area, Sci.
Rep., 15, 1315, <a href="https://doi.org/10.1038/s41598-024-84318-y" target="_blank">https://doi.org/10.1038/s41598-024-84318-y</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib39"><label>39</label><mixed-citation>
      
Jia, B., Zhou, J., Tang, Z., Xu, Z., Chen, X., and Fang, W.: Effective
stochastic streamflow simulation method based on Gaussian mixture model, J.
Hydrol., 605, 127366, <a href="https://doi.org/10.1016/j.jhydrol.2021.127366" target="_blank">https://doi.org/10.1016/j.jhydrol.2021.127366</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib40"><label>40</label><mixed-citation>
      
Jiang, H., Geng, D., Fu, J., Liu, M., Qi, Y., Min, L., Wang, S., and Shen,
Y.: Quantifying nitrate dynamics in deep vadose zone under irrigated
farmland in the North China Plain: Insights from continuous in-situ
monitoring, Agr. Ecosyst. Environ., 396, 109956,
<a href="https://doi.org/10.1016/j.agee.2025.109956" target="_blank">https://doi.org/10.1016/j.agee.2025.109956</a>, 2026.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib41"><label>41</label><mixed-citation>
      
Kalinin, A. A., Arevalo, J., Serrano, E., Vulliard, L., Tsang, H.,
Bornholdt, M., Muñoz, A. F., Sivagurunathan, S., Rajwa, B., Carpenter,
A. E., Way, G. P., and Singh, S.: A versatile information retrieval
framework for evaluating profile strength and similarity, Nat. Commun., 16,
5181, <a href="https://doi.org/10.1038/s41467-025-60306-2" target="_blank">https://doi.org/10.1038/s41467-025-60306-2</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib42"><label>42</label><mixed-citation>
      
Karimanzira, D., Weis, J., Wunsch, A., Ritzau, L., Liesch, T., and Ohmer,
M.: Application of machine learning and deep neural networks for spatial
prediction of groundwater nitrate concentration to improve land use
management practices, Front. Water, 5, 1193142,
<a href="https://doi.org/10.3389/frwa.2023.1193142" target="_blank">https://doi.org/10.3389/frwa.2023.1193142</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib43"><label>43</label><mixed-citation>
      
Kaur, H., Bansod, B. S., Khungar, P., and Dhawan, C.: Combining clustering
and ensemble learning for groundwater quality monitoring: a data-driven
framework for sustainable water management, Environ. Sci. Pollut. R., 32,
13862–13903, <a href="https://doi.org/10.1007/s11356-025-36477-2" target="_blank">https://doi.org/10.1007/s11356-025-36477-2</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib44"><label>44</label><mixed-citation>
      
Khalil, M., Zhang, C., Ye, Z., and Zhang, P.: PegasosQSVM: A Quantum Machine
Learning Approach for Accurate Fake News Detection, Appl. Artif. Intell.,
39, 2457207, <a href="https://doi.org/10.1080/08839514.2025.2457207" target="_blank">https://doi.org/10.1080/08839514.2025.2457207</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib45"><label>45</label><mixed-citation>
      
Kobak, D. and Berens, P.: The art of using t-SNE for single-cell
transcriptomics, Nat. Commun., 10, 5416,
<a href="https://doi.org/10.1038/s41467-019-13056-x" target="_blank">https://doi.org/10.1038/s41467-019-13056-x</a>, 2019.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib46"><label>46</label><mixed-citation>
      
Lamichhane, P. and Rawat, D. B.: Quantum Machine Learning: Recent Advances,
Challenges and Perspectives, IEEE Access, 19, 94057–94105,
<a href="https://doi.org/10.1109/ACCESS.2025.3573244" target="_blank">https://doi.org/10.1109/ACCESS.2025.3573244</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib47"><label>47</label><mixed-citation>
      
Li, X. and Yu, L.: Mapping complex cropping patterns in China (2018–2021) at 10 m resolution: a data-driven framework based on multi-product integration and Google satellite embedding, Earth Syst. Sci. Data, 18, 5601–5626, <a href="https://doi.org/10.5194/essd-18-5601-2026" target="_blank">https://doi.org/10.5194/essd-18-5601-2026</a>, 2026.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib48"><label>48</label><mixed-citation>
      
Li, J., Zhu, D., Zhang, S., Yang, G., Zhao, Y., Zhou, C., Lin, Y., and Zou,
S.: Application of the hydrochemistry, stable isotopes and MixSIAR model to
identify nitrate sources and transformations in surface water and
groundwater of an intensive agricultural karst wetland in Guilin, China,
Ecotox. Environ. Safe, 231, 113205,
<a href="https://doi.org/10.1016/j.ecoenv.2022.113205" target="_blank">https://doi.org/10.1016/j.ecoenv.2022.113205</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib49"><label>49</label><mixed-citation>
      
Li, X., Wang, Y., Xue, B., A, Y., Zhang, X., and Wang, G.: Attribution of
runoff and hydrological drought changes in an ecologically vulnerable basin
in semi-arid regions of China, Hydrol. Process., 37, e15003,
<a href="https://doi.org/10.1002/hyp.15003" target="_blank">https://doi.org/10.1002/hyp.15003</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib50"><label>50</label><mixed-citation>
      
Li, R., Feng, K., An, T., Cheng, P., Wei, L., Zhao, Z., Xu, X., and Zhu, L.:
Enhanced insights into effluent prediction in wastewater treatment plants:
Comprehensive deep learning model explanation based on shap, ACS ES&amp;T
Water, 4, 1904–1915, <a href="https://doi.org/10.1021/acsestwater.4c00040" target="_blank">https://doi.org/10.1021/acsestwater.4c00040</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib51"><label>51</label><mixed-citation>
      
Li, J., Liu, G., and Shen, Z.: Integrating spatiotemporal variability of
non-point source pollution and best management practice efficiency to
improve adaptive watershed management, Water Res., 284, 124042,
<a href="https://doi.org/10.1016/j.watres.2025.124042" target="_blank">https://doi.org/10.1016/j.watres.2025.124042</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib52"><label>52</label><mixed-citation>
      
Li, S., Zheng, T., Farchi, A., Bocquet, M., and Gentine, P.: Probabilistic
data assimilation for ensemble distribution projections with generative
machine learning: A Lorenz'96 proof-of-concept, Geophys. Res. Lett., 52,
e2024GL112523, <a href="https://doi.org/10.1029/2024GL112523" target="_blank">https://doi.org/10.1029/2024GL112523</a>, 2025a.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib53"><label>53</label><mixed-citation>
      
Li, X., Liu, M., Min, L., and Shen, Y.: Nitrogen transport and
transformation processes in the typical deep vadose zone in the central
North China Plain, Chinese Journal of Eco-Agriculture, 33, 2359–2370,
<a href="https://doi.org/10.12357/cjea.20250148" target="_blank">https://doi.org/10.12357/cjea.20250148</a>, 2025b.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib54"><label>54</label><mixed-citation>
      
Li, X., Zhou, J., Zhou, W., Mao, L., Wang, C., Hao, Y., and Bian, P.:
Hydrochemical Characteristics of Shallow Groundwater and Analysis of
Vegetation Water Sources in the Ulan Buh Desert, Water, 17, 3058,
<a href="https://doi.org/10.3390/w17213058" target="_blank">https://doi.org/10.3390/w17213058</a>, 2025c.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib55"><label>55</label><mixed-citation>
      
Li, Y., Jiang, Z., Wu, S., Gao, J., and Sharma, A.: Systematic evaluation
and correction of extreme water level in global storm surge simulation,
Geophys. Res. Lett., 53, e2025GL119793,
<a href="https://doi.org/10.1029/2025GL119793" target="_blank">https://doi.org/10.1029/2025GL119793</a>, 2026.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib56"><label>56</label><mixed-citation>
      
Liao, H., Wang, D. S., Sitdikov, I., Salcedo, C., Seif, A., and Minev, Z.
K.: Machine learning for practical quantum error mitigation, Nat. Mach.
Intell., 6, 1478–1486, <a href="https://doi.org/10.1038/s42256-024-00927-2" target="_blank">https://doi.org/10.1038/s42256-024-00927-2</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib57"><label>57</label><mixed-citation>
      
Liu, H., Yang, J., Ye, M., James, S. C., Tang, Z., Dong, J., and Xing, T.:
Using t-distributed Stochastic Neighbor Embedding (t-SNE) for cluster
analysis and spatial zone delineation of groundwater geochemistry data, J.
Hydrol., 597, 126146, <a href="https://doi.org/10.1016/j.jhydrol.2021.126146" target="_blank">https://doi.org/10.1016/j.jhydrol.2021.126146</a>, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib58"><label>58</label><mixed-citation>
      
Liu, M., Min, L., Wu, L., Pei, H., and Shen, Y.: Evaluating nitrate
transport and accumulation in the deep vadose zone of the intensive
agricultural region, North China Plain, Sci. Total Environ., 825, 153894,
<a href="https://doi.org/10.1016/j.scitotenv.2022.153894" target="_blank">https://doi.org/10.1016/j.scitotenv.2022.153894</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib59"><label>59</label><mixed-citation>
      
Liu, S., Hao, Y., Wang, H., Zheng, X., Yu, X., Meng, X., Qiu, Y., Li, S.,
and Zheng, T.: Bidirectional potential effects of DON transformation in
vadose zones on groundwater nitrate contamination: Different contributions
to nitrification and denitrification, J. Hazard. Mater., 448, 130976,
<a href="https://doi.org/10.1016/j.jhazmat.2023.130976" target="_blank">https://doi.org/10.1016/j.jhazmat.2023.130976</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib60"><label>60</label><mixed-citation>
      
Liu, M., Geng, D., Wu, L., Min, L., Wang, S., and Shen, Y.: The impact of
agricultural land use change on water and nitrate fluxes in the deep vadose
zone, the North China Plain, J. Hydrol.-Reg. Stud., 62, 102914,
<a href="https://doi.org/10.1016/j.ejrh.2025.102914" target="_blank">https://doi.org/10.1016/j.ejrh.2025.102914</a>, 2025a.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib61"><label>61</label><mixed-citation>
      
Liu, N., Chen, M., Gao, D., Wu, Y., and Wang, X.: Identification of
hydrogeochemical processes in shallow groundwater using multivariate
statistical analysis and inverse geochemical modeling, Environ. Monit.
Assess., 197, 135, <a href="https://doi.org/10.1007/s10661-024-13528-8" target="_blank">https://doi.org/10.1007/s10661-024-13528-8</a>, 2025b.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib62"><label>62</label><mixed-citation>
      
Liu, X., Yue, F. J., Li, L., Zhou, F., Wen, H., Yan, Z., Wang, L., Wong, W.
W., Liu, C. Q., and Li, S. L.: Chronic nitrogen legacy in the aquifers of
China, Commun. Earth Environ., 6, 58,
<a href="https://doi.org/10.1038/s43247-025-02016-7" target="_blank">https://doi.org/10.1038/s43247-025-02016-7</a>, 2025c.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib63"><label>63</label><mixed-citation>
      
Luo, Y., Yan, J., McClure, S. C., and Li, F.: Socioeconomic and
environmental factors of poverty in China using geographically weighted
random forest regression model, Environ. Sci. Pollut. R., 29,
33205–33217, <a href="https://doi.org/10.1007/s11356-021-17513-3" target="_blank">https://doi.org/10.1007/s11356-021-17513-3</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib64"><label>64</label><mixed-citation>
      
Ma, Z., Ye, C., Lu, C., Wei, Q., Zhu, R., Xie, X., Zhong, S., Chu, W., and
Xu, Z.: A Robust Gaussian Process Paradigm for Predictive Modeling on Small
Data sets in Environmental Science: A Case Study in Ballasted Flocculation,
Environ. Sci. Technol., 60, 748–759,
<a href="https://doi.org/10.1021/acs.est.5c12617" target="_blank">https://doi.org/10.1021/acs.est.5c12617</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib65"><label>65</label><mixed-citation>
      
Mao, H., Wang, G., Liao, F., Shi, Z., Zhang, H., Chen, X., Qiao, Z., Li, B.,
and Bai, Y.: Spatial variability of source contributions to nitrate in
regional groundwater based on the positive matrix factorization and Bayesian
model, J. Hazard. Mater., 445, 130569,
<a href="https://doi.org/10.1016/j.jhazmat.2022.130569" target="_blank">https://doi.org/10.1016/j.jhazmat.2022.130569</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib66"><label>66</label><mixed-citation>
      
Merabet, K., Di Nunno, F., Granata, F., Kim, S., Adnan, R. M., Heddam, S.,
Kisi, O., and Zounemat-Kermani, M.: Predicting water quality variables using
gradient boosting machine: global versus local explainability using SHapley
Additive Explanations (SHAP), Earth Sci. Inform., 18, 298,
<a href="https://doi.org/10.1007/s12145-025-01796-y" target="_blank">https://doi.org/10.1007/s12145-025-01796-y</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib67"><label>67</label><mixed-citation>
      
Miao, P., Zhu, X., Zhang, S., Li, W., Zhou, J., and Chen, Z.: Long-term high
nitrogen surplus in intensive apple-planting regions results in huge legacy
nitrogen in the vadose zone: Potential or real risk to the groundwater,
Agr. Ecosyst. Environ., 357, 108682,
<a href="https://doi.org/10.1016/j.agee.2023.108682" target="_blank">https://doi.org/10.1016/j.agee.2023.108682</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib68"><label>68</label><mixed-citation>
      
Michalek, A. T., Villarini, G., Kim, T., Quintero, F., Krajewski, W., and
Scoccimarro, E.: Evaluation of CMIP6 HighResMIP for hydrologic modeling of
annual maximum discharge in Iowa, Water Resour. Res., 59, e2022WR034166,
<a href="https://doi.org/10.1029/2022WR034166" target="_blank">https://doi.org/10.1029/2022WR034166</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib69"><label>69</label><mixed-citation>
      
Naresh, V. S. and Reddi, S.: Quantum-enhanced predictive analytics in
healthcare: benchmarking QSVM and QNN on medical datasets, Measurement, 258,
119099, <a href="https://doi.org/10.1016/j.measurement.2025.119099" target="_blank">https://doi.org/10.1016/j.measurement.2025.119099</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib70"><label>70</label><mixed-citation>
      
Niu, H., McCallum, G. B., Chang, A. B., Khan, K., and Azam, S.: Exploring
unsupervised feature extraction algorithms: tackling high dimensionality in
small datasets, Sci. Rep., 15, 21973,
<a href="https://doi.org/10.1038/s41598-025-07725-9" target="_blank">https://doi.org/10.1038/s41598-025-07725-9</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib71"><label>71</label><mixed-citation>
      
Nolte, A., Heudorfer, B., Bender, S., and Hartmann, J.: Multi-site deep
learning for groundwater level prediction across global datasets: toward
scalable applications under data scarcity, J. Hydroinform., 27, 1632–1651,
<a href="https://doi.org/10.2166/hydro.2025.095" target="_blank">https://doi.org/10.2166/hydro.2025.095</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib72"><label>72</label><mixed-citation>
      
Oliveira Santos, V., Costa Rocha, P. A., Thé, J. V. G., and Gharabaghi,
B.: Optimizing the Architecture of a Quantum–Classical Hybrid Machine
Learning Model for Forecasting Ozone Concentrations: Air Quality Management
Tool for Houston, Texas, Atmosphere, 16, 255,
<a href="https://doi.org/10.3390/atmos16030255" target="_blank">https://doi.org/10.3390/atmos16030255</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib73"><label>73</label><mixed-citation>
      
Peng, D., Gui, Z., Wei, W., Li, F., Gui, J., Wu, H., and Gong, J.:
Sampling-enabled scalable manifold learning unveils the discriminative
cluster structure of high-dimensional data, Nat. Mach. Intell., 7,
1669–1684, <a href="https://doi.org/10.1038/s42256-025-01112-9" target="_blank">https://doi.org/10.1038/s42256-025-01112-9</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib74"><label>74</label><mixed-citation>
      
Pérez-Salinas, A., Cervera-Lierta, A., Gil-Fuster, E., and Latorre, J.:
Data re-uploading for a universal quantum classifier, Quantum, 4, 226,
<a href="https://doi.org/10.22331/q-2020-02-06-226" target="_blank">https://doi.org/10.22331/q-2020-02-06-226</a>, 2020.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib75"><label>75</label><mixed-citation>
      
Ranga, D., Rana, A., Prajapat, S., Kumar, P., Kumar, K., and Vasilakos, A.
V.: Quantum machine learning: Exploring the role of data encoding
techniques, challenges, and future directions, Mathematics, 12, 3318,
<a href="https://doi.org/10.3390/math12213318" target="_blank">https://doi.org/10.3390/math12213318</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib76"><label>76</label><mixed-citation>
      
Ransom, K. M., Nolan, B. T., Stackelberg, P. E., Belitz, K., and Fram, M.
S.: Machine learning predictions of nitrate in groundwater used for drinking
supply in the conterminous United States, Sci. Total Environ., 807, 151065,
<a href="https://doi.org/10.1016/j.scitotenv.2021.151065" target="_blank">https://doi.org/10.1016/j.scitotenv.2021.151065</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib77"><label>77</label><mixed-citation>
      
Ren, M., Sun, W., and Chen, S.: Combining machine learning models through
multiple data division methods for PM2.5 forecasting in Northern Xinjiang,
China, Environ. Monit. Assess., 193, 476,
<a href="https://doi.org/10.1007/s10661-021-09233-5" target="_blank">https://doi.org/10.1007/s10661-021-09233-5</a>, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib78"><label>78</label><mixed-citation>
      
Saberian, M., Zafarmomen, N., Neupane, A., Panthi, K., and Samadi, V.:
HydroQuantum: A new quantum-driven Python package for hydrological
simulation, Environ. Modell. Softw., 195, 106736,
<a href="https://doi.org/10.1016/j.envsoft.2025.106736" target="_blank">https://doi.org/10.1016/j.envsoft.2025.106736</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib79"><label>79</label><mixed-citation>
      
Saha, G. K., Rahmani, F., Shen, C., Li, L., and Cibin, R.: A deep
learning-based novel approach to generate continuous daily stream nitrate
concentration for nitrate data-sparse watersheds, Sci. Total Environ., 878,
162930, <a href="https://doi.org/10.1016/j.scitotenv.2023.162930" target="_blank">https://doi.org/10.1016/j.scitotenv.2023.162930</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib80"><label>80</label><mixed-citation>
      
Schuld, M. and Killoran, N.: Quantum machine learning in feature Hilbert
spaces, Phys. Rev. Lett., 122, 040504,
<a href="https://doi.org/10.1103/PhysRevLett.122.040504" target="_blank">https://doi.org/10.1103/PhysRevLett.122.040504</a>, 2019.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib81"><label>81</label><mixed-citation>
      
Sebestyen, S. D., Shanley, J. B., Boyer, E. W., Kendall, C., and Doctor, D.
H.: Coupled hydrological and biogeochemical processes controlling
variability of nitrogen species in streamflow during autumn in an upland
forest, Water Resour. Res., 50, 1569–1591,
<a href="https://doi.org/10.1002/2013WR013670" target="_blank">https://doi.org/10.1002/2013WR013670</a>, 2014.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib82"><label>82</label><mixed-citation>
      
Silva, R. and Melo-Pinto, P.: t-SNE: A study on reducing the dimensionality
of hyperspectral data for the regression problem of estimating oenological
parameters, Artif. Intell. Agric., 7, 58–68,
<a href="https://doi.org/10.1016/j.aiia.2023.02.003" target="_blank">https://doi.org/10.1016/j.aiia.2023.02.003</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib83"><label>83</label><mixed-citation>
      
Stock, B. C., Jackson, A. L., Ward, E. J., Parnell, A. C., Phillips, D. L.,
and Semmens, B. X.: Analyzing mixing systems using a new generation of
Bayesian tracer mixing models, PeerJ, 6, e5096,
<a href="https://doi.org/10.7717/peerj.5096" target="_blank">https://doi.org/10.7717/peerj.5096</a>, 2018.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib84"><label>84</label><mixed-citation>
      
Su, Y., Wu, Z., Zheng, X., Qiu, Y., Ma, Z., Ren, Y., and Bai, Y.:
Harmonizing remote sensing and ground data for forest aboveground biomass
estimation, Ecol. Inform., 86, 103002,
<a href="https://doi.org/10.1016/j.ecoinf.2025.103002" target="_blank">https://doi.org/10.1016/j.ecoinf.2025.103002</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib85"><label>85</label><mixed-citation>
      
Sun, H., Zheng, W., Wang, S., Ma, L., Min, L., and Shen, Y.: Variation of
nitrate sources affected by precipitation with different intensities in
groundwater in the piedmont plain area of alluvial-pluvial fan, J. Environ.
Manage., 367, 121885, <a href="https://doi.org/10.1016/j.jenvman.2024.121885" target="_blank">https://doi.org/10.1016/j.jenvman.2024.121885</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib86"><label>86</label><mixed-citation>
      
Tang, W. and Carey, S. K.: Classifying annual daily hydrographs in Western
North America using t-distributed stochastic neighbour embedding, Hydrol.
Process., 36, e14473, <a href="https://doi.org/10.1002/hyp.14473" target="_blank">https://doi.org/10.1002/hyp.14473</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib87"><label>87</label><mixed-citation>
      
Tanner, K. B., Cardall, A. C., and Williams, G. P.: A spatial long-term
trend analysis of estimated chlorophyll-a concentrations in Utah Lake using
Earth observation data, Remote Sens., 14, 3664,
<a href="https://doi.org/10.3390/rs14153664" target="_blank">https://doi.org/10.3390/rs14153664</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib88"><label>88</label><mixed-citation>
      
Thunyawatcharakul, P., Cho, K. H., and Chotpantarat, S.: Predicting Arsenic
Speciation in Coastal Aquifers Using Machine Learning: A Case Study of the
Chonburi and Rayong Groundwater Basins, Thailand, ACS ES&amp;T Water, 5,
5011–5024, <a href="https://doi.org/10.1021/acsestwater.4c01082" target="_blank">https://doi.org/10.1021/acsestwater.4c01082</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib89"><label>89</label><mixed-citation>
      
Tian, D., Zhao, X., Gao, L., Jiang, T., Liang, Z., Yang, Z., Zhang, P., Wu,
Q., Ren, K., Yang, C., Li, R., Li, S., Cao, Y., Xuan, Y., Chen, J., and Zhu,
A.: A framework for tracing the sources of nitrate in surface water through
remote sensing data coupled with machine learning, npj Clean Water, 8, 43,
<a href="https://doi.org/10.1038/s41545-025-00473-3" target="_blank">https://doi.org/10.1038/s41545-025-00473-3</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib90"><label>90</label><mixed-citation>
      
Tollefson, J.: Google AI model creates maps of Earth `at any place and
time', Nature, 644, 313, <a href="https://doi.org/10.1038/d41586-025-02412-1" target="_blank">https://doi.org/10.1038/d41586-025-02412-1</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib91"><label>91</label><mixed-citation>
      
Torres-Martínez, J. A., Mora, A., Mahlknecht, J., Daessle, L. W.,
Cervantes-Avilés, P. A., and Ledesma-Ruiz, R. L.: Estimation of nitrate
pollution sources and transformations in groundwater of an intensive
livestock-agricultural area (Comarca Lagunera), combining major ions, stable
isotopes and MixSIAR model, Environ. Pollut., 269, 115445,
<a href="https://doi.org/10.1016/j.envpol.2020.115445" target="_blank">https://doi.org/10.1016/j.envpol.2020.115445</a>, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib92"><label>92</label><mixed-citation>
      
Tung, P. Y., Sheikh, H. A., Ball, M., Nabiei, F., and Harrison, R.: SIGMA:
Spectral interpretation using gaussian mixtures and autoencoder, Geochem.
Geophy. Geosy., 24, e2022GC010530, <a href="https://doi.org/10.1029/2022GC010530" target="_blank">https://doi.org/10.1029/2022GC010530</a>,
2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib93"><label>93</label><mixed-citation>
      
Udu, A. G., Salman, M. T., Ghalati, M. K., Lecchini-Visintini, A., Siddle,
D., and Dong, H.: Emerging SMOTE and GAN-variants for Data Augmentation in
Imbalance Machine Learning Tasks: A Review, IEEE Access, 13, 113838–113853,
<a href="https://doi.org/10.1109/ACCESS.2025.3584532" target="_blank">https://doi.org/10.1109/ACCESS.2025.3584532</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib94"><label>94</label><mixed-citation>
      
Van Katwyk, P., Fox-Kemper, B., Seroussi, H., Nowicki, S., and Bergen, K.: A
variational LSTM emulator of sea level contribution from the Antarctic ice
sheet, J. Adv. Model. Earth Sy., 15, e2023MS003899,
<a href="https://doi.org/10.1029/2023MS003899" target="_blank">https://doi.org/10.1029/2023MS003899</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib95"><label>95</label><mixed-citation>
      
Vedavyasa, K. V. and Kumar, A.: Classification Analysis of Transition Metal
Compounds Using Quantum Machine Learning, Adv. Quantum Technol., 8, 2400081,
<a href="https://doi.org/10.1002/qute.202400081" target="_blank">https://doi.org/10.1002/qute.202400081</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib96"><label>96</label><mixed-citation>
      
Wang, S. Q., Zheng, W. B., and Kong, X. L.: Spatial distribution
characteristics of nitrate in shallow groundwater of the agricultural area
of the North China Plain, Chinese Journal of Eco-Agriculture, 26,
1476–1482, <a href="https://doi.org/10.13930/j.cnki.cjea.180639" target="_blank">https://doi.org/10.13930/j.cnki.cjea.180639</a>, 2018.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib97"><label>97</label><mixed-citation>
      
Wang, S., Chen, J., Zhang, S., Zhang, X., Chen, D., and Zhou, J.:
Hydrochemical evolution characteristics, controlling factors, and high
nitrate hazards of shallow groundwater in a typical agricultural area of
Nansi Lake Basin, North China, Environ. Res., 223, 115430,
<a href="https://doi.org/10.1016/j.envres.2023.115430" target="_blank">https://doi.org/10.1016/j.envres.2023.115430</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib98"><label>98</label><mixed-citation>
      
Wang, C., Yang, J., and Zhang, B.: A fault diagnosis method using improved
prototypical network and weighting similarity-Manhattan distance with
insufficient noisy data, Measurement, 226, 114171,
<a href="https://doi.org/10.1016/j.measurement.2024.114171" target="_blank">https://doi.org/10.1016/j.measurement.2024.114171</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib99"><label>99</label><mixed-citation>
      
Wang, J., Hao, X., Liu, X., Ouyang, W., Li, T., Cui, X., Pei, J., Zhang, S.,
Zhu, W., and Jin, R.: Groundwater–surface water exchange affects nitrate
fate in a seasonal freeze–thaw watershed: Sources, migration and removal,
J. Hydrol., 654, 132803, <a href="https://doi.org/10.1016/j.jhydrol.2025.132803" target="_blank">https://doi.org/10.1016/j.jhydrol.2025.132803</a>,
2025a.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib100"><label>100</label><mixed-citation>
      
Wang, J., Wang, M., Zhang, C., Fan, J., and Lv, F.: Vertical Migration
Characteristics and Effect Factors of BaP in Contaminated Soil Under
Rainwater Infiltration, Water Air Soil Poll., 236, 915,
<a href="https://doi.org/10.1007/s11270-025-08578-8" target="_blank">https://doi.org/10.1007/s11270-025-08578-8</a>, 2025b.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib101"><label>101</label><mixed-citation>
      
Wang, N., Zhou, Q., Gao, J., and Wang, Z.: Evaluating the efficacy of PCA
and t-SNE in optimizing input features for groundwater level simulation
using machine learning models, Environ. Earth Sci., 84, 336,
<a href="https://doi.org/10.1007/s12665-025-12357-3" target="_blank">https://doi.org/10.1007/s12665-025-12357-3</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib102"><label>102</label><mixed-citation>
      
Weller, D. L., Murphy, C. M., Johnson, S., Green, H., Michalenko, E., Love,
T. M., and Strawn, L. K.: Land use, weather, and water quality factors
associated with fecal contamination of northeastern streams that span an
urban-rural gradient, Front. Water, 3, 741676,
<a href="https://doi.org/10.3389/frwa.2021.741676" target="_blank">https://doi.org/10.3389/frwa.2021.741676</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib103"><label>103</label><mixed-citation>
      
Williams, M. R., Ford, W. I., and Mumbi, R. C. K.: Preferential flow in the
shallow vadose zone: Effect of rainfall intensity, soil moisture,
connectivity, and agricultural management, Hydrol. Process., 37, 15057,
<a href="https://doi.org/10.1002/hyp.15057" target="_blank">https://doi.org/10.1002/hyp.15057</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib104"><label>104</label><mixed-citation>
      
Wu, H., Song, F., Min, L., Li, J., Shen, Y., Huang, Y., Fan, H., Liu, J.,
and Fu, S.: Exploring recharge mechanisms of soil water in the thick
unsaturated zone using water isotopes in the North China Plain, Catena, 234,
107615, <a href="https://doi.org/10.1016/j.catena.2023.107615" target="_blank">https://doi.org/10.1016/j.catena.2023.107615</a>, 2024.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib105"><label>105</label><mixed-citation>
      
Wu, Y., Wang, J., Liu, Z., Li, C., Niu, Y., and Jiang, X.: Seasonal nitrate
input drives the spatiotemporal variability of regional surface
water-groundwater interactions, nitrate sources and transformations, J.
Hydrol., 655, 132973, <a href="https://doi.org/10.1016/j.jhydrol.2025.132973" target="_blank">https://doi.org/10.1016/j.jhydrol.2025.132973</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib106"><label>106</label><mixed-citation>
      
Xiong, X., Hao, G., Xu, L., Li, Z., Zhang, X., Lu, J., and Zhang, Z.:
Characteristics of spatial distributions for nitrate and traceability of
pollution in shallow groundwater of the Qingshui River Basin, China, Hydrol.
Res., 56, 920–936, <a href="https://doi.org/10.2166/nh.2025.047" target="_blank">https://doi.org/10.2166/nh.2025.047</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib107"><label>107</label><mixed-citation>
      
Xu, G., Su, X., Yuan, Z., Ji, L., Li, N., and Liang, H.: Nitrogen behavior
during artificial groundwater recharge through ponds: A case study in
Xiong'an New Area, Environ. Geochem. Hlth., 44, 2545–2561,
<a href="https://doi.org/10.1007/s10653-021-01041-7" target="_blank">https://doi.org/10.1007/s10653-021-01041-7</a>, 2021.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib108"><label>108</label><mixed-citation>
      
Xu, J.: sherlockjjobs/Quantum-RF-framework: Quantum-RF-framework (Version groundwater-nitrate-prediction), Zenodo [data set] and [code], <a href="https://doi.org/10.5281/zenodo.22329075" target="_blank">https://doi.org/10.5281/zenodo.22329075</a>,  2026.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib109"><label>109</label><mixed-citation>
      
Yan, R. and Huang, J. J.: Confident learning-based Gaussian mixture model
for leakage detection in water distribution networks, Water Res., 247,
120773, <a href="https://doi.org/10.1016/j.watres.2023.120773" target="_blank">https://doi.org/10.1016/j.watres.2023.120773</a>, 2023.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib110"><label>110</label><mixed-citation>
      
Yang, W., Long, D., Scanlon, B. R., Burek, P., Zhang, C., Han, Z., Butler,
J., Pan, Y., Lei, X., and Wada, Y.: Human intervention will stabilize
groundwater storage across the North China Plain, Water Resour. Res., 58,
e2021WR030884, <a href="https://doi.org/10.1029/2021WR030884" target="_blank">https://doi.org/10.1029/2021WR030884</a>, 2022.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib111"><label>111</label><mixed-citation>
      
Zhang, M., Zhi, Y. Y., Shi, J. C., and Wu, L. S.: Apportionment and
uncertainty analysis of nitrate sources based on the dual isotope approach
and a Bayesian isotope mixing model at the watershed scale, Sci. Total
Environ., 639, 1175–1187, <a href="https://doi.org/10.1016/j.scitotenv.2018.05.239" target="_blank">https://doi.org/10.1016/j.scitotenv.2018.05.239</a>,
2018.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib112"><label>112</label><mixed-citation>
      
Zhang, W., Xin, C., and Yu, S.: A review of heavy metal migration and its
influencing factors in karst groundwater, Northern and Southern China,
Water, 15, 3690, <a href="https://doi.org/10.3390/w15203690" target="_blank">https://doi.org/10.3390/w15203690</a>, 2023.


    </mixed-citation></ref-html>
<ref-html id="bib1.bib113"><label>113</label><mixed-citation>
      
Zhang, J., Zhang, L., Zheng, T., Jin, M., Kang, F., Jiang, J., Yuan, Z., and
Luo, J.: Tracing Nitrate Contamination Sources and Transformations in a
Rural-Urban Karst Groundwater System in North China Using Multiple Isotopes
and Simmr Modeling, Water Resour. Res., 61, e2025WR040156,
<a href="https://doi.org/10.1029/2025WR040156" target="_blank">https://doi.org/10.1029/2025WR040156</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib114"><label>114</label><mixed-citation>
      
Zhao, Y., Liu, J., Zhang, X., Li, Q., and Wu, J.: Integrated Machine
Learning and Health Risk Assessment for Groundwater Nitrate Contamination in
Handan City, China, Water, 18, 1174, <a href="https://doi.org/10.3390/w18101174" target="_blank">https://doi.org/10.3390/w18101174</a>,
2026.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib115"><label>115</label><mixed-citation>
      
Zheng, Y., Zhang, X., Zhou, Y., Zhang, Y., Zhang, T., and Farmani, R.: Deep
representation learning enables cross-basin water quality prediction under
data-scarce conditions, npj Clean Water, 8, 33,
<a href="https://doi.org/10.1038/s41545-025-00466-2" target="_blank">https://doi.org/10.1038/s41545-025-00466-2</a>, 2025.

    </mixed-citation></ref-html>
<ref-html id="bib1.bib116"><label>116</label><mixed-citation>
      
Zhu, J. J., Yang, M., and Ren, Z. J.: Machine learning in environmental
research: common pitfalls and best practices, Environ. Sci. Technol., 57,
17671–17689, <a href="https://doi.org/10.1021/acs.est.3c00026" target="_blank">https://doi.org/10.1021/acs.est.3c00026</a>, 2023.

    </mixed-citation></ref-html>--></article>
