<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.4 20241031//EN" "JATS-journalpublishing1-4.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.4" xml:lang="en">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">Oalib</journal-id>
      <journal-title-group>
        <journal-title>Open Access Library Journal</journal-title>
      </journal-title-group>
      <issn pub-type="epub">2333-9721</issn>
      <issn pub-type="ppub">2333-9705</issn>
      <publisher>
        <publisher-name>Scientific Research Publishing</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.4236/oalib.1115081</article-id>
      <article-id pub-id-type="publisher-id">Oalib-154348</article-id>
      <article-categories>
        <subj-group>
          <subject>Article</subject>
        </subj-group>
        <subj-group>
          <subject>Biomedical</subject>
          <subject>Life Sciences</subject>
          <subject>Business</subject>
          <subject>Economics</subject>
          <subject>Chemistry</subject>
          <subject>Materials Science</subject>
          <subject>Computer Science</subject>
          <subject>Communications</subject>
          <subject>Earth</subject>
          <subject>Environmental Sciences</subject>
          <subject>Engineering</subject>
          <subject>Medicine</subject>
          <subject>Healthcare</subject>
          <subject>Physics</subject>
          <subject>Mathematics</subject>
          <subject>Social Sciences</subject>
          <subject>Humanities</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>SatMAE-Agri: Masked Spatiotemporal Autoencoding for Self-Supervised Learning on Satellite Image Time Series</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <name name-style="western">
            <surname>Sabiraguha</surname>
            <given-names>Aimé-Emmanuel</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <contrib-id contrib-id-type="orcid">0000-0002-9727-7182</contrib-id>
          <name name-style="western">
            <surname>Sindayigaya</surname>
            <given-names>Ildephonse</given-names>
          </name>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <name name-style="western">
            <surname>Havyarimana</surname>
            <given-names>Vincent</given-names>
          </name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <name name-style="western">
            <surname>Kamdjoug</surname>
            <given-names>Jean Robert Kala</given-names>
          </name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <name name-style="western">
            <surname>Niyongabo</surname>
            <given-names>Prime</given-names>
          </name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <name name-style="western">
            <surname>Haremarugira</surname>
            <given-names>Sylvain</given-names>
          </name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
      </contrib-group>
      <aff id="aff1"><label>1</label> Ecole Doctorale, Université du Burundi, Bujumbura, Burundi </aff>
      <aff id="aff2"><label>2</label> Faculté de Droit, Université Lumière de Bujumbura, Bujumbura, Burundi </aff>
      <aff id="aff3"><label>3</label> Centre de Recherche en Droits et Bien-être de l’Enfant, Ecole Doctorale, Université du Burundi, Bujumbura, Burundi </aff>
      <aff id="aff4"><label>4</label> Ecole Normale Supérieure (ENS), Bujumbura, Burundi </aff>
      <aff id="aff5"><label>5</label> Douala Campus, Université Catholique d’Afrique Centrale (UCAC), Yaoundé, Cameroun </aff>
      <aff id="aff6"><label>6</label> Institut Supérieur des Cadres Militaires (ISCAM), Bujumbura, Burundi </aff>
      <aff id="aff7"><label>7</label> Institut des Sciences Agronomiques du Burundi, Université du Burundi, Gitega, Burundi </aff>
      <author-notes>
        <fn fn-type="conflict" id="fn-conflict">
          <p>The authors declare no conflicts of interest regarding the publication of this paper.</p>
        </fn>
      </author-notes>
      <pub-date pub-type="epub">
        <day>02</day>
        <month>09</month>
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="collection">
        <month>09</month>
        <year>2026</year>
      </pub-date>
      <volume>13</volume>
      <issue>09</issue>
      <fpage>1</fpage>
      <lpage>10</lpage>
      <history>
        <date date-type="received">
          <day>28</day>
          <month>02</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>27</day>
          <month>09</month>
          <year>2026</year>
        </date>
        <date date-type="published">
          <day>30</day>
          <month>09</month>
          <year>2026</year>
        </date>
      </history>
      <permissions>
        <copyright-statement>© 2026 by the authors and Scientific Research Publishing Inc.</copyright-statement>
        <copyright-year>2026</copyright-year>
        <license license-type="open-access">
          <license-p> This article is an open access article distributed under the terms and conditions of the Creative Commons Attribution (CC BY) license ( <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link> ). </license-p>
        </license>
      </permissions>
      <self-uri content-type="doi" xlink:href="https://doi.org/10.4236/oalib.1115081">https://doi.org/10.4236/oalib.1115081</self-uri>
      <abstract>
        <p>Satellite image time series (SITS) provide valuable information for agricultural monitoring, yet supervised learning approaches remain limited by the scarcity of labeled data, particularly in developing regions. To address this challenge, we propose SatMAE-Agri, a masked spatiotemporal autoencoder for self-supervised representation learning from multi-temporal Sentinel-2 satellite imagery. The proposed method extends masked autoencoding to spatiotemporal remote sensing data by jointly modeling spatial structure and temporal evolution. Satellite images are divided into non-overlapping patches and embedded into a latent space, where both spatial and temporal positional encodings are added. A high proportion of spatiotemporal tokens is randomly masked, and the encoder processes only the visible tokens. A lightweight decoder then reconstructs the masked patches, enabling the model to learn meaningful representations without manual annotations. We evaluate the method on Sentinel-2 image time series over agricultural regions in Burundi. Experimental results show that the model successfully reconstructs heavily masked patches and captures consistent spatial and temporal patterns across crop fields. The learned representations are suitable for downstream agricultural tasks such as crop classification and change detection. This work demonstrates that masked spatiotemporal modeling is a promising direction for label-efficient learning in satellite-based agricultural monitoring.</p>
      </abstract>
      <kwd-group kwd-group-type="author-generated" xml:lang="en">
        <kwd>Masked Autoencoder</kwd>
        <kwd>Self-Supervised Learning</kwd>
        <kwd>Satellite Image Time Series</kwd>
        <kwd>Sentinel-2</kwd>
        <kwd>Spatiotemporal Modeling</kwd>
        <kwd>Remote Sensing</kwd>
        <kwd>Agricultural Monitoring</kwd>
        <kwd>Representation Learning</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec1">
      <title>1. Introduction</title>
      <p>Satellite image time series (SITS) provide rich information for monitoring agricultural dynamics, crop phenology, and land-use changes. However, obtaining labeled data for large-scale agricultural analysis remains expensive and time-consuming. This makes self-supervised learning (SSL) [<xref ref-type="bibr" rid="B1">1</xref>][<xref ref-type="bibr" rid="B2">2</xref>] an attractive solution for learning useful representations without manual annotation.</p>
      <p>Recently, masked autoencoding has emerged as a powerful SSL strategy in computer vision [<xref ref-type="bibr" rid="B2">2</xref>], where parts of the input are hidden, and the model learns to reconstruct them. While this idea has been widely explored for natural images, its application to spatiotemporal satellite data remains underexplored, especially for agricultural regions in developing countries.</p>
      <p>In this work, we propose SatMAE-Agri, a Masked Spatiotemporal Autoencoder designed for Sentinel-2 satellite image time series over agricultural areas in Burundi [<xref ref-type="bibr" rid="B3">3</xref>].</p>
      <p>Our contributions are:</p>
      <p>A spatiotemporal masked autoencoder tailored for satellite image time series.A patch-based embedding strategy for multi-spectral Sentinel-2 data.A high-ratio masking scheme encouraging robust temporal-spatial representation learning.A self-supervised training pipeline for agricultural monitoring without labels.</p>
      <p>Self-supervised learning methods such as contrastive learning [<xref ref-type="bibr" rid="B4">4</xref>] and masked modeling have significantly improved representation learning without labels [<xref ref-type="bibr" rid="B4">4</xref>]. Masked Autoencoders (MAE) demonstrated that reconstructing missing image patches enables efficient learning of visual features [<xref ref-type="bibr" rid="B4">4</xref>].</p>
      <p>Recent works have adapted SSL to remote sensing imagery. However, many approaches focus on single-date images rather than time series. Satellite image time series introduce temporal dependencies that standard MAE architectures do not fully exploit.</p>
      <p>Yet, few methods combine temporal modeling with masked reconstruction in a unified SSL framework [<xref ref-type="bibr" rid="B5">5</xref>][<xref ref-type="bibr" rid="B6">6</xref>].</p>
      <p>Our method bridges this gap by extending masked autoencoding to multi-temporal, multi-spectral satellite data.</p>
    </sec>
    <sec id="sec2">
      <title>2. Methods and Methodology</title>
      <p>We propose a Masked Spatiotemporal Autoencoder (SatMAE-Agri) that learns representations from satellite image sequences by reconstructing masked patches.</p>
      <p><xref ref-type="fig" rid="fig1">Figure 1</xref><xref ref-type="fig" rid="fig1">Figure 1</xref> illustrates the overall architecture of the proposed SatMAE-Agri framework. The model follows a masked spatiotemporal autoencoding strategy where satellite image time series are divided into patches, embedded, partially masked, and reconstructed through a Transformer-based encoder-decoder architecture [<xref ref-type="bibr" rid="B7">7</xref>][<xref ref-type="bibr" rid="B8">8</xref>].</p>
      <fig id="fig1">
        <label>Figure 1</label>
        <graphic xlink:href="https://html.scirp.org/file/1115081-rId17.jpeg?20260930012352" />
      </fig>
      <p><bold>Figure 1.</bold>Overview of the SatMAE-Agri masked spatiotemporal autoencoder framework.</p>
      <sec id="sec2dot1">
        <title>2.1. Patch Embedding</title>
        <p>Each satellite image frame has shape:</p>
        <disp-formula id="FD1">
          <mml:math>
            <mml:mrow>
              <mml:mi>T</mml:mi>
              <mml:mo>×</mml:mo>
              <mml:mi>C</mml:mi>
              <mml:mo>×</mml:mo>
              <mml:mi>H</mml:mi>
              <mml:mo>×</mml:mo>
              <mml:mi>W</mml:mi>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>where</p>
        <p><italic>T</italic> = number of time steps.<italic>C</italic> = number of spectral bands.<italic>H</italic>, <italic>W</italic> = spatial dimensions.</p>
        <disp-formula id="FD2">
          <mml:math display="inline">
            <mml:mrow>
              <mml:mi>N</mml:mi>
              <mml:mo>=</mml:mo>
              <mml:mfrac>
                <mml:mi>H</mml:mi>
                <mml:mi>P</mml:mi>
              </mml:mfrac>
              <mml:mo>×</mml:mo>
              <mml:mfrac>
                <mml:mi>W</mml:mi>
                <mml:mi>P</mml:mi>
              </mml:mfrac>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>Each patch is flattened into a vector of size <inline-formula><mml:math display="inline"><mml:mrow><mml:msup><mml:mi> P </mml:mi><mml:mn> 2 </mml:mn></mml:msup><mml:mo> ⋅ </mml:mo><mml:mi> C </mml:mi></mml:mrow></mml:math></inline-formula> and projected into a latent embedding space using a linear layer:</p>
        <disp-formula id="FD3">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mi>x</mml:mi>
                <mml:mrow>
                  <mml:mi>t</mml:mi>
                  <mml:mo>,</mml:mo>
                  <mml:mi>i</mml:mi>
                </mml:mrow>
              </mml:msub>
              <mml:mo>∈</mml:mo>
              <mml:msup>
                <mml:mi>ℝ</mml:mi>
                <mml:mrow>
                  <mml:msup>
                    <mml:mi>P</mml:mi>
                    <mml:mn>2</mml:mn>
                  </mml:msup>
                  <mml:mi>C</mml:mi>
                </mml:mrow>
              </mml:msup>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <disp-formula id="FD4">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mi>z</mml:mi>
                <mml:mrow>
                  <mml:mi>t</mml:mi>
                  <mml:mo>,</mml:mo>
                  <mml:mi>i</mml:mi>
                </mml:mrow>
              </mml:msub>
              <mml:mo>=</mml:mo>
              <mml:msub>
                <mml:mi>W</mml:mi>
                <mml:mi>p</mml:mi>
              </mml:msub>
              <mml:msub>
                <mml:mi>x</mml:mi>
                <mml:mrow>
                  <mml:mi>t</mml:mi>
                  <mml:mo>,</mml:mo>
                  <mml:mi>i</mml:mi>
                </mml:mrow>
              </mml:msub>
              <mml:mo>+</mml:mo>
              <mml:msub>
                <mml:mi>b</mml:mi>
                <mml:mi>p</mml:mi>
              </mml:msub>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>where:</p>
        <p><inline-formula><mml:math display="inline"><mml:mrow><mml:msub><mml:mi> x </mml:mi><mml:mrow><mml:mi> t </mml:mi><mml:mo> , </mml:mo><mml:mi> i </mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the flattened patch at time step <inline-formula><mml:math><mml:mi> t </mml:mi></mml:math></inline-formula> and spatial index <inline-formula><mml:math><mml:mi> i </mml:mi></mml:math></inline-formula> .<inline-formula><mml:math display="inline"><mml:mrow><mml:msub><mml:mi> W </mml:mi><mml:mi> p </mml:mi></mml:msub><mml:mo> ∈ </mml:mo><mml:msup><mml:mi> ℝ </mml:mi><mml:mrow><mml:mi> D </mml:mi><mml:mo> × </mml:mo><mml:msup><mml:mi> P </mml:mi><mml:mn> 2 </mml:mn></mml:msup><mml:mi> C </mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> is the learnable projection matrix.<inline-formula><mml:math display="inline"><mml:mrow><mml:msub><mml:mi> z </mml:mi><mml:mrow><mml:mi> t </mml:mi><mml:mo> , </mml:mo><mml:mi> i </mml:mi></mml:mrow></mml:msub><mml:mo> ∈ </mml:mo><mml:msup><mml:mi> ℝ </mml:mi><mml:mi> D </mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> is the patch embedding.</p>
        <disp-formula id="FD5">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mstyle mathvariant="bold" mathsize="normal">
                  <mml:mi>z</mml:mi>
                </mml:mstyle>
                <mml:mrow>
                  <mml:mi>t</mml:mi>
                  <mml:mo>,</mml:mo>
                  <mml:mi>i</mml:mi>
                </mml:mrow>
              </mml:msub>
              <mml:mo>=</mml:mo>
              <mml:mtext>Linear</mml:mtext>
              <mml:mrow>
                <mml:mo>(</mml:mo>
                <mml:mrow>
                  <mml:msub>
                    <mml:mrow>
                      <mml:mtext>Patch</mml:mtext>
                    </mml:mrow>
                    <mml:mrow>
                      <mml:mi>t</mml:mi>
                      <mml:mo>,</mml:mo>
                      <mml:mi>i</mml:mi>
                    </mml:mrow>
                  </mml:msub>
                </mml:mrow>
                <mml:mo>)</mml:mo>
              </mml:mrow>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>This converts the satellite sequence into a sequence of spatiotemporal tokens.</p>
        <p><xref ref-type="fig" rid="fig2">Figure 2</xref><xref ref-type="fig" rid="fig2">Figure 2</xref> illustrates how Sentinel-2 images are split into spatial patches that are later converted into tokens.</p>
        <fig id="fig2">
          <label>Figure 2</label>
          <graphic xlink:href="https://html.scirp.org/file/1115081-rId40.jpeg?20260930012352" />
        </fig>
        <p><xref ref-type="fig" rid="fig2">Figure 2</xref><bold>.</bold>Example of patch extraction from a Sentinel-2 image. </p>
        <p>The image is divided into non-overlapping patches that are flattened and projected into embedding vectors.</p>
        <p>Each patch is flattened into a vector of size <inline-formula><mml:math display="inline"><mml:mrow><mml:msup><mml:mi> P </mml:mi><mml:mn> 2 </mml:mn></mml:msup><mml:mo> ⋅ </mml:mo><mml:mi> C </mml:mi></mml:mrow></mml:math></inline-formula> and projected into a latent embedding space using a linear layer. This converts the satellite sequence into a sequence of spatiotemporal tokens.</p>
      </sec>
      <sec id="sec2dot2">
        <title>2.2. Positional Encoding</title>
        <p>To preserve spatial and temporal order, we add spatiotemporal positional encodings.</p>
        <p>Each token embedding receives:</p>
        <p>A spatial positional embedding indicating its location within the image grid.A temporal positional embedding indicating its time step.</p>
        <p>The final token representation is:</p>
        <disp-formula id="FD6">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mstyle mathvariant="bold" mathsize="normal">
                  <mml:mi>h</mml:mi>
                </mml:mstyle>
                <mml:mrow>
                  <mml:mi>t</mml:mi>
                  <mml:mo>,</mml:mo>
                  <mml:mi>i</mml:mi>
                </mml:mrow>
              </mml:msub>
              <mml:mo>=</mml:mo>
              <mml:msub>
                <mml:mstyle mathvariant="bold" mathsize="normal">
                  <mml:mi>z</mml:mi>
                </mml:mstyle>
                <mml:mrow>
                  <mml:mi>t</mml:mi>
                  <mml:mo>,</mml:mo>
                  <mml:mi>i</mml:mi>
                </mml:mrow>
              </mml:msub>
              <mml:mo>+</mml:mo>
              <mml:msubsup>
                <mml:mstyle mathvariant="bold" mathsize="normal">
                  <mml:mi>p</mml:mi>
                </mml:mstyle>
                <mml:mi>i</mml:mi>
                <mml:mrow>
                  <mml:mtext>space</mml:mtext>
                </mml:mrow>
              </mml:msubsup>
              <mml:mo>+</mml:mo>
              <mml:msubsup>
                <mml:mstyle mathvariant="bold" mathsize="normal">
                  <mml:mi>p</mml:mi>
                </mml:mstyle>
                <mml:mi>t</mml:mi>
                <mml:mrow>
                  <mml:mtext>time</mml:mtext>
                </mml:mrow>
              </mml:msubsup>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>where:</p>
        <p><inline-formula><mml:math display="inline"><mml:mrow><mml:msubsup><mml:mi> e </mml:mi><mml:mi> i </mml:mi><mml:mrow><mml:mtext> space </mml:mtext></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is the spatial positional embedding.<inline-formula><mml:math display="inline"><mml:mrow><mml:msubsup><mml:mi> e </mml:mi><mml:mi> t </mml:mi><mml:mrow><mml:mtext> time </mml:mtext></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is the temporal positional embedding.</p>
        <p>This enables the model to understand where and when each patch occurs.</p>
      </sec>
      <sec id="sec2dot3">
        <title>2.3. Masking Strategy</title>
        <p>We apply random masking to a high percentage (e.g., 75%) of tokens across both space and time.</p>
        <p>Steps:</p>
        <p>1) Flatten all spatiotemporal tokens into a sequence.</p>
        <p>2) Randomly select a subset to keep (visible tokens).</p>
        <p>3) Mask the remaining tokens.</p>
        <p>4) Encode only visible tokens.</p>
        <p>A high masking ratio forces the model to:</p>
        <p>Learn global spatial context.Exploit temporal correlations.Develop robust representations rather than memorizing textures.</p>
        <p>Let <inline-formula><mml:math display="inline"><mml:mrow><mml:mi mathvariant="script"> Z </mml:mi><mml:mo> = </mml:mo><mml:mrow><mml:mo> { </mml:mo><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi> z </mml:mi><mml:mo> ˜ </mml:mo></mml:mover><mml:mrow><mml:mi> t </mml:mi><mml:mo> , </mml:mo><mml:mi> i </mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo> } </mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> be the full token set.</p>
        <p>We randomly sample a visible subset:</p>
        <disp-formula id="FD7">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mi mathvariant="script">Z</mml:mi>
                <mml:mrow>
                  <mml:mtext>vis</mml:mtext>
                </mml:mrow>
              </mml:msub>
              <mml:mo>⊂</mml:mo>
              <mml:mi mathvariant="script">Z</mml:mi>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>The masked set is:</p>
        <disp-formula id="FD8">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mi mathvariant="script">Z</mml:mi>
                <mml:mrow>
                  <mml:mtext>mask</mml:mtext>
                </mml:mrow>
              </mml:msub>
              <mml:mo>=</mml:mo>
              <mml:mi mathvariant="script">Z</mml:mi>
              <mml:mo>\</mml:mo>
              <mml:msub>
                <mml:mi mathvariant="script">Z</mml:mi>
                <mml:mrow>
                  <mml:mtext>vis</mml:mtext>
                </mml:mrow>
              </mml:msub>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>The encoder processes only</p>
        <disp-formula id="FD9">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mi mathvariant="script">Z</mml:mi>
                <mml:mrow>
                  <mml:mtext>vis</mml:mtext>
                </mml:mrow>
              </mml:msub>
            </mml:mrow>
          </mml:math>
        </disp-formula>
      </sec>
      <sec id="sec2dot4">
        <title>2.4. Architecture</title>
        <p>The model consists of:</p>
        <p>Encoder</p>
        <p>Input: visible spatiotemporal tokens.Backbone: Transformer encoder layers.Output: latent representation of visible tokens.</p>
        <p>Decoder</p>
        <p>Mask tokens are reinserted.Full token sequence (visible + mask tokens) is processed.The decoder predicts original pixel values of masked patches.</p>
        <p>The architecture is lightweight on the encoder side and heavier on the decoder, making training efficient.</p>
      </sec>
      <sec id="sec2dot5">
        <title>2.5. Experimental Setup</title>
        <p>2.5.1. Dataset</p>
        <p>We use Sentinel-2 satellite image time series over agricultural regions in Burundi. Each sample consists of:</p>
        <p>Multi-spectral bands (e.g., 10 bands).Multiple time steps (e.g., 3 - 6 dates).Patches extracted from the same geographic location.</p>
        <p>80% of patch locations are used for training, and 20% for validation.</p>
        <p>2.5.2. Preprocessing</p>
        <p>Cloud filtering (if available).Per-band normalization (mean/std).Patch extraction (e.g., 32 × 32 pixels).</p>
        <p>2.5.3. Training Details</p>
        <p>Loss: Mean Squared Error (MSE) on masked patches.</p>
        <p>The reconstruction loss is computed only on masked patches:</p>
        <disp-formula id="FD10">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msub>
                <mml:mi>ℒ</mml:mi>
                <mml:mrow>
                  <mml:mtext>rec</mml:mtext>
                </mml:mrow>
              </mml:msub>
              <mml:mo>=</mml:mo>
              <mml:mfrac>
                <mml:mn>1</mml:mn>
                <mml:mrow>
                  <mml:mrow>
                    <mml:mo>|</mml:mo>
                    <mml:mrow>
                      <mml:msub>
                        <mml:mi mathvariant="script">Z</mml:mi>
                        <mml:mrow>
                          <mml:mtext>mask</mml:mtext>
                        </mml:mrow>
                      </mml:msub>
                    </mml:mrow>
                    <mml:mo>|</mml:mo>
                  </mml:mrow>
                </mml:mrow>
              </mml:mfrac>
              <mml:munder>
                <mml:mstyle displaystyle="true" mathsize="140%">
                  <mml:mo>∑</mml:mo>
                </mml:mstyle>
                <mml:mrow>
                  <mml:mrow>
                    <mml:mo>(</mml:mo>
                    <mml:mrow>
                      <mml:mi>t</mml:mi>
                      <mml:mo>,</mml:mo>
                      <mml:mi>i</mml:mi>
                    </mml:mrow>
                    <mml:mo>)</mml:mo>
                  </mml:mrow>
                  <mml:mo>∈</mml:mo>
                  <mml:msub>
                    <mml:mi mathvariant="script">Z</mml:mi>
                    <mml:mrow>
                      <mml:mtext>mask</mml:mtext>
                    </mml:mrow>
                  </mml:msub>
                </mml:mrow>
              </mml:munder>
              <mml:msubsup>
                <mml:mrow>
                  <mml:mrow>
                    <mml:mo>‖</mml:mo>
                    <mml:mrow>
                      <mml:msub>
                        <mml:mi>x</mml:mi>
                        <mml:mrow>
                          <mml:mi>t</mml:mi>
                          <mml:mo>,</mml:mo>
                          <mml:mi>i</mml:mi>
                        </mml:mrow>
                      </mml:msub>
                      <mml:mo>−</mml:mo>
                      <mml:msub>
                        <mml:mover accent="true">
                          <mml:mi>x</mml:mi>
                          <mml:mo>^</mml:mo>
                        </mml:mover>
                        <mml:mrow>
                          <mml:mi>t</mml:mi>
                          <mml:mo>,</mml:mo>
                          <mml:mi>i</mml:mi>
                        </mml:mrow>
                      </mml:msub>
                    </mml:mrow>
                    <mml:mo>‖</mml:mo>
                  </mml:mrow>
                </mml:mrow>
                <mml:mn>2</mml:mn>
                <mml:mn>2</mml:mn>
              </mml:msubsup>
            </mml:mrow>
          </mml:math>
        </disp-formula>
        <p>where:</p>
        <p><inline-formula><mml:math display="inline"><mml:mrow><mml:msub><mml:mi> x </mml:mi><mml:mrow><mml:mi> t </mml:mi><mml:mo> , </mml:mo><mml:mi> i </mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the original patch.</p>
        <p><inline-formula><mml:math display="inline"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mi> x </mml:mi><mml:mo> ^ </mml:mo></mml:mover><mml:mrow><mml:mi> t </mml:mi><mml:mo> , </mml:mo><mml:mi> i </mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the reconstructed patch.</p>
        <p>Optimizer: Adam.Batch size: 2 - 16.Masking ratio: 75%.</p>
        <p>The model parameters<italic>θ</italic> are optimized by:</p>
        <disp-formula id="FD11">
          <mml:math display="inline">
            <mml:mrow>
              <mml:msup>
                <mml:mi>θ</mml:mi>
                <mml:mtext>*</mml:mtext>
              </mml:msup>
              <mml:mo>=</mml:mo>
              <mml:mtext>arg</mml:mtext>
              <mml:munder>
                <mml:mrow>
                  <mml:mtext>min</mml:mtext>
                </mml:mrow>
                <mml:mi>θ</mml:mi>
              </mml:munder>
              <mml:msub>
                <mml:mi>ℒ</mml:mi>
                <mml:mrow>
                  <mml:mtext>rec</mml:mtext>
                </mml:mrow>
              </mml:msub>
            </mml:mrow>
          </mml:math>
        </disp-formula>
      </sec>
    </sec>
    <sec id="sec3">
      <title>3. Results</title>
      <p>The model successfully reconstructs masked patches from heavily masked inputs. Even with only 25% visible tokens, reconstructed patches preserve:</p>
      <p>Spatial structures (Good field boundaries, patterns).Temporal consistency across frames.</p>
      <p>Training loss decreases steadily, indicating effective learning of spatiotemporal features.</p>
      <p>Qualitative results show that the model captures agricultural texture and seasonal changes despite the absence of labels.</p>
      <p>To qualitatively evaluate the reconstruction capability of the proposed model, we visualize examples of masked inputs and their reconstructed outputs. Even with a high masking ratio, the model successfully recovers spatial structures and preserves temporal consistency across frames.</p>
      <p>Results from <xref ref-type="fig" rid="fig3">Figure 3</xref><xref ref-type="fig" rid="fig3">Figure 3</xref> show the examples: Left: masked input patches. Middle: reconstructed output. Right: original satellite patches. The model accurately restores spatial patterns and temporal structures despite heavy masking.</p>
      <p>The training process of the proposed SatMAE-Agri model is stable, with the reconstruction loss consistently decreasing over epochs. This indicates that the encoder-decoder architecture successfully learns meaningful spatiotemporal representations from masked satellite patches. </p>
      <p>According to <xref ref-type="fig" rid="fig4">Figure 4</xref><xref ref-type="fig" rid="fig4">Figure 4</xref>, the steady decrease demonstrates effective learning of spatiotemporal features despite the high masking ratio.</p>
      <p>To further evaluate the effectiveness of the learned representations, we analyze additional experimental results. </p>
      <fig id="fig3">
        <label>Figure 3</label>
        <graphic xlink:href="https://html.scirp.org/file/1115081-rId65.jpeg?20260930012352" />
      </fig>
      <p><bold>Figure 3.</bold>Qualitative reconstruction.</p>
      <fig id="fig4">
        <label>Figure 4</label>
        <graphic xlink:href="https://html.scirp.org/file/1115081-rId66.jpeg?20260930012352" />
      </fig>
      <p><bold>Figure 4.</bold>Training reconstruction loss over epochs.</p>
      <fig id="fig5">
        <label>Figure 5</label>
        <graphic xlink:href="https://html.scirp.org/file/1115081-rId67.jpeg?20260930012352" />
      </fig>
      <p><bold>Figure 5.</bold>Performance comparison using features learned by SatMAE-Agri on a downstream agricultural task.</p>
      <p>These results confirm that the proposed masked spatiotemporal modeling approach captures meaningful temporal and spatial information useful for agricultural analysis (see <xref ref-type="fig" rid="fig5">Figure 5</xref><xref ref-type="fig" rid="fig5">Figure 5</xref>).</p>
    </sec>
    <sec id="sec4">
      <title>4. Discussion</title>
      <p>The high masking ratio encourages the model to infer missing information using both spatial context and temporal evolution, which is crucial for agricultural monitoring.</p>
      <p>The learned representations can be transferred to downstream tasks such as:</p>
      <p>Crop type classification.Yield prediction.Change detection.</p>
      <p>A limitation is that reconstruction-based SSL focuses on low-level details; future work could integrate contrastive or predictive objectives.</p>
      <p>Beyond reconstruction quality, we analyze the structure of the learned representations. </p>
      <p><xref ref-type="fig" rid="fig6">Figure 6</xref><xref ref-type="fig" rid="fig6">Figure 6</xref> illustrates the organization of feature embeddings extracted from the encoder, showing that patches with similar temporal behavior tend to cluster together. This suggests that the model captures meaningful spatiotemporal patterns relevant to agricultural dynamics.</p>
      <p>As is clearly shown in <xref ref-type="fig" rid="fig6">Figure 6</xref><xref ref-type="fig" rid="fig6">Figure 6</xref>, similar agricultural patches form coherent clusters, indicating that the model learns semantically meaningful representations.</p>
      <fig id="fig6">
        <label>Figure 6</label>
        <graphic xlink:href="https://html.scirp.org/file/1115081-rId68.jpeg?20260930012352" />
      </fig>
      <p><bold>Figure 6.</bold>Visualization of learned feature embeddings (e.g., PCA/t-SNE).</p>
    </sec>
    <sec id="sec5">
      <title>5. Conclusions</title>
      <p>We proposed SatMAE-Agri, a masked spatiotemporal autoencoder for self-supervised learning on satellite image time series. The method effectively learns representations from agricultural Sentinel-2 data without labels by reconstructing masked patches across space and time.</p>
      <p>This work demonstrates that masked modeling is a promising direction for label-efficient agricultural monitoring in data-scarce regions.</p>
      <p><bold>Author Contributions</bold></p>
      <p>Aimé-Emmanuel Sabiraguha: Conceptualization, investigation, data curation, and writing—original draft preparation. Ildephonse Sindayigaya : Conceptualization, methodology, formal analysis, writing original draft preparation, supervision, and writing—review and editing. Vincent Havyarimana: Conceptualization and data curation. Jean Robert Kala Kamdjoug: Conceptualization and project administration. Prime Niyongabo: Supervision and project administration. Sylvain Haremarugira: Resources, investigation and funding acquisition.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <title>References</title>
      <ref id="B1">
        <label>1.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Sallam, M. and Ali Shnan, M. (2025) Enhancing Semantic Image Retrieval Using Self-Supervised Learning: A Label-Efficient Approach. <italic>Babylonian Journal of Machine Learning</italic>, 2025, 42-60. https://doi.org/10.58496/bjml/2025/004 <pub-id pub-id-type="doi">10.58496/bjml/2025/004</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.58496/bjml/2025/004">https://doi.org/10.58496/bjml/2025/004</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Sallam, M.</string-name>
              <string-name>Shnan, M.</string-name>
            </person-group>
            <year>2025</year>
            <article-title>Enhancing Semantic Image Retrieval Using Self-Supervised Learning: A Label-Efficient Approach</article-title>
            <source>Babylonian Journal of Machine Learning</source>
            <volume>2025</volume>
            <pub-id pub-id-type="doi">10.58496/bjml/2025/004</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B2">
        <label>2.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Mohy, A.A., Bassioni, H.A., Elgendi, E.O. and Hassan, T.M. (2026) Innovations in Safety Management for Construction Sites: The Role of Deep Learning and Computer Vision Techniques. <italic>Construction Innovation</italic>, 26, 551-578. https://doi.org/10.1108/ci-04-2023-0062 <pub-id pub-id-type="doi">10.1108/ci-04-2023-0062</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1108/ci-04-2023-0062">https://doi.org/10.1108/ci-04-2023-0062</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Mohy, A.A.</string-name>
              <string-name>Bassioni, H.A.</string-name>
              <string-name>Elgendi, E.O.</string-name>
              <string-name>Hassan, T.M.</string-name>
            </person-group>
            <year>2026</year>
            <article-title>Innovations in Safety Management for Construction Sites: The Role of Deep Learning and Computer Vision Techniques</article-title>
            <source>Construction Innovation</source>
            <volume>26</volume>
            <pub-id pub-id-type="doi">10.1108/ci-04-2023-0062</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B3">
        <label>3.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Al-Nofaie, S.M., Sharaf, S. and Molla, R. (2025) Design Trends and Comparative Analysis of Lightweight Block Ciphers for IoTs. <italic>Applied Sciences</italic>, 15, Article 7740. https://doi.org/10.3390/app15147740 <pub-id pub-id-type="doi">10.3390/app15147740</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3390/app15147740">https://doi.org/10.3390/app15147740</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Al-Nofaie, S.M.</string-name>
              <string-name>Sharaf, S.</string-name>
              <string-name>Molla, R.</string-name>
            </person-group>
            <year>2025</year>
            <article-title>Design Trends and Comparative Analysis of Lightweight Block Ciphers for IoTs</article-title>
            <source>Applied Sciences</source>
            <volume>15</volume>
            <elocation-id>7740</elocation-id>
            <pub-id pub-id-type="doi">10.3390/app15147740</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B4">
        <label>4.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Liu, S., Bi, H., Liu, L., Yang, N. and Peng, T. (2025) Fine-Grained Graph Domain Adaptation via Instance Contrastive Learning. <italic>Expert Systems with Applications</italic>, 296, Article 129034. https://doi.org/10.1016/j.eswa.2025.129034 <pub-id pub-id-type="doi">10.1016/j.eswa.2025.129034</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.eswa.2025.129034">https://doi.org/10.1016/j.eswa.2025.129034</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Liu, S.</string-name>
              <string-name>Bi, H.</string-name>
              <string-name>Liu, L.</string-name>
              <string-name>Yang, N.</string-name>
              <string-name>Peng, T.</string-name>
            </person-group>
            <year>2025</year>
            <article-title>Fine-Grained Graph Domain Adaptation via Instance Contrastive Learning</article-title>
            <source>Expert Systems with Applications</source>
            <volume>296</volume>
            <elocation-id>129034</elocation-id>
            <pub-id pub-id-type="doi">10.1016/j.eswa.2025.129034</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B5">
        <label>5.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Liu, C., Zhang, J., Chen, K., Wang, M., Zou, Z. and Shi, Z. (2025) Remote Sensing Spatiotemporal Vision-Language Models: A Comprehensive Survey. <italic>IEEE</italic><italic>Geoscience</italic><italic>and</italic><italic>Remote</italic><italic>Sensing</italic><italic>Magazine</italic>, 14, 383-423. https://doi.org/10.1109/mgrs.2025.3598283 <pub-id pub-id-type="doi">10.1109/mgrs.2025.3598283</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1109/mgrs.2025.3598283">https://doi.org/10.1109/mgrs.2025.3598283</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Liu, C.</string-name>
              <string-name>Zhang, J.</string-name>
              <string-name>Chen, K.</string-name>
              <string-name>Wang, M.</string-name>
              <string-name>Zou, Z.</string-name>
              <string-name>Shi, Z.</string-name>
            </person-group>
            <year>2025</year>
            <article-title>Remote Sensing Spatiotemporal Vision-Language Models: A Comprehensive Survey</article-title>
            <source>IEEE Geoscience and Remote Sensing Magazine</source>
            <volume>14</volume>
            <pub-id pub-id-type="doi">10.1109/mgrs.2025.3598283</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B6">
        <label>6.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Consens, M.E., Default, C., Wainberg, M., <italic>et al</italic>. (2025) Transformers and Genome Language Models. <italic>Nature Machine Intelligence</italic>, 7, 346-362.</mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Consens, M.E.</string-name>
              <string-name>Default, C.</string-name>
              <string-name>Wainberg, M.</string-name>
            </person-group>
            <year>2025</year>
            <article-title>Transformers and Genome Language Models</article-title>
            <source>Nature Machine Intelligence</source>
            <volume>7</volume>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B7">
        <label>7.</label>
        <citation-alternatives>
          <mixed-citation publication-type="journal">Sabiraguha, A., Havyarimana, V., Niyongabo, P., Kamdjoug, J.R.K., Sindayigaya, I. and Niyonsaba, T. (2023) Digital in Higher Education in Burundi. <italic>Open Journal of Social Sciences</italic>, 11, 284-297. https://doi.org/10.4236/jss.2023.1111019 <pub-id pub-id-type="doi">10.4236/jss.2023.1111019</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.4236/jss.2023.1111019">https://doi.org/10.4236/jss.2023.1111019</ext-link></mixed-citation>
          <element-citation publication-type="journal">
            <person-group person-group-type="author">
              <string-name>Sabiraguha, A.</string-name>
              <string-name>Havyarimana, V.</string-name>
              <string-name>Niyongabo, P.</string-name>
              <string-name>Kamdjoug, J.R.K.</string-name>
              <string-name>Sindayigaya, I.</string-name>
              <string-name>Niyonsaba, T.</string-name>
            </person-group>
            <year>2023</year>
            <article-title>Digital in Higher Education in Burundi</article-title>
            <source>Open Journal of Social Sciences</source>
            <volume>11</volume>
            <pub-id pub-id-type="doi">10.4236/jss.2023.1111019</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
      <ref id="B8">
        <label>8.</label>
        <citation-alternatives>
          <mixed-citation publication-type="other">Sindayigaya, I. (2023) The Overview of Burundi in the Image of the African Charter on Rights and Welfare of the Child. <italic>Beijing Law Review</italic>, 14, 812-827. https://doi.org/10.4236/blr.2023.142044 <pub-id pub-id-type="doi">10.4236/blr.2023.142044</pub-id><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.4236/blr.2023.142044">https://doi.org/10.4236/blr.2023.142044</ext-link></mixed-citation>
          <element-citation publication-type="other">
            <person-group person-group-type="author">
              <string-name>Sindayigaya, I.</string-name>
            </person-group>
            <year>2023</year>
            <article-title>The Overview of Burundi in the Image of the African Charter on Rights and Welfare of the Child</article-title>
            <source>Beijing Law Review</source>
            <volume>14</volume>
            <pub-id pub-id-type="doi">10.4236/blr.2023.142044</pub-id>
          </element-citation>
        </citation-alternatives>
      </ref>
    </ref-list>
  </back>
</article>