<article article-type="article" specific-use="production" xml:lang="en" xmlns:hw="org.highwire.hpp" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:ref="http://schema.highwire.org/Reference" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:nlm="http://schema.highwire.org/NLM/Journal" xmlns:a="http://www.w3.org/2005/Atom" xmlns:c="http://schema.highwire.org/Compound" xmlns:hpp="http://schema.highwire.org/Publishing" xmlns:hwp="http://schema.highwire.org/Journal" xmlns:l="http://schema.highwire.org/Linking" xmlns:r="http://schema.highwire.org/Revision" xmlns:x="http://www.w3.org/1999/xhtml" xmlns:app="http://www.w3.org/2007/app"><front><journal-meta><journal-id journal-id-type="hwp">biorxiv</journal-id><journal-id journal-id-type="publisher-id">BIORXIV</journal-id><journal-title>bioRxiv</journal-title><abbrev-journal-title abbrev-type="publisher">bioRxiv</abbrev-journal-title><publisher><publisher-name>Cold Spring Harbor Laboratory</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.1101/2024.05.08.593025</article-id><article-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1</article-id><article-id pub-id-type="other" hwp:sub-type="pisa-master">biorxiv;2024.05.08.593025</article-id><article-id pub-id-type="other" hwp:sub-type="slug">2024.05.08.593025</article-id><article-id pub-id-type="other" hwp:sub-type="atom-slug">2024.05.08.593025</article-id><article-id pub-id-type="other" hwp:sub-type="tag">2024.05.08.593025</article-id><article-version>1.1</article-version><article-categories><subj-group subj-group-type="author-type"><subject>Regular Article</subject></subj-group><subj-group subj-group-type="heading"><subject>New Results</subject></subj-group><subj-group subj-group-type="hwp-journal-coll" hwp:journal-coll-id="Genetics" hwp:journal="biorxiv"><subject>Genetics</subject></subj-group></article-categories><title-group><article-title hwp:id="article-title-1">Investigating spatial dynamics in spatial omics data with StarTrail</article-title></title-group><author-notes hwp:id="author-notes-1"><corresp id="cor1" hwp:id="corresp-1" hwp:rev-id="xref-corresp-1-1 xref-corresp-1-2"><label>*</label>Corresponding author(s). E-mail(s): <email hwp:id="email-1">yunli@med.unc.edu</email>; <email hwp:id="email-2">didongli@unc.edu</email>;</corresp><fn fn-type="con" hwp:id="fn-1"><p hwp:id="p-1">Contributing authors: <email hwp:id="email-3">jiawenn@email.unc.edu</email>; <email hwp:id="email-4">caiweix@unc.edu</email>; <email hwp:id="email-5">quansun@live.unc.edu</email>; <email hwp:id="email-6">gwwang@ncsu.edu</email>; <email hwp:id="email-7">gaorav_gupta@med.unc.edu</email>; <email hwp:id="email-8">ah3758@drexel.edu</email>;</p></fn></author-notes><contrib-group hwp:id="contrib-group-1"><contrib contrib-type="author" hwp:id="contrib-1"><name name-style="western" hwp:sortable="Chen Jiawen"><surname>Chen</surname><given-names>Jiawen</given-names></name><xref ref-type="aff" rid="a1" hwp:id="xref-aff-1-1" hwp:rel-id="aff-1">1</xref></contrib><contrib contrib-type="author" hwp:id="contrib-2"><name name-style="western" hwp:sortable="Xiong Caiwei"><surname>Xiong</surname><given-names>Caiwei</given-names></name><xref ref-type="aff" rid="a1" hwp:id="xref-aff-1-2" hwp:rel-id="aff-1">1</xref></contrib><contrib contrib-type="author" hwp:id="contrib-3"><name name-style="western" hwp:sortable="Sun Quan"><surname>Sun</surname><given-names>Quan</given-names></name><xref ref-type="aff" rid="a1" hwp:id="xref-aff-1-3" hwp:rel-id="aff-1">1</xref></contrib><contrib contrib-type="author" hwp:id="contrib-4"><name name-style="western" hwp:sortable="Wang Geoffery W."><surname>Wang</surname><given-names>Geoffery W.</given-names></name><xref ref-type="aff" rid="a2" hwp:id="xref-aff-2-1" hwp:rel-id="aff-2">2</xref></contrib><contrib contrib-type="author" hwp:id="contrib-5"><name name-style="western" hwp:sortable="Gupta Gaorav P."><surname>Gupta</surname><given-names>Gaorav P.</given-names></name><xref ref-type="aff" rid="a3" hwp:id="xref-aff-3-1" hwp:rel-id="aff-3">3</xref><xref ref-type="aff" rid="a4" hwp:id="xref-aff-4-1" hwp:rel-id="aff-4">4</xref></contrib><contrib contrib-type="author" hwp:id="contrib-6"><name name-style="western" hwp:sortable="Halder Aritra"><surname>Halder</surname><given-names>Aritra</given-names></name><xref ref-type="aff" rid="a5" hwp:id="xref-aff-5-1" hwp:rel-id="aff-5">5</xref></contrib><contrib contrib-type="author" corresp="yes" hwp:id="contrib-7"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-9275-4189</contrib-id><name name-style="western" hwp:sortable="Li Yun"><surname>Li</surname><given-names>Yun</given-names></name><xref ref-type="aff" rid="a1" hwp:id="xref-aff-1-4" hwp:rel-id="aff-1">1</xref><xref ref-type="aff" rid="a6" hwp:id="xref-aff-6-1" hwp:rel-id="aff-6">6</xref><xref ref-type="aff" rid="a7" hwp:id="xref-aff-7-1" hwp:rel-id="aff-7">7</xref><xref ref-type="corresp" rid="cor1" hwp:id="xref-corresp-1-1" hwp:rel-id="corresp-1">*</xref><object-id pub-id-type="other" hwp:sub-type="orcid" xlink:href="http://orcid.org/0000-0002-9275-4189"/></contrib><contrib contrib-type="author" corresp="yes" hwp:id="contrib-8"><name name-style="western" hwp:sortable="Li Didong"><surname>Li</surname><given-names>Didong</given-names></name><xref ref-type="aff" rid="a1" hwp:id="xref-aff-1-5" hwp:rel-id="aff-1">1</xref><xref ref-type="corresp" rid="cor1" hwp:id="xref-corresp-1-2" hwp:rel-id="corresp-1">*</xref></contrib><aff id="a1" hwp:id="aff-1" hwp:rev-id="xref-aff-1-1 xref-aff-1-2 xref-aff-1-3 xref-aff-1-4 xref-aff-1-5"><label>1</label><institution hwp:id="institution-1">Department of Biostatistics, The University of North Carolina at Chapel Hill</institution>, Chapel Hill, 27599, North Carolina, <country>USA</country></aff><aff id="a2" hwp:id="aff-2" hwp:rev-id="xref-aff-2-1"><label>2</label><institution hwp:id="institution-2">Department of Statistics, North Carolina State University</institution>, Raleigh, 27695, North Carolina, <country>USA</country></aff><aff id="a3" hwp:id="aff-3" hwp:rev-id="xref-aff-3-1"><label>3</label><institution hwp:id="institution-3">Department of Radiation Oncology, The University of North Carolina at Chapel Hill</institution>, Chapel Hill, 27599, North Carolina, <country>USA</country></aff><aff id="a4" hwp:id="aff-4" hwp:rev-id="xref-aff-4-1"><label>4</label><institution hwp:id="institution-4">Lineberger Comprehensive Cancer Center, The University of North Carolina at Chapel Hill</institution>, Chapel Hill, 27599, North Carolina, <country>USA</country></aff><aff id="a5" hwp:id="aff-5" hwp:rev-id="xref-aff-5-1"><label>5</label><institution hwp:id="institution-5">Department of Epidemiology and Biostatistics, Drexel University</institution>, Philadelphia, 10587, Pennsylvania, <country>USA</country></aff><aff id="a6" hwp:id="aff-6" hwp:rev-id="xref-aff-6-1"><label>6</label><institution hwp:id="institution-6">Department of Genetics, The University of North Carolina at Chapel Hill</institution>, Chapel Hill, 27599, North Carolina, <country>USA</country></aff><aff id="a7" hwp:id="aff-7" hwp:rev-id="xref-aff-7-1"><label>7</label><institution hwp:id="institution-7">Department of Computer science, The University of North Carolina at Chapel Hill</institution>, Chapel Hill, 27599, North Carolina, <country>USA</country></aff></contrib-group><pub-date pub-type="epub-original" date-type="pub" publication-format="electronic" hwp:start="2024"><year>2024</year></pub-date><pub-date pub-type="hwp-created" hwp:start="2024-05-10T21:33:23-07:00">
    <day>10</day><month>5</month><year>2024</year>
  </pub-date><pub-date pub-type="hwp-received" hwp:start="2024-05-10T21:33:23-07:00">
    <day>10</day><month>5</month><year>2024</year>
  </pub-date><pub-date pub-type="epub" hwp:start="2024-05-10T21:42:52-07:00">
    <day>10</day><month>5</month><year>2024</year>
  </pub-date><pub-date pub-type="epub-version" hwp:start="2024-05-10T21:42:52-07:00">
    <day>10</day><month>5</month><year>2024</year>
  </pub-date><elocation-id>2024.05.08.593025</elocation-id><history hwp:id="history-1">
<date date-type="received" hwp:start="2024-05-08"><day>08</day><month>5</month><year>2024</year></date>
<date date-type="rev-recd" hwp:start="2024-05-08"><day>08</day><month>5</month><year>2024</year></date>
<date date-type="accepted" hwp:start="2024-05-10"><day>10</day><month>5</month><year>2024</year></date>
</history><permissions><copyright-statement hwp:id="copyright-statement-1">© 2024, Posted by Cold Spring Harbor Laboratory</copyright-statement><copyright-year>2024</copyright-year><license license-type="creative-commons" xlink:href="http://creativecommons.org/licenses/by-nc-nd/4.0/" hwp:id="license-1"><p hwp:id="p-2">This pre-print is available under a Creative Commons License (Attribution-NonCommercial-NoDerivs 4.0 International), CC BY-NC-ND 4.0, as described at <ext-link l:rel="related" l:ref-type="uri" l:ref="http://creativecommons.org/licenses/by-nc-nd/4.0/" ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by-nc-nd/4.0/" hwp:id="ext-link-1">http://creativecommons.org/licenses/by-nc-nd/4.0/</ext-link></p></license></permissions><self-uri xlink:href="593025.pdf" content-type="pdf" xlink:role="full-text"/><self-uri l:ref="forthcoming:yes" c:role="http://schema.highwire.org/variant/abstract" xlink:role="abstract" content-type="xhtml+xml" hwp:variant="yes"/><self-uri l:ref="forthcoming:yes" c:role="http://schema.highwire.org/variant/full-text" xlink:href="file:/content/biorxiv/vol0/issue2024/pdf/2024.05.08.593025v1.pdf" hwp:variant="yes" content-type="pdf" xlink:role="full-text"/><self-uri l:ref="forthcoming:yes" c:role="http://schema.highwire.org/variant/full-text" xlink:role="full-text" content-type="xhtml+xml" hwp:variant="yes"/><self-uri l:ref="forthcoming:yes" c:role="http://schema.highwire.org/variant/source" xlink:role="source" content-type="xml" xlink:show="none" hwp:variant="yes"/><self-uri l:ref="forthcoming:yes" c:role="http://schema.highwire.org/variant/original" xlink:role="original" content-type="xml" xlink:show="none" hwp:variant="yes" xlink:href="593025.xml"/><self-uri content-type="abstract" xlink:href="file:/content/biorxiv/vol0/issue2024/abstracts/2024.05.08.593025v1/2024.05.08.593025v1.htslp"/><self-uri content-type="fulltext" xlink:href="file:/content/biorxiv/vol0/issue2024/fulltext/2024.05.08.593025v1/2024.05.08.593025v1.htslp"/><abstract hwp:id="abstract-1"><title hwp:id="title-1">Abstract</title><p hwp:id="p-3">Spatial omics technologies revolutionize our view of biological processes within tissues. However, existing methods fail to capture localized, sharp changes characteristic of critical events (e.g. tumor development). Here, we present StarTrail, a novel gradient based method that powerfully defines rapidly changing regions and detects “cliff genes”, genes exhibiting drastic expression changes at highly localized or disjoint boundaries. StarTrail, the first to leverage spatial gradients for spatial omics data, also quantifies <italic toggle="yes">directional</italic> dynamics. Across multiple datasets, StarTrail accurately delineates boundaries (e.g., brain layers, tumor-immune boundaries), and detects cliff genes that may regulate molecular crosstalk at these biologically relevant boundaries but are missed by existing methods. For instance, StarTrail precisely pinpointed the cancer-immune interface in a HER2+ breast cancer dataset, unveiled key cliff genes including a potential prognostic biomarker <italic toggle="yes">IGSF3</italic>, highlighting NK-, B-cell mediated immunity, and B cell receptor signaling pathways missed by all spatial variable gene methods attempted. StarTrail, filling important gaps in current literature, enables deeper insights into tissue spatial architecture.</p></abstract><kwd-group kwd-group-type="author" hwp:id="kwd-group-1"><title hwp:id="title-2">Keywords</title><kwd hwp:id="kwd-1">Boundary detection</kwd><kwd hwp:id="kwd-2">cliff gene</kwd><kwd hwp:id="kwd-3">Gaussian process</kwd><kwd hwp:id="kwd-4">gradients</kwd></kwd-group><counts><page-count count="49"/></counts></article-meta><notes hwp:id="notes-1"><notes notes-type="competing-interest-statement" hwp:id="notes-2"><title hwp:id="title-3">Competing Interest Statement</title><p hwp:id="p-4">The authors have declared no competing interest.</p></notes></notes></front><body><sec id="s1" hwp:id="sec-1"><label>1</label><title hwp:id="title-4">Main</title><p hwp:id="p-5">Spatial data, with its inherent geographical, topological or geometric coordinates, enables structured views of cellular and inter-cellular dynamics and allows researchers to discern patterns, relationships, and changes over different spatial scales. Leveraging spatial information has proven invaluable in numerous fields including biomedical imaging studies, environmental science and computational biology [see, for e.g., 1–3]. In recent years, spatial omics technologies have emerged as valuable tools in biological and biomedical studies [<xref ref-type="bibr" rid="c4" hwp:id="xref-ref-4-1" hwp:rel-id="ref-4">4</xref>]. Their significance lie in the unique capability to provide omics measurements while simultaneously retaining spatial location information, e.g. gene expression in spatial transcriptomics (ST), and protein expression in spatial proteomics. Notable technologies in this domain include 10X Visium [<xref ref-type="bibr" rid="c5" hwp:id="xref-ref-5-1" hwp:rel-id="ref-5">5</xref>], Slide-seqV2 [<xref ref-type="bibr" rid="c6" hwp:id="xref-ref-6-1" hwp:rel-id="ref-6">6</xref>], and MERFISH [<xref ref-type="bibr" rid="c7" hwp:id="xref-ref-7-1" hwp:rel-id="ref-7">7</xref>] for ST, and CODEX [<xref ref-type="bibr" rid="c8" hwp:id="xref-ref-8-1" hwp:rel-id="ref-8">8</xref>] earmarked for spatial proteomics. The dual insight has transformed our understanding of cellular organization within tissues.</p><p hwp:id="p-6">The ST literature has seen several focused developments that have garnered attention and gained popularity. Methodology employed in the detection of spatially variable genes (SVGs) is one of the most popular [see, e.g., 9, 10] areas. It aims to detect genes that exhibit differential expression (DE) across spatial locations. SVG detection methods enable the identification of important genes that are associated with tissue architecture, function, or pathology, based on their spatial expression patterns. Note that SVGs and DE genes (DEGs) have been used exchangeably in ST literature. Here we use SVGs for genes exhibiting spatial changes without the need to pre-define spatial regions and use DEGs for genes showing differences between two pre-defined regions. Besides SVG and DEG detection, there is an increasing emphasis on developing methods to detect spatial clusters, also often referred to as spatial domains [see, e.g., 11–14], specifically using histology images and spatial coordinates. These spatial clustering methods aim to reliably represent the natural biological patterns found within tissues.</p><p hwp:id="p-7">However, at least three notable limitations are present in the aforementioned methodologies. First, although methods for identifying SVGs can reveal genes that vary across space, they are unable to pinpoint the exact locations of these variations. In the clustering context, the spatial domains defined can vary substantially depending on which genes are selected for analysis. Second, no existing method is sufficiently powerful to identify omics features (e.g., genes, proteins) that exhibit drastic changes along highly localized or disjoint boundaries. We name such genes cliff genes, distinct from SVGs, which typically manifest global changes or at least across larger areas. Third, existing methods provides no directionality information. In other words, no method quantifies how exactly the expression of any given gene changes in the 2-dimensional (2D) space. Owing to these methodological gaps, the underlying intricacies that drive the manifestation of localized boundaries in spatial omics data remain largely unexplored. Many current approaches rely on DE analyses at specific boundary intersections [<xref ref-type="bibr" rid="c15" hwp:id="xref-ref-15-1" hwp:rel-id="ref-15">15</xref>, <xref ref-type="bibr" rid="c16" hwp:id="xref-ref-16-1" hwp:rel-id="ref-16">16</xref>], but this barely touches on the depth of spatial dynamics involved.</p><p hwp:id="p-8">To address these limitations, we introduce StarTrail (Scalable spaTiAl gRadienT pRocess Approx-Imation utiLizing nearest-neighbor Gaussian process). StarTrail is the first method that leverages 2D spatial gradient for the analysis of spatial omics data. Spatial gradients, a concept borrowed from calculus, measuring rate of changes across spatial locations, offer perspectives complementary to raw omics measurements and thus into nuances of spatial data changes missed by existing approaches. Specifically in response to the three aforementioned limitations, StarTrail provides quantitative, directional, super-resolution gradient estimates that illuminate more precisely where and how changes occur. It is important to note that our use of the term ‘gradient’ specifically refers to its mathematical sense as the derivative of a function, describing the rate of changes in an omics feature relative to spatial coordinates. This definition is distinct from the more philosophical or intuitive use of ‘gradient’ found in some other studies [see, e.g., 17, 18]. Our focus is on the concrete and quantitative aspects of spatial data variation.</p><p hwp:id="p-9">As a motivating example, in cancer samples, the tumor region(s) can be rather small (e.g., newly developed or developing cancerous regions) and disjoint (multiple primary or a combination of primary and metastatic regions) with each having its own localized boundary. Existing methods proposed to identify SVGs are under-powered to identify omics features that manifest drastic changes along these boundaries because they can easily be an negligible portion of the entire tissue space examined. These boundaries, however, frequently coincide with vital biological junctures, such as the interface that distinguishes diseased from healthy tissues or areas of distinct biological functions [see, e.g., 17]. Complemented with histology images, gradient analyses are further enriched, e.g.,g enabling the analysis of gradient along pathologist-annotated boundaries. Previous research primarily concentrated on constructing slide topological maps using low-dimensional features, typically one-dimensional (1D) attributes such as pseudo-time or iso-depth [see, e.g., 17, 19–22]. These studies often leaned on stringent assumptions, such as assuming that every gene’s gradient direction is either null or aligned with the 1D latent variable [<xref ref-type="bibr" rid="c19" hwp:id="xref-ref-19-1" hwp:rel-id="ref-19">19</xref>], or the gradient calculation depends heavily on the choice of the model [<xref ref-type="bibr" rid="c20" hwp:id="xref-ref-20-1" hwp:rel-id="ref-20">20</xref>]. However, accurately inferring gene-wise spatial gradients [<xref ref-type="bibr" rid="c19" hwp:id="xref-ref-19-2" hwp:rel-id="ref-19">19</xref>] and appropriately applying gradients to understand spatial dynamics of each gene expression remain as open and challenging tasks.</p><p hwp:id="p-10">StarTrail is highly flexible and versatile: it can accommodate various spatial omics features, encompassing genes, proteins, as well as annotations or derived features such as estimated cell type proportions. StarTrail harnesses spatial gradients to project the vector field of each omics feature, pinpointing regions of the most rapid change by their heightened gradient values. This enables the delineation of omics boundaries through principal curve [<xref ref-type="bibr" rid="c23" hwp:id="xref-ref-23-1" hwp:rel-id="ref-23">23</xref>] (see <xref ref-type="sec" rid="s2a" hwp:id="xref-sec-3-1" hwp:rel-id="sec-3">section 2.1</xref> below). Moreover, when boundaries are pre-defined (e.g., based on pathologists’ annotations), StarTrail adeptly calculates gradients along their normal vectors (wombling, see <xref ref-type="sec" rid="s2a" hwp:id="xref-sec-3-2" hwp:rel-id="sec-3">section 2.1</xref> below), defining “cliff genes” whose gene expressions shift most sharply across the boundaries. Incorporating gradient metrics and Wombling analysis [<xref ref-type="bibr" rid="c24" hwp:id="xref-ref-24-1" hwp:rel-id="ref-24">24</xref>], Star-Trail also enhances downstream analysis potentials. For instance, StarTrail can categorize SVGs based on their gradients relative to predefined boundaries. Additionally, StarTrail can bolster clustering analysis by leveraging the gradient metric, providing more refined insight into spatial data structures, compared with only utilizing the original space. Notably, StarTrail achieves a marked increase in computational efficiency, compared to the original spatial gradient process [<xref ref-type="bibr" rid="c25" hwp:id="xref-ref-25-1" hwp:rel-id="ref-25">25</xref>, <xref ref-type="bibr" rid="c26" hwp:id="xref-ref-26-1" hwp:rel-id="ref-26">26</xref>], facilitating rapid and sophisticated spatial data analysis.</p><p hwp:id="p-11">We demonstrate the versatility of StarTrail on a spectrum of spatial omics data types and datasets, including omics data generated by technologies such as 10X Visium [<xref ref-type="bibr" rid="c5" hwp:id="xref-ref-5-2" hwp:rel-id="ref-5">5</xref>], Spatial Transcriptomics [<xref ref-type="bibr" rid="c5" hwp:id="xref-ref-5-3" hwp:rel-id="ref-5">5</xref>], and CODEX [<xref ref-type="bibr" rid="c8" hwp:id="xref-ref-8-2" hwp:rel-id="ref-8">8</xref>], as well as derived features such as estimated cell type proportions.</p></sec><sec id="s2" hwp:id="sec-2"><label>2</label><title hwp:id="title-5">Results</title><sec id="s2a" hwp:id="sec-3" hwp:rev-id="xref-sec-3-1 xref-sec-3-2"><label>2.1</label><title hwp:id="title-6">Scalable spatial gradient process via StarTrail</title><p hwp:id="p-12">The overview of StarTrail is illustrated in <xref rid="fig1" ref-type="fig" hwp:id="xref-fig-1-1" hwp:rel-id="F1">Fig. 1</xref>. To understand the relationships between spatial co-ordinates <italic toggle="yes">s</italic> and omics feature <italic toggle="yes">Y</italic>, StarTrail harnesses the Gaussian Process (GP), <italic toggle="yes">Z</italic>. Designed to be versatile, StarTrail operates seamlessly with both 2D and 3D coordinates. The input data, <italic toggle="yes">Y</italic> can en-compass various spatial omics or spatial annotations, including metrics such as gene expression, protein abundance, or cell type proportions. In this illustration, we use gene expression as an example. The sub-sequent gradient is adeptly modeled using a spatial gradient process denoted by ∇<italic toggle="yes">Z</italic>. This forms a joint distribution of [<italic toggle="yes">Z</italic>, ∇<italic toggle="yes">Z</italic>], with cross-covariance characterized by <italic toggle="yes">K</italic> and its derivatives up to the second order [see, for e.g., 25–27]. While GPs offer the advantageous property wherein their derivative remains a GP, the large sample size (e.g. the number of spots/cells in ST) in spatial omics data makes the use of the original spatial gradient process untenable [<xref ref-type="bibr" rid="c25" hwp:id="xref-ref-25-2" hwp:rel-id="ref-25">25</xref>, <xref ref-type="bibr" rid="c26" hwp:id="xref-ref-26-2" hwp:rel-id="ref-26">26</xref>]. For instance, it takes over 40 hours to fit the original spatial gradient process on ∼ 3000 spots for one gene. To address this, StarTrail introduces a scalable spatial gradient process which utilizes finite differences to approximate gradients. Inspired by both plug-in GP [<xref ref-type="bibr" rid="c28" hwp:id="xref-ref-28-1" hwp:rel-id="ref-28">28</xref>] and the nearest-neighbor (NN) GP [<xref ref-type="bibr" rid="c29" hwp:id="xref-ref-29-1" hwp:rel-id="ref-29">29</xref>, <xref ref-type="bibr" rid="c30" hwp:id="xref-ref-30-1" hwp:rel-id="ref-30">30</xref>], StarTrail establishes a NNGP model. The integration of NN facilitates a significant computational economy, reducing the cost of fitting the GP from O(<italic toggle="yes">n</italic><sup>3</sup>) to O(<italic toggle="yes">nm</italic><sup>3</sup>), where <italic toggle="yes">n</italic> represents the total sample size and <italic toggle="yes">m</italic> is the number of neighboring locations used [<xref ref-type="bibr" rid="c29" hwp:id="xref-ref-29-2" hwp:rel-id="ref-29">29</xref>, <xref ref-type="bibr" rid="c30" hwp:id="xref-ref-30-2" hwp:rel-id="ref-30">30</xref>].</p><fig id="fig1" position="float" fig-type="figure" orientation="portrait" hwp:id="F1" hwp:rev-id="xref-fig-1-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIG1</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F1</object-id><object-id pub-id-type="publisher-id">fig1</object-id><label>Figure 1.</label><caption hwp:id="caption-1"><p hwp:id="p-13">StarTrail overview. StarTrail takes spatial omics data as its input (top left). Utilizing a scalable spatial gradient process, StarTrail models the intricate relationship between spatial coordinates <italic toggle="yes">s</italic> and omics features <italic toggle="yes">Y</italic>, efficiently inferring gradients across a tissue region (top right). With the inferred gradients, StarTrail can identify boundaries by pinpointing areas exhibiting the highest gradient values (bottom left). Additionally, StarTrail employs Wombling analysis across predefined boundaries to detect “cliff genes”—genes with significant expression changes at these boundaries (bottom middle). Moreover, these gradients could be leveraged to enhance downstream analyses, such as clustering (bottom right).</p></caption><graphic xlink:href="593025v1_fig1" position="float" orientation="portrait" hwp:id="graphic-1"/></fig><p hwp:id="p-14">Leveraging the advantages of GPs, StarTrail offers the ability to infer spatial omics data in previously unmeasured regions, thereby enhancing the resolution to any desired level. Utilizing the inferred gradient, ∇<italic toggle="yes">Z</italic>, StarTrail enables a detailed examination of the spatial dynamics. More precisely, it enables the evaluation of gradients in any chosen direction. For preliminary analysis, for instance, the process initiates by deducing the gradient along the directions represented by <italic toggle="yes">e</italic><sub>1</sub> := (1, 0) (horizontal direction) and <italic toggle="yes">e</italic><sub>2</sub> := (0, 1) (vertical direction), respectively. The gradient at every point is then calculated as <inline-formula hwp:id="inline-formula-1"><inline-graphic xlink:href="593025v1_inline1.gif" hwp:id="inline-graphic-1"/></inline-formula>, from which we define the gradient flow map which characterizes the direction of local change for each spot/cell. Following this, the <italic toggle="yes">L</italic><sup>2</sup>-norm of gradients <inline-formula hwp:id="inline-formula-2"><inline-graphic xlink:href="593025v1_inline2.gif" hwp:id="inline-graphic-2"/></inline-formula> serves to determine the magnitude of the gradient. This nuanced approach enables us to categorize the dynamics of omics data—rapid changes are indicated by high gradient values, while areas exhibiting slow changes are characterized by low-magnitude gradients. For omics features exhibiting swift alterations across a region, boundaries can be identified along the area of maximal gradient, by using a non-parametric curve fitting algorithm to delineate principal curves [for more details, see 23].</p><p hwp:id="p-15">In addition to the ability to infer gradients and delineate boundaries, we recognize that it is equally crucial to evaluate the directional gradients along a given boundary. The boundary is usually annotated by pathologists or experts with domain knowledge. Delving into omics alterations across specific boundaries provides invaluable biomedical insights, particularly in scenarios where these boundaries demarcate pivotal biological structures or lurking conditions. To leverage pathologist-annotated boundaries or boundaries informed by external information, such as cell type proportion demarcations, we engage the “Wombling” technique. This approach involves statistical inference on directional derivatives evaluated along the normal direction (also known as the orthogonal direction) of pre-designated boundary segments. Subsequent integration across the entire curve produces a valid measure that can be used to track whether a curve forms a “wombling boundary” [see 24, Section 3.3, for more detail]. Upon analyzing the curve gradient inferred using wombling, we introduce the concept of a “cliff gene”. A cliff gene designates a gene that exhibits pronounced shifts across specific boundaries. Analogously, we can expand this concept to other domains and similarly define a “cliff cell type” or a “cliff protein”, each signifying an omics feature that manifests significant variations over demarcated regions.</p><p hwp:id="p-16">Utilizing the inferred gradients provides perspectives on spatial dynamics beyond the original feature space, enhancing and enriching downstream analyses. Using StarTrail inferred spatial metrics, we can precisely link SVGs to specific spatial boundaries based on their gradients. This granularity enables a more targeted and precise analysis. Furthermore, the integration of gradient metrics enhances clustering analysis, delineating distinct clusters based on spatial intricacies.</p></sec><sec id="s2b" hwp:id="sec-4"><label>2.2</label><title hwp:id="title-7">StarTrail accurately estimates spatial gradient and defines rapid change area</title><p hwp:id="p-17">We first benchmark the performance of StarTrail using a series of simulation tests designed to mirror a variety of spatial patterns. They range from trigonometric functions (<xref rid="figED1" ref-type="fig" hwp:id="xref-fig-5-1" hwp:rel-id="F5">Extended Data Fig. 1</xref>, first row), to common patterns including linear gradients and hot-spots, as well as regional patterns (<xref rid="figED1" ref-type="fig" hwp:id="xref-fig-5-2" hwp:rel-id="F5">Extended Data Fig. 1</xref>, second row-last row). For each simulated scenario, we calculated gradients along the directions <italic toggle="yes">e</italic><sub>1</sub> = (1, 0) and <italic toggle="yes">e</italic><sub>2</sub> = (0, 1). The results show that StarTrail achieves high accuracy in recovering the gradients along these axes. To pinpoint areas of rapid changes in any given direction, we computed the <italic toggle="yes">L</italic><sup>2</sup>-norm of the gradients (referred to <italic toggle="yes">L</italic><sup>2</sup> in the following context). This metric successfully identified regions with the highest gradient magnitude, corresponding to the points of most significant changes in our simulation models. Linear gradients and hot-spots represent patterns of changes gradually from one constant to another constant, while regional patterns exhibit abrupt jumps from one constant to another. Consequently, we have chosen to use <italic toggle="yes">L</italic><sup>2</sup> as the indicator to mark areas undergoing rapid changes when applying StarTrail to real-world data analysis.</p><p hwp:id="p-18">After confirming StarTrail’s efficacy through simulations, we turned our attention to real-world spatial omics datasets, starting with ST data. We first analyzed 10x Visium data from human dorsolateral prefrontal cortex (DLPFC)[<xref ref-type="bibr" rid="c16" hwp:id="xref-ref-16-2" hwp:rel-id="ref-16">16</xref>]. The DLPFC dataset is known for its well-defined seven-layer architecture, including layers L1 through L6 and white matter (WM), making it an exemplary instance for studying layer-enriched gene expression. The spatially resolved expression has important clinical implications for neurological disorders such as autism spectrum disorder and schizophrenia [<xref ref-type="bibr" rid="c16" hwp:id="xref-ref-16-3" hwp:rel-id="ref-16">16</xref>]. StarTrail aptly captures the intricate relationship between spatial coordinates and gene expression levels, as well as their spatial gradients. This capability is demonstrated through the precise modeling of several key layer marker genes (<xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-1" hwp:rel-id="F2">Fig. 2b-d</xref>, other example genes in <xref rid="figED2" ref-type="fig" hwp:id="xref-fig-6-1" hwp:rel-id="F6">Extended Data Fig. 2</xref>). <italic toggle="yes">MBP</italic>, a marker gene enriched in WM, shows spatial gradients with the highest <italic toggle="yes">L</italic><sup>2</sup> values concentrated at the boundary between WM and layer L6. StarTrail further identifies the boundary using the principal curve fitted to the points with <italic toggle="yes">L</italic><sup>2</sup> exceeding its 90% quantile. The resulting boundary (<xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-2" hwp:rel-id="F2">Fig. 2b</xref>) accurately defines the zone of the most dramatic expression changes. Additionally, the direction of these gradients is effectively summarized by aggregating the horizontal and vertical gradients ([<xref ref-type="bibr" rid="c31" hwp:id="xref-ref-31-1" hwp:rel-id="ref-31">31</xref>], <xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-3" hwp:rel-id="F2">Fig. 2c</xref>). The gradient flow map generated from this approach provides an insightful visualization of gene expression dynamics, visualizing the change direction of gene expression, and pinpointing locations of extreme expression (singularities), where neighboring gradients converge (indicating maximal expression) or diverge (indicating minimal expression). StarTrail also accurately estimated spatial gradients for other layer-specific markers such as <italic toggle="yes">PCP4</italic> for layer L5 and <italic toggle="yes">HPCAL1</italic> for layer L1, further demonstrating the robustness and accuracy of StarTrail in deciphering spatial expression patterns. Moreover, one fundamental advantage of employing a spatial gradient process lies in its ability to construct a spatial model between spatial locations and their corresponding omics feature as well as gradients. Consequently, StarTrail is equipped to extrapolate spatial omics features to areas that were not originally sampled, effectively filling in the gaps and providing a more complete super-resolutional picture of the spatial dynamics (<xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-4" hwp:rel-id="F2">Fig. 2b</xref>).</p><fig id="fig2" position="float" fig-type="figure" orientation="portrait" hwp:id="F2" hwp:rev-id="xref-fig-2-1 xref-fig-2-2 xref-fig-2-3 xref-fig-2-4 xref-fig-2-5 xref-fig-2-6 xref-fig-2-7 xref-fig-2-8 xref-fig-2-9 xref-fig-2-10"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIG2</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F2</object-id><object-id pub-id-type="publisher-id">fig2</object-id><label>Figure 2.</label><caption hwp:id="caption-2"><p hwp:id="p-19">Gradient analysis. <bold>DLPFC</bold>: (a) Histology image of DLPFC slide 151676. (b) A workflow outlines the process of detecting gene expression boundaries, using <italic toggle="yes">MBP</italic> as an example. StarTrail first smoothes raw gene expression and estimates the spatial gradients, and then fits a principal curve to spots where the <italic toggle="yes">L</italic><sup>2</sup> norm of the gradients exceeds the 0.9 quantile, effectively delineating the gene expression boundary. (c) Gradient flow map for <italic toggle="yes">MBP</italic>, arrows colored by smoothed and normalized expression (d) Inferred boundaries for <italic toggle="yes">PCP4</italic> and <italic toggle="yes">HPCAL1</italic>, points colored by smoothed and normalized expression (e) Pathologist annotation. <bold>HER2+ breast cancer</bold>: (f) left: pathologist annotation, middle: estimated cancer epithelial cell proportion, right: estimated <italic toggle="yes">L</italic><sup>2</sup>. <bold>CODEX</bold>: (g) left: cell type assignment, right: detected boundary for CD15, points colored by smoothed and normalized abundance.</p></caption><graphic xlink:href="593025v1_fig2" position="float" orientation="portrait" hwp:id="graphic-2"/></fig><p hwp:id="p-20">Moving to a finer-resolution and non-layered dataset, we next applied StarTrail on HER2+ breast cancer data from the Spatial Transcriptomics platform (focusing on slide H1 shown in <xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-5" hwp:rel-id="F2">Fig. 2f</xref>). Breast cancer encompasses a variety of subtypes, among which HER2-positive (HER2+) subtype is particularly known for the overexpression of <italic toggle="yes">ERBB2</italic>, commonly referred to as <italic toggle="yes">HER2</italic> [<xref ref-type="bibr" rid="c32" hwp:id="xref-ref-32-1" hwp:rel-id="ref-32">32</xref>]. The dataset was annotated by pathologist into six distinct histo-pathological regions: noninvasive ductal carcinoma in-situ (DCIS), invasive breast cancer (IBC), adipose tissue, immune infiltrate, breast glands, or connective tissue (<xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-6" hwp:rel-id="F2">Fig. 2f</xref> left panel). As mentioned, a key advantage of StarTrail is its adaptability to various spatial measurements. For this dataset, we applied StarTrail to model the spatial distribution of cell type proportions inferred by RCTD (<xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-7" hwp:rel-id="F2">Fig. 2f</xref> middle panel) [<xref ref-type="bibr" rid="c33" hwp:id="xref-ref-33-1" hwp:rel-id="ref-33">33</xref>]. Cancer epithelial cells, as expected, are predominantly located within cancerous regions [<xref ref-type="bibr" rid="c32" hwp:id="xref-ref-32-2" hwp:rel-id="ref-32">32</xref>, <xref ref-type="bibr" rid="c34" hwp:id="xref-ref-34-1" hwp:rel-id="ref-34">34</xref>, <xref ref-type="bibr" rid="c35" hwp:id="xref-ref-35-1" hwp:rel-id="ref-35">35</xref>]. StarTrail’s inference of spatial gradients showed accurate delineation, particularly with high <italic toggle="yes">L</italic><sup>2</sup> values around the DCIS zones (<xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-8" hwp:rel-id="F2">Fig. 2f</xref> right panel). Notably, the steepest gradients were identified at the interface between the DCIS and the immune infiltrate areas. This observation suggests that the most pronounced changes in cell type composition occurs at the boundary of these two regions, potentially offering insights into the tumor microenvironment and the interface of cancerous and immune responses. As evidenced in <xref rid="figED3" ref-type="fig" hwp:id="xref-fig-7-1" hwp:rel-id="F7">Extended Data Fig. 3</xref>, StarTrail also works well on other cell types.</p><fig id="fig3" position="float" fig-type="figure" orientation="portrait" hwp:id="F3" hwp:rev-id="xref-fig-3-1 xref-fig-3-2 xref-fig-3-3 xref-fig-3-4 xref-fig-3-5 xref-fig-3-6 xref-fig-3-7 xref-fig-3-8 xref-fig-3-9 xref-fig-3-10"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIG3</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F3</object-id><object-id pub-id-type="publisher-id">fig3</object-id><label>Figure 3.</label><caption hwp:id="caption-3"><p hwp:id="p-21">Wombling analysis. <bold>DLPFC</bold>: (a) Annotation boundaries. The arrows indicate the normal direction for a single boundary: between white matter (WM) and layer 6 (L6). (b) Selected genes with top-tier gradients over single boundaries. (c) An example for the normal direction when combining two boundaries: one between WM and L6 and the other between layer 5 (L5) and L6. (d) Selected genes with top-tier weighted-sum gradients over multi-boundaries. <bold>HER2+ breast cancer</bold>: (e) Cancer epithelial cell boundaries. The arrows indicate the normal direction across the boundaries. (f) Top 2 genes with the highest gradients over the two cancer epithelial cell boundaries combined. (g) Top 2 genes with the lowest gradients over the multi-boundaries. (h) Top gene with the highest gradient over the multi-boundaries only detected by StarTrail. (i) Upset plot for detected cliff genes and SVGs. (j) Top 10 pathways identified by cliff genes with the largest gene ratio. All the gene expression in the figure are smoothed and normalized. epi: epithelial.</p></caption><graphic xlink:href="593025v1_fig3" position="float" orientation="portrait" hwp:id="graphic-3"/></fig><p hwp:id="p-22">StarTrail was further tested on spatial proteomic data from human tonsils measured by the CODEX platform [<xref ref-type="bibr" rid="c36" hwp:id="xref-ref-36-1" hwp:rel-id="ref-36">36</xref>] (<xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-9" hwp:rel-id="F2">Fig. 2g</xref>, <xref rid="figED4" ref-type="fig" hwp:id="xref-fig-8-1" hwp:rel-id="F8">Extended Data Fig. 4</xref>). The spatial pattern of the tonsil tissue is characterized by the presence of germinal centers (GC), known for heightened abundance of CD15 and a predominance of B cells. StarTrail successfully delineated spatial gradients within this complex tissue structure. The highest <italic toggle="yes">L</italic><sup>2</sup> values are indicative of the steepest expression changes, occurring at the GC boundary. Notably, this occurs at the transition zones between two types of B cells.</p><fig id="fig4" position="float" fig-type="figure" orientation="portrait" hwp:id="F4" hwp:rev-id="xref-fig-4-1 xref-fig-4-2 xref-fig-4-3"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIG4</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F4</object-id><object-id pub-id-type="publisher-id">fig4</object-id><label>Figure 4.</label><caption hwp:id="caption-4"><p hwp:id="p-23">Enhanced clustering analysis. (a) Integrating <italic toggle="yes">L</italic><sup>2</sup> to gene expression enlarges the difference of mean gene expression between its enriched layer and other layers. (b) Integrating <italic toggle="yes">L</italic><sup>2</sup> to clustering analysis enhanced clustering, shown by higher ARI. (c) The clustering result from five methods. Here, ‘Without gradient’ (the first row) represents the best clustering result (highest ARI) using only gene expression in SVG and cliff gene sets. ‘With gradient’ (the second row) represents the best clustering result using either <italic toggle="yes">L</italic><sup>2</sup> alone, or Gene+<italic toggle="yes">L</italic><sup>2</sup> in SVG and cliff gene sets. We matched the color of predicted clusters to truth for visualization. See <xref rid="figS11" ref-type="fig" hwp:id="xref-fig-25-1" hwp:rel-id="F25">Supplementary Fig. 11</xref> for raw result.</p></caption><graphic xlink:href="593025v1_fig4" position="float" orientation="portrait" hwp:id="graphic-4"/></fig><p hwp:id="p-24">Our results demonstrate that, across various datasets, StarTrail can precisely estimate spatial gradients and delineate areas of rapid changes. StarTrail’s ability to handle multiple data types makes it a unifying tool for integrated omics analysis. StarTrail thus boasts great potential to facilitate the integrative analysis of multiple assays, such as dual gene expression studies or joint analyses of gene expression and proteomics with either computationally inferred or pathological annotations. For example, we could calculate the gradients of one gene and compare that to the gradients of the cell type proportion on the same slide. This adaptability is essential to the current landscape of biological research, where the integration of different omics is increasingly crucial for a deeper understanding of complex biological patterns and dynamics.</p></sec><sec id="s2c" hwp:id="sec-5"><label>2.3</label><title hwp:id="title-8">Spatial Wombling enables the identification of cliff genes</title><p hwp:id="p-25">One of the primary aims in spatial omics research is to illuminate variations between different tissue regions. Based on histology images, pathologists often mark biologically relevant boundaries, which is a labor-intensive process. In the absence of such annotations, computational methods, such as image segmentation or cell type borders identification, can infer such information. These boundaries between tissue regions represent not merely physical divisions, but also transition zones where critical biological changes occur, such as the shift from healthy to diseased tissue. Investigating these boundaries is key to understanding the underlying causes and mechanisms behind the formation of distinct regions, especially in the context of health and disease research, where understanding the factors that influence disease development and progression is critical.</p><p hwp:id="p-26">StarTrail leverages these boundaries to identify “cliff genes”, which exhibit sharp expression shifts across boundaries. This cliff-gene approach transcends traditional DE analysis between the two domains separated by the boundaries, by revealing localized gene expression changes that might be missed in DE analysis, where signals are diluted over larger areas. In addition, StarTrail’s capability extends beyond a single boundary. It can adeptly handle multiple boundaries for which traditional DE analyses often struggle due to their intrinsic limitations to two-region comparisons. StarTrail’s flexibility in accommo-dating different boundary shapes, be they open or closed curves, circumvents the need to assign spots or cells to specific regions, a requirement in DE analysis [<xref ref-type="bibr" rid="c16" hwp:id="xref-ref-16-4" hwp:rel-id="ref-16">16</xref>]. This allows for a more dynamic, flexible, and finer-resolution DE and SVG analysis. Focusing on boundary-induced expression changes is essential for understanding the spatial structure of tissues and changes across regions.</p><sec id="s2c1" hwp:id="sec-6"><title hwp:id="title-9">DLPFC</title><p hwp:id="p-27">Delving further into the DLPFC analysis [<xref ref-type="bibr" rid="c16" hwp:id="xref-ref-16-5" hwp:rel-id="ref-16">16</xref>], we continued to explore StarTrail’s capabilities to decipher gene expression gradients perpendicular to pathologically annotated boundaries (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-1" hwp:rel-id="F3">Fig. 3a</xref>). By aggregating these gradients along a single boundary, we were able to identify genes that demonstrate substantial shifts in expression across different regions (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-2" hwp:rel-id="F3">Fig. 3b</xref>, <xref rid="figED5" ref-type="fig" hwp:id="xref-fig-9-1" hwp:rel-id="F9">Extended Data Fig. 5</xref>,<xref rid="figED6" ref-type="fig" hwp:id="xref-fig-10-1" hwp:rel-id="F10">6</xref>, Supplementary Table 1). For example, StarTrail successfully identified <italic toggle="yes">PLP1</italic>, previously noted for pronounced gradients at the WM boundary (boundary 6), as a cliff gene with the highest gradient (<italic toggle="yes">L</italic><sup>2</sup> = 81.21) along boundary 6 across all genes. Other genes identified by StarTrail with top-rank gradients along a single boundary also conform to anticipated spatial expression profiles. To validate the accuracy of the estimated gradients, we extended both outward and inward from the pathologist annotated boundaries [<xref ref-type="bibr" rid="c37" hwp:id="xref-ref-37-1" hwp:rel-id="ref-37">37</xref>], with each step encompassing one delineated layer of spots (<xref rid="figS11" ref-type="fig" hwp:id="xref-fig-25-2" hwp:rel-id="F25">Supplementary Fig. 1</xref>,<xref rid="figS2" ref-type="fig" hwp:id="xref-fig-16-1" hwp:rel-id="F16">2</xref>). For every delineated layer of spots, we computed the average gene expression counts. The concordance between the estimated boundary gradients and the slopes of gene expression change at the boundaries was striking (Pearson correlation between estimated gradient and slope in cliff genes: boundary0: 0.97, boundary1: 0.94, boundary2: 0.99, boundary3: 0.94, boundary4: 0.80, boundary5: 0.95, <xref rid="figED7" ref-type="fig" hwp:id="xref-fig-11-1" hwp:rel-id="F11">Extended Data Fig. 7</xref>, <xref rid="figS2" ref-type="fig" hwp:id="xref-fig-16-2" hwp:rel-id="F16">Supplementary Fig. 2</xref>). Genes with a steeper slope across the boundary consistently exhibited higher estimated boundary gradients. This agreement attests to the utility and robustness of our gradient estimation method, confirming that the boundary gradients inferred by StarTrail accurately reflect the actual changes in gene expression as one moves across the pathologically defined boundaries. Gradient estimation along boundaries goes beyond naive exploratory summaries of the change in spatial gene expression. Our boundary-informed supervised analysis thereby facilitates the discovery of “cliff genes” which are intricately linked to biological boundaries.</p><p hwp:id="p-28">The biological intricacy of the DLPFC necessitates an analysis that is not confined to single boundary definitions. For example, while <italic toggle="yes">MBP</italic> expression shows a consistent increase from the top right to the bottom left, as indicated by multiple positive boundary gradients, contrasting genes like <italic toggle="yes">PCP4</italic> display heterogeneity in their expression patterns. This is marked by their positive and negative gradients across different single boundaries (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-3" hwp:rel-id="F3">Fig. 3b</xref>). This variability accentuates the need to accommodate multiple boundaries into spatial omics analyses. When engaging with multiple boundaries, StarTrail aggregates information by taking the weighted sum of gradients along multiple boundaries. For instance, StarTrail enables the delineation of WM-enriched genes through analyzing gradients over boundary 6 (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-4" hwp:rel-id="F3">Fig. 3a</xref>). Similarly, StarTrail helps to identify genes with high expression in L6 by analyzing the weighted sum of gradients along boundaries 5 and 6 (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-5" hwp:rel-id="F3">Fig. 3c</xref>). By analyzing multiple boundary pairs (for example, boundaries 5,6 for layer L6, boundaries 4,5 for layer L5, boundaries 3,4 for layer L4, boundaries 2,3 for layer L3, boundaries 1,2 for layer L2), StarTrail accurately identifies genes enriched in specific layers or regions (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-6" hwp:rel-id="F3">Fig. 3d</xref>, <xref rid="figED8" ref-type="fig" hwp:id="xref-fig-12-1" hwp:rel-id="F12">Extended Data Fig. 8</xref>).</p></sec><sec id="s2c2" hwp:id="sec-7"><title hwp:id="title-10">Breast cancer</title><p hwp:id="p-29">In addition, StarTrail’s ability to harness estimated gradients for detecting boundaries in various contexts, such as gene expression or cell proportions, extends its utility to the simultaneous modeling of multi-omics data. This capacity is particularly advantageous with co-assay data, for example gene expression and cell type annotation, where cliff genes can be identified in relation to cell-type-defined boundaries. In our study of HER2+ breast cancer, we applied StarTrail, focusing squarely on the previously identified cancer epithelial cell boundary, to estimate boundary gradients for gene expression (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-7" hwp:rel-id="F3">Fig. 3e-h</xref>, <xref rid="figED9" ref-type="fig" hwp:id="xref-fig-13-1" hwp:rel-id="F13">Extended Data Fig. 9</xref>, Supplementary Table 2). <italic toggle="yes">ERBB2</italic>, known as a marker gene of cancer epithelial cells, was pinpointed by StarTrail with the highest boundary gradient, demon-strating its efficacy in identifying genes critical to cancer pathology. Further examination of genes with pronounced boundary gradients revealed their biological significance in cancer progression. For instance, <italic toggle="yes">MGP</italic>, whose gene expression was found to be negatively associated with patients’ overall survival, is a promising prognostic biomarker [<xref ref-type="bibr" rid="c38" hwp:id="xref-ref-38-1" hwp:rel-id="ref-38">38</xref>]. Similarly, increased <italic toggle="yes">CD24</italic> expression in breast cancer tissues compared to normal tissues, as observed in TCGA, underscores its potential as a potential biomarker for prognosis [<xref ref-type="bibr" rid="c39" hwp:id="xref-ref-39-1" hwp:rel-id="ref-39">39</xref>]. When contrasting the top genes identified by cell type-specific boundaries with those discerned by pathologist-annotated cancer boundaries, we observed remarkable concordance (<xref rid="figED9" ref-type="fig" hwp:id="xref-fig-13-2" hwp:rel-id="F13">Extended Data Fig. 9</xref>). This suggests that, in the absence of pathological annotations, cell type boundaries can serve as an effective proxy for delineating the dynamic interface between cancerous and non-cancerous regions. These results demonstrate StarTrail’s precision in delineating spatial expression boundaries and corroborate the robustness of StarTrail’s gradient estimation techniques. StarTrail’s ability to accurately identify genes with drastic expression changes at cellular boundaries enables researchers to focus on regions of rapid changes, which often harbor critical biological insights. In the context of disease pathology, such as cancer, discovery of these rapidly changing zones can provide a wealth of information regarding the mechanisms of disease progression, invasion, and interaction with the surrounding tissue microenvironment. By targeting these rapidly transitional areas, researchers can gain a more comprehensive understanding of omics dynamics and cellular heterogeneity at the tumor margin, which is essential for the development of targeted therapies and better predictive prognostic markers.</p></sec></sec><sec id="s2d" hwp:id="sec-8"><label>2.4</label><title hwp:id="title-11">StarTrail-identified cliff genes and pathways offer novel insights</title><p hwp:id="p-30">We benchmark StarTrail’s performance in identifying cliff genes against established SVG detection methods in the human DLPFC and HER2+ breast cancer dataset. We chose four widely-used SVG methods: Spark-X [<xref ref-type="bibr" rid="c40" hwp:id="xref-ref-40-1" hwp:rel-id="ref-40">40</xref>], SoMDE [<xref ref-type="bibr" rid="c41" hwp:id="xref-ref-41-1" hwp:rel-id="ref-41">41</xref>], nnSVG [<xref ref-type="bibr" rid="c42" hwp:id="xref-ref-42-1" hwp:rel-id="ref-42">42</xref>], and SpatialDE [<xref ref-type="bibr" rid="c43" hwp:id="xref-ref-43-1" hwp:rel-id="ref-43">43</xref>].</p><sec id="s2d1" hwp:id="sec-9"><title hwp:id="title-12">DLPFC</title><p hwp:id="p-31">Analyzing the DLPFC 151576 dataset, among 11,686 genes, Spark-X, SpatialDE, nnSVG, and SoMDE identified 9,673, 5,266, 1,622 and 1,185 SVGs respectively; and StarTrail identified 758 cliff genes (<xref rid="figS3" ref-type="fig" hwp:id="xref-fig-17-1" hwp:rel-id="F17">Supplementary Fig. 3</xref>, Supplementary Table 1). Notably, in this dataset, most cliff genes detected by StarTrail were identified by some SVG method(s), with only 7 StarTrail-exclusive genes missed by all four SVG methods (see Methods for threshold recommendations). Investigation into these 7 unique genes revealed a commonality: each exhibited rather localized patterns of hot spots along at least one pre-defined boundary (<xref rid="figS4" ref-type="fig" hwp:id="xref-fig-18-1" hwp:rel-id="F18">Supplementary Fig. 4</xref>). To substantiate the validity of StarTrail findings beyond visual assessments, we checked cliff genes that StarTrail identified from one single slide against genes identified by combining information across 8 slides from the spatialLIBD study [<xref ref-type="bibr" rid="c16" hwp:id="xref-ref-16-6" hwp:rel-id="ref-16">16</xref>]. Reassuringly, we observed a substantial overlap: for instance <italic toggle="yes">CCDC80, MEX3D</italic>, and <italic toggle="yes">RRP8</italic> showed significance at the L6-WM boundary, while <italic toggle="yes">MSX1</italic> and <italic toggle="yes">ID4</italic> were significant at the L1-L2 boundary (as detailed in Supplementary Table S4C of the spatialLIBD paper).</p></sec><sec id="s2d2" hwp:id="sec-10"><title hwp:id="title-13">Breast cancer</title><p hwp:id="p-32">we similarly compared cliff genes identified by StarTrail with genes detected by the SVG methods. In this dataset, spatialDE, nnSVG, and Spark-X identified 192, 458, and 1,585 SVGs respectively. StarTrail identified 463 cliff genes at either of the two cancer epithelial cell boundaries, and the weighted-sum of the two boundaries gradients led to 100 cliff genes (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-8" hwp:rel-id="F3">Fig. 3i</xref>, Supplementary Table 2, <xref rid="figS5" ref-type="fig" hwp:id="xref-fig-19-1" hwp:rel-id="F19">Supplementary Fig. 5</xref>, <xref rid="figS6" ref-type="fig" hwp:id="xref-fig-20-1" hwp:rel-id="F20">6</xref>). Note that SoMDE did not detect any SVGs and was thus omitted from further analysis. Among the 463 (100) cliff genes, 212 (12) were StarTrail-exclusive, not recognized by any SVG method. The 12 genes (<xref rid="figS6" ref-type="fig" hwp:id="xref-fig-20-2" hwp:rel-id="F20">Supplementary Fig. 6</xref>) manifest high expression in either cancer in situ region (e.g. <italic toggle="yes">IGSF3</italic>) or the immune infiltrate region (e.g. <italic toggle="yes">MBP</italic>). <italic toggle="yes">IGSF3</italic> (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-9" hwp:rel-id="F3">Fig. 3h</xref>), exhibiting the largest gradient in StarTrail’s multi-boundaries analysis but undetected by all SVG methods, is part of the immunoglobulin superfamily (IgSF), which includes members whose gene expression levels are known to vary in breast cancer, making them potential prognostic biomarkers. [<xref ref-type="bibr" rid="c44" hwp:id="xref-ref-44-1" hwp:rel-id="ref-44">44</xref>]. <italic toggle="yes">IGSF3</italic>, in particular, has been highlighted as a possible target for Chimeric antigen receptor (CAR) therapy and is considered essential in 80% of head and neck squamous cell carcinoma cell lines (classified as “common-essential” by DepMap) [<xref ref-type="bibr" rid="c45" hwp:id="xref-ref-45-1" hwp:rel-id="ref-45">45</xref>]. Moreover, TCGA DE analysis between tumor and adjacent normal tissue revealed significant differences in over 90% of IgSF genes in at least one cancer type. Specifically, <italic toggle="yes">IGSF3</italic> demonstrates a marked over-expression in breast invasive carcinoma samples from TCGA, highlighting its relevance and potential impact in the context of breast cancer [<xref ref-type="bibr" rid="c46" hwp:id="xref-ref-46-1" hwp:rel-id="ref-46">46</xref>].</p><p hwp:id="p-33">StarTrail not only identifies cliff genes that offer insights missed by existing methods but also uncovers undetected pathways. Gene Ontology (GO) analysis was conducted to delve into the pathways of these cliff genes (<xref rid="fig3" ref-type="fig" hwp:id="xref-fig-3-10" hwp:rel-id="F3">Fig. 3g</xref>) [<xref ref-type="bibr" rid="c47" hwp:id="xref-ref-47-1" hwp:rel-id="ref-47">47</xref>]. Cliff genes identified at either boundary revealed 54 pathways missed by SVG methods; and using cliff genes detected with the multi-boundaries approach revealed 30 missed pathways (Supplementary Tables 3-8). Notably, GO results highlight terms pertinent to the response to interleukin-1 (IL-1, GO:0071347 and GO:0070555, <xref rid="figS7" ref-type="fig" hwp:id="xref-fig-21-1" hwp:rel-id="F21">Supplementary Fig. 7</xref>), one of the major pro-inflammatory cytokines known to be elevated in various tumor types including breast cancer. IL-1 has been linked to tumor progression through its role in promoting the expression of genes involved in metastasis, angiogenesis, and growth factors [<xref ref-type="bibr" rid="c48" hwp:id="xref-ref-48-1" hwp:rel-id="ref-48">48</xref>]. Among the 30 GO terms uncovered from StarTrail’s multi-boundaries approach, multiple are related to immune response, including the regulation of natural killer cell mediated immunity (GO:0002715, <xref rid="figS8" ref-type="fig" hwp:id="xref-fig-22-1" hwp:rel-id="F22">Supplementary Fig. 8</xref>) and the positive regulation of myeloid leukocyte mediated immunity (GO:0002888), presenting granular insights into the immune dynamics within the tumor microenvironment. As StarTrail is designed to detect localized spatial patterns, it holds the promise of offering superior power and resolution in scenarios where multiple unconnected hot spots occur around boundaries, uncovering localized patterns that may well be overlooked by traditional SVG detection methods.</p><p hwp:id="p-34">So far we have concentrated on the absolute magnitudes of the gradients. However, the gradients’ sign can provide orthogonal insights. Focusing on cliff genes characterized by a negative gradient (indicating lower expression within the cancerous regions), we observed a strong association with immune-related processes (Supplementary Table 5). Among the top 15 significant pathways identified by StarTrail, an impressive 14 were directly tied to immunity, antigens, and B cell functions. This starkly contrasts with the findings from SVG detection methods, where nnSVG, Spark-X, and SpatialDE identified 4, 3 and 6 pathways related to immunity.</p></sec></sec><sec id="s2e" hwp:id="sec-11"><label>2.5</label><title hwp:id="title-14">Enhanced downstream analysis with StarTrail inferred finer spatial dynamics</title><p hwp:id="p-35">In addition to StarTrail’s direct benefits of precisely demarcating boundaries and detecting cliff genes, integrating StarTrail inferred gradient information can significantly enhance the inferential capabilities of traditional spatial methodology. Although a complete evaluation is beyond the scope of this work, we demonstrate in this section how harnessing the output of StarTrail inferred spatial dynamics can bolster spatial domain detection, commonly referred to as clustering analysis.</p><p hwp:id="p-36">It is important to note that we are not introducing a new clustering method. Rather, we performed integration by feeding StarTrail inferred spatial dynamics as input to existing clustering methods. We propose two approaches for integration: first, using the <italic toggle="yes">L</italic><sup>2</sup> norm of gradients as the input for clustering, and second, combining the original omics data with the <italic toggle="yes">L</italic><sup>2</sup> norm of gradients as input. Since high <italic toggle="yes">L</italic><sup>2</sup> values typically correspond to areas of rapid omics profile change, which often corresponds to biological boundaries, the inclusion of <italic toggle="yes">L</italic><sup>2</sup> enhances the distinction between different spatial regions. To illustrate this, we show four genes with diverse expression patterns (<xref rid="fig4" ref-type="fig" hwp:id="xref-fig-4-1" hwp:rel-id="F4">Fig. 4a</xref>). When <italic toggle="yes">L</italic><sup>2</sup> values are included with gene expression, the contrast between areas enriched for each gene becomes more pronounced.</p><p hwp:id="p-37">We applied several clustering algorithms, including both traditional clustering methods like K-means [<xref ref-type="bibr" rid="c49" hwp:id="xref-ref-49-1" hwp:rel-id="ref-49">49</xref>, <xref ref-type="bibr" rid="c50" hwp:id="xref-ref-50-1" hwp:rel-id="ref-50">50</xref>] and spatial clustering methods: BayesSpace [<xref ref-type="bibr" rid="c12" hwp:id="xref-ref-12-1" hwp:rel-id="ref-12">12</xref>], SpaGCN [<xref ref-type="bibr" rid="c11" hwp:id="xref-ref-11-1" hwp:rel-id="ref-11">11</xref>], Stardust [<xref ref-type="bibr" rid="c14" hwp:id="xref-ref-14-1" hwp:rel-id="ref-14">14</xref>] and stLearn [<xref ref-type="bibr" rid="c13" hwp:id="xref-ref-13-1" hwp:rel-id="ref-13">13</xref>], to the DLPFC dataset based on gene expression alone, <italic toggle="yes">L</italic><sup>2</sup> norm alone, and the sum of gene expression and <italic toggle="yes">L</italic><sup>2</sup> (Gene+<italic toggle="yes">L</italic><sup>2</sup>). In our analysis, we used two sets of genes: highly variable genes (selected by SPARK [<xref ref-type="bibr" rid="c9" hwp:id="xref-ref-9-1" hwp:rel-id="ref-9">9</xref>]) and cliff gene sets (genes with absolute boundary gradients greater than 5 at any boundary). The choice of gene set (either SVGs or cliff genes) and input metric(s) can both influence clustering outcomes, with no universally optimal selection for all methods. However, when we compared the best Adjusted Rand Index (ARI) [<xref ref-type="bibr" rid="c51" hwp:id="xref-ref-51-1" hwp:rel-id="ref-51">51</xref>] scores for clustering methods with and without gradients, the inclusion of StarTrail inferred gradients consistently outperformed analyses relying solely on gene expression (<xref rid="fig4" ref-type="fig" hwp:id="xref-fig-4-2" hwp:rel-id="F4">Fig. 4b,c</xref>, <xref rid="figED10" ref-type="fig" hwp:id="xref-fig-14-1" hwp:rel-id="F14">Extended Data Fig. 10</xref>, <xref rid="figS11" ref-type="fig" hwp:id="xref-fig-25-3" hwp:rel-id="F25">Supplementary Fig. 11</xref>). Notably, K-Means clustering with <italic toggle="yes">L</italic><sup>2</sup> demonstrated a marked improvement, revealing a distinct and drastically clearer layer structure (<xref rid="fig4" ref-type="fig" hwp:id="xref-fig-4-3" hwp:rel-id="F4">Fig. 4b,c</xref>).</p><p hwp:id="p-38">These findings underscore the potential of including StarTrail inferred spatial dynamics to enhance the resolution and accuracy of clustering analyses in spatial omics studies, offering a more comprehensive and granular understanding of tissue architecture and function.</p></sec></sec><sec id="s3" hwp:id="sec-12"><label>3</label><title hwp:id="title-15">Discussion</title><p hwp:id="p-39">StarTrail represents a pioneering and paradigm-shifting approach in the spatial omics field as the first method that employs a rigorous yet efficient spatial gradient framework to empower inference at highly localized, potentially disjoint regions that existing methods have little power. This innovative technique substantially enhances our understanding of spatial omics feature by providing directional, quantitative, and high-resolution gradient (i.e., rate of change) information. Augmenting the original omics measurements with such gradient information, StarTrail sheds new light on tissue structure and function.</p><p hwp:id="p-40">The fundamental strength of StarTrail lies in its utilization of gradient information, which has multiple advantages in practical applications. The gradient flow map, a central feature of StarTrail, provides an intuitive visualization of omics dynamics, revealing the direction of changes across tissue samples. This is particularly valuable in identifying areas of rapid transition within the tissue, thereby aiding in the detection of critical biological boundaries. Another significant application is the identification of “cliff genes” – genes that exhibit dramatic shifts in expression at specific tissue interfaces. This ability to pinpoint genes that are highly variable across spatial boundaries has profound implications beyond the routine analysis of SVGs. Furthermore, StarTrail is specifically tailored to precisely demarcate local boundaries and to detect local omics patterns near pre-defined boundaries. It is therefore complementary to SVG methods proposed in the literature. For instance, significant SVGs with small gradient values at any boundary suggest alternative boundaries unknown to us, or the presence of potential new clusters (Supplementary Note, <xref rid="figS9" ref-type="fig" hwp:id="xref-fig-23-1" hwp:rel-id="F23">Supplementary Figs. 9</xref>, <xref rid="figS10" ref-type="fig" hwp:id="xref-fig-24-1" hwp:rel-id="F24">10</xref>). For downstream analysis such as clustering analysis, the inclusion of gradient information significantly enhances the resolution and accuracy of identifying distinct spatial domains within tissue samples.</p><p hwp:id="p-41">Another pivotal advantage of StarTrail is its computational efficiency. Utilizing NNGP models, Star-Trail dramatically reduces the computational cost required for analysis – from 40 hours to a mere 2 minutes per gene per slide, as evidenced in our analysis of the DLPFC dataset. This efficiency is achieved without sacrificing accuracy, making StarTrail a practical tool for large-scale studies. Furthermore, Star-Trail’s flexibility in handling multiple data types – from measured transcriptomics or proteomics data to computationally inferred annotations – makes it a versatile tool for various spatial omics applications and the analysis of co-assays.</p><p hwp:id="p-42">Looking forward, there are several promising directions for extending StarTrail’s capabilities. For example, the development of spatial gradient processes that model multiple omics features simultaneously is highly warranted. This could provide a more comprehensive understanding of the complex interactions among multiple genes or proteins within the same spatial context. Another promising direction is the development of clustering methods that are specially designed to incorporate gradient information. Such methods could offer a more powerful approach to segmenting tissues based on dynamic changes, leading to more accurate and biologically relevant clustering outcomes.</p><p hwp:id="p-43">In conclusion, the introduction of StarTrail to the spatial omics toolkit has the potential to open doors to a deeper understanding of the spatial dynamics of biological tissues. This novel approach promises to enhance our comprehension of tissue structure and function by leveraging the rich information contained within spatial gradients.</p></sec></body><back><ack hwp:id="ack-1"><title hwp:id="title-16">Acknowledgments</title><p hwp:id="p-44">DL was supported by NIH grants R01 AG079291, R56 LM013784, R01 HL149683 and UM1 TR004406. JC, GPG, YL are supported by a Department of Defense Breast Cancer Research Program Award (HT94252310961). JC, QS and YL are supported by R01 AG079291, U01 HG011720, U01 DA052713, and P50 HD103573.</p></ack><sec id="s4" hwp:id="sec-13"><label>4</label><title hwp:id="title-17">Method</title><sec id="s4a" hwp:id="sec-14"><title hwp:id="title-18">The StarTrail pipeline</title><p hwp:id="p-45">The analysis in StarTrail begins by inputting data from spatial omics experiments or its annotation. For omics data, such as gene expression or protein abundance, we first implement a smoothing step to enhance signal quality and reduce noise. The omics measurement of each spot/cell is smoothed as the weighted sum of its neighbors and itself. This is followed by data normalization using the SCTransform function in Seurat [<xref ref-type="bibr" rid="c52" hwp:id="xref-ref-52-1" hwp:rel-id="ref-52">52</xref>]. For cell type proporiton, only the smoothing step is applied. This initial phase refines raw data, preparing them for subsequent analysis.</p><p hwp:id="p-46">The core of StarTrail’s workflow involves employing the NNGP [<xref ref-type="bibr" rid="c29" hwp:id="xref-ref-29-3" hwp:rel-id="ref-29">29</xref>] [<xref ref-type="bibr" rid="c30" hwp:id="xref-ref-30-3" hwp:rel-id="ref-30">30</xref>]. Here, we model the relationship between coordinates <italic toggle="yes">s</italic> and omics feature (e.g., gene expression) <italic toggle="yes">Y</italic> using the following Bayesian hierarchical model [<xref ref-type="bibr" rid="c53" hwp:id="xref-ref-53-1" hwp:rel-id="ref-53">53</xref>] [<xref ref-type="bibr" rid="c54" hwp:id="xref-ref-54-1" hwp:rel-id="ref-54">54</xref>]
<disp-formula id="ueqn1" hwp:id="disp-formula-1">
<graphic xlink:href="593025v1_ueqn1.gif" position="float" orientation="portrait" hwp:id="graphic-5"/>
</disp-formula>
where <italic toggle="yes">μ</italic> is a mean function, <italic toggle="yes">Z</italic> ∼ GP(0, <italic toggle="yes">K</italic>) is a zero mean GP with covariance function <italic toggle="yes">K</italic>, and <italic toggle="yes">ϵ</italic>(<italic toggle="yes">s</italic>) ∼ <italic toggle="yes">N</italic> (0, <italic toggle="yes">τ</italic> <sup>2</sup>) is a zero mean white-noise process capturing measurement error, also known as nuggets. In particular, the GP <italic toggle="yes">Z</italic> is a stochastic process with finite d imensional r ealization b eing multivariate Gaussian: for any <italic toggle="yes">s</italic><sub>1</sub>, …, <italic toggle="yes">s</italic><sub><italic toggle="yes">n</italic></sub> ∈ ℝ<sup>2</sup>,
<disp-formula id="ueqn2" hwp:id="disp-formula-2">
<graphic xlink:href="593025v1_ueqn2.gif" position="float" orientation="portrait" hwp:id="graphic-6"/>
</disp-formula>
where ∑<sub><italic toggle="yes">ij</italic></sub> = <italic toggle="yes">K</italic>(<italic toggle="yes">s</italic><sub><italic toggle="yes">i</italic></sub>, <italic toggle="yes">s</italic><sub><italic toggle="yes">j</italic></sub>) is the covariance matrix calculated based on predefined kernels. Here we choose the Matérn kernel as the covariance function due to its flexibility in controlling smoothness [<xref ref-type="bibr" rid="c55" hwp:id="xref-ref-55-1" hwp:rel-id="ref-55">55</xref>]:
<disp-formula id="ueqn3" hwp:id="disp-formula-3">
<graphic xlink:href="593025v1_ueqn3.gif" position="float" orientation="portrait" hwp:id="graphic-7"/>
</disp-formula>
where <italic toggle="yes">K</italic><sub><italic toggle="yes">ν</italic></sub> is the modified Bessel function of the second kind.</p><p hwp:id="p-47">Traditional GP modeling is computationally intensive, primarily due to the inverse of the <italic toggle="yes">n</italic> by <italic toggle="yes">n</italic> covariance matrix ∑, at a cost of <italic toggle="yes">O</italic>(<italic toggle="yes">n</italic><sup>3</sup>). StarTrail, adopting the NNGP approach, significantly improves efficiency and scalability, crucial for handling large datasets common in spatial omics studies. NNGP, by focusing on the <italic toggle="yes">m</italic>-nearest neighbors and a sparse approximation to the covariance, reduces the computational cost to <italic toggle="yes">O</italic>(<italic toggle="yes">nm</italic><sup>3</sup>) ≪ <italic toggle="yes">O</italic>(<italic toggle="yes">n</italic><sup>3</sup>) [<xref ref-type="bibr" rid="c29" hwp:id="xref-ref-29-4" hwp:rel-id="ref-29">29</xref>] [<xref ref-type="bibr" rid="c30" hwp:id="xref-ref-30-4" hwp:rel-id="ref-30">30</xref>]. Here we chose <italic toggle="yes">m</italic> = 10.</p><p hwp:id="p-48">We implemented the spNNGP function (“response” algorithm) in the spNNGP package [<xref ref-type="bibr" rid="c30" hwp:id="xref-ref-30-5" hwp:rel-id="ref-30">30</xref>]. The priors for the kernel parameters and additional tuning parameters are available in our GITHUB repository.</p><p hwp:id="p-49">After fitting the model, the next step is gradient estimation. In the traditional spatial gradient process, the gradient ∇<italic toggle="yes">Z</italic> and curvature ∇<sup>2</sup><italic toggle="yes">Z</italic> are estimated by inferring joint distribution of [<italic toggle="yes">Z</italic>, ∇<italic toggle="yes">Z</italic>, ∇<sup>2</sup><italic toggle="yes">Z</italic>] with cross-covariance characterized by
<disp-formula id="ueqn4" hwp:id="disp-formula-4">
<graphic xlink:href="593025v1_ueqn4.gif" position="float" orientation="portrait" hwp:id="graphic-8"/>
</disp-formula>
Given the Matérn kernel is ⌈<italic toggle="yes">ν</italic>⌉ − 1 times mean square differentiable (MSD) [<xref ref-type="bibr" rid="c55" hwp:id="xref-ref-55-2" hwp:rel-id="ref-55">55</xref>], we set the prior of <italic toggle="yes">ν</italic> to be supported on (2, 5) to ensure that the GP with kernel <italic toggle="yes">K</italic> is 2-times MSD, i.e., ∇<sup>2</sup><italic toggle="yes">Z</italic> is well-defined. Due to the nature of NNGP, and because we do not fix <italic toggle="yes">ν</italic> in the kernel, the closed form of ∇<italic toggle="yes">K</italic> is not available. As a result, we propose to estimate the gradients using finite differences [<xref ref-type="bibr" rid="c56" hwp:id="xref-ref-56-1" hwp:rel-id="ref-56">56</xref>], a mathematical method for approximating derivatives. Here, we explain the case for 2D coordinates, calculating the gradient in the direction of <italic toggle="yes">e</italic><sub>1</sub> = (1, 0) and <italic toggle="yes">e</italic><sub>2</sub> = (0, 1). We denote the posterior mean of <italic toggle="yes">f</italic> as <inline-formula hwp:id="inline-formula-3"><inline-graphic xlink:href="593025v1_inline3.gif" hwp:id="inline-graphic-3"/></inline-formula>, and for any given location <italic toggle="yes">s</italic>, <inline-formula hwp:id="inline-formula-4"><inline-graphic xlink:href="593025v1_inline4.gif" hwp:id="inline-graphic-4"/></inline-formula> and <inline-formula hwp:id="inline-formula-5"><inline-graphic xlink:href="593025v1_inline5.gif" hwp:id="inline-graphic-5"/></inline-formula> are approximated by
<disp-formula id="ueqn5" hwp:id="disp-formula-5">
<graphic xlink:href="593025v1_ueqn5.gif" position="float" orientation="portrait" hwp:id="graphic-9"/>
</disp-formula>
where <italic toggle="yes">h</italic> is a sufficiently small step size specified by users. According to our empirical observation, we recommended and set as default <italic toggle="yes">h</italic> = 0.8<italic toggle="yes">ι</italic>, where <italic toggle="yes">ι</italic> = min<sub><italic toggle="yes">i,j</italic></sub> ∥<italic toggle="yes">s</italic><sub><italic toggle="yes">i</italic></sub> − <italic toggle="yes">s</italic><sub><italic toggle="yes">j</italic></sub>∥ is known as the minimal separation of the locations <inline-formula hwp:id="inline-formula-6"><inline-graphic xlink:href="593025v1_inline6.gif" hwp:id="inline-graphic-6"/></inline-formula> .</p><p hwp:id="p-50">Similarly, the curvatures at <italic toggle="yes">s</italic> are approximated by:
<disp-formula id="ueqn6" hwp:id="disp-formula-6">
<graphic xlink:href="593025v1_ueqn6.gif" position="float" orientation="portrait" hwp:id="graphic-10"/>
</disp-formula>
Consequently, the gradient at <italic toggle="yes">s</italic> can be estimated by <inline-formula hwp:id="inline-formula-7"><inline-graphic xlink:href="593025v1_inline7.gif" hwp:id="inline-graphic-7"/></inline-formula>.</p><p hwp:id="p-51">Wombling analysis in StarTrail calculates gradients along a given curve <italic toggle="yes">γ</italic> : [0, <italic toggle="yes">T</italic>] → ℝ<sup>2</sup>, taking segment points <inline-formula hwp:id="inline-formula-8"><inline-graphic xlink:href="593025v1_inline8.gif" hwp:id="inline-graphic-8"/></inline-formula> as input and adopting a piece-wise linear approximation. For each line segment <italic toggle="yes">γ</italic>(<italic toggle="yes">t</italic>) with starting point <inline-formula hwp:id="inline-formula-9"><inline-graphic xlink:href="593025v1_inline9.gif" hwp:id="inline-graphic-9"/></inline-formula> and ending point <inline-formula hwp:id="inline-formula-10"><inline-graphic xlink:href="593025v1_inline10.gif" hwp:id="inline-graphic-10"/></inline-formula>, let <inline-formula hwp:id="inline-formula-11"><inline-graphic xlink:href="593025v1_inline11.gif" hwp:id="inline-graphic-11"/></inline-formula> be the unit vector representing this segment. Then the normal unit vector <italic toggle="yes">v</italic><sub><italic toggle="yes">i</italic></sub> = <italic toggle="yes">Ru</italic><sub><italic toggle="yes">i</italic></sub> with <inline-formula hwp:id="inline-formula-12"><inline-graphic xlink:href="593025v1_inline12.gif" hwp:id="inline-graphic-12"/></inline-formula> being the rotation matrix of <inline-formula hwp:id="inline-formula-13"><inline-graphic xlink:href="593025v1_inline13.gif" hwp:id="inline-graphic-13"/></inline-formula>.</p><p hwp:id="p-52">An approximate to <inline-formula hwp:id="inline-formula-14"><inline-graphic xlink:href="593025v1_inline14.gif" hwp:id="inline-graphic-14"/></inline-formula> is given by the Simpson’s rule [<xref ref-type="bibr" rid="c57" hwp:id="xref-ref-57-1" hwp:rel-id="ref-57">57</xref>][<xref ref-type="bibr" rid="c58" hwp:id="xref-ref-58-1" hwp:rel-id="ref-58">58</xref>]
<disp-formula id="ueqn7" hwp:id="disp-formula-7">
<graphic xlink:href="593025v1_ueqn7.gif" position="float" orientation="portrait" hwp:id="graphic-11"/>
</disp-formula>
where
<disp-formula id="ueqn8" hwp:id="disp-formula-8">
<graphic xlink:href="593025v1_ueqn8.gif" position="float" orientation="portrait" hwp:id="graphic-12"/>
</disp-formula>
Then the average gradient along the curve <italic toggle="yes">γ</italic> is approximated as
<disp-formula id="ueqn9" hwp:id="disp-formula-9">
<graphic xlink:href="593025v1_ueqn9.gif" position="float" orientation="portrait" hwp:id="graphic-13"/>
</disp-formula>
For easier comparison across curves, we scaled the approximated gradient of each curve using its piece-wise linearly approximated length <inline-formula hwp:id="inline-formula-15"><inline-graphic xlink:href="593025v1_inline15.gif" hwp:id="inline-graphic-15"/></inline-formula>:
<disp-formula id="ueqn10" hwp:id="disp-formula-10">
<graphic xlink:href="593025v1_ueqn10.gif" position="float" orientation="portrait" hwp:id="graphic-14"/>
</disp-formula>
When considering multiple boundaries <italic toggle="yes">γ</italic><sub>1</sub>, …, <italic toggle="yes">γ</italic><sub><italic toggle="yes">k</italic></sub>, the weighted-sum gradient is approximated as
<disp-formula id="ueqn11" hwp:id="disp-formula-11">
<graphic xlink:href="593025v1_ueqn11.gif" position="float" orientation="portrait" hwp:id="graphic-15"/>
</disp-formula>
</p></sec><sec id="s4b" hwp:id="sec-15"><title hwp:id="title-19">Boundary detection in StarTrail</title><p hwp:id="p-53">StarTrail excels in precisely detecting boundaries within spatial omics data. This process begins by identifying spots or cells that exhibit high <italic toggle="yes">L</italic><sup>2</sup>, specifically those exceeding a predefined quantile threshold (default is 0.9). Subsequently, the dbSCAN algorithm [<xref ref-type="bibr" rid="c59" hwp:id="xref-ref-59-1" hwp:rel-id="ref-59">59</xref>] is deployed to cluster these points, creating an initial map of areas where substantial changes coalesce, and thus suggesting potential boundary locations. For each cluster identified by dbSCAN, a principal curve is fitted. This curve acts as a smooth line that passes through the center of each cluster, effectively delineating the boundary. The default number of nodes in each principal curve is set as the number of points in the cluster divided by ten. Through this methodical approach, StarTrail offers a nuanced and precise way to detect boundaries within spatial omics data. By combining rigorous statistical techniques with advanced clustering algorithms, StarTrail effectively highlights regions of rapid changes, providing valuable insights into the spatial organization and underlying biological dynamics of the sampled tissue.</p></sec><sec id="s4c" hwp:id="sec-16"><title hwp:id="title-20">Cliff gene gradient threshold</title><p hwp:id="p-54">For DLPFC, we used 0.25<italic toggle="yes">σ</italic><sub>0.9</sub><italic toggle="yes">/ι</italic> as the threshold, where. <italic toggle="yes">σ</italic><sub>0.9</sub> is the 0.9 quantile of the standard deviation of smoothed gene expression, and ι is the minimal separation between spots. 0.25 here accounts for the change spanning across two rows/columns of spots over half of the given boundary. For HER2+ analysis, we used 0.5<italic toggle="yes">σ</italic><sub>0.9ι</sub>. We note here that this threshold reflects the actual change of gene expression and is therefore not a probability. Thus, users could adjust this threshold based on their needs.</p></sec><sec id="s4d" hwp:id="sec-17"><title hwp:id="title-21">Gene+<italic toggle="yes">L</italic><sup>2</sup></title><p hwp:id="p-55">Gene+<italic toggle="yes">L</italic><sup>2</sup> is designed to augment the original gene expression data with a gradual spatial change information. To avoid overshadowing the inherent characteristics of gene expression, we meticulously regulate the magnitude of the <italic toggle="yes">L</italic><sup>2</sup> term. Specifically, we use a scaled <italic toggle="yes">L</italic><sup>2</sup> (divided by the range max <italic toggle="yes">L</italic><sup>2</sup> − min <italic toggle="yes">L</italic><sup>2</sup>). This approach ensures that the alteration to each value does not exceed 1, preserving the nuanced nature of the original data.</p></sec><sec id="s4e" hwp:id="sec-18"><title hwp:id="title-22">Simulations</title><p hwp:id="p-56">To mimic a spectrum of patterns in ST data, we simulated data under several distinct settings. For simulation 1 (<xref rid="figED1" ref-type="fig" hwp:id="xref-fig-5-3" hwp:rel-id="F5">Extended Data Fig. 1</xref> top row), we simulated data from trigonometric functions:
<disp-formula id="ueqn12" hwp:id="disp-formula-12">
<graphic xlink:href="593025v1_ueqn12.gif" position="float" orientation="portrait" hwp:id="graphic-16"/>
</disp-formula>
where <italic toggle="yes">s</italic><sub><italic toggle="yes">g</italic>,1</sub> and <italic toggle="yes">s</italic><sub><italic toggle="yes">g</italic>,2</sub> are vectors spaced equally from 0 to 1 in increments of 0.05. In the second simulation (<xref rid="figED1" ref-type="fig" hwp:id="xref-fig-5-4" hwp:rel-id="F5">Extended Data Fig. 1</xref> second row), we created a linear gradient pattern, or ‘streak’, where the data transitions gradually from a background level to a higher level towards the center in the (0, 1) direction. The third simulation involved the creation of a ‘hot-spot’ pattern (<xref rid="figED1" ref-type="fig" hwp:id="xref-fig-5-5" hwp:rel-id="F5">Extended Data Fig. 1</xref> third row), which similarly sees a gradual change from the background parameter to a center of a circle. The fourth simulation produced a ‘layer’ pattern (<xref rid="figED1" ref-type="fig" hwp:id="xref-fig-5-6" hwp:rel-id="F5">Extended Data Fig. 1</xref> bottom row), characterized by a sharp transition from one level to another, resembling discrete layers. All the simulated data can be accessed in our GitHub repository.</p></sec><sec id="s4f" hwp:id="sec-19"><title hwp:id="title-23">Clustering analysis</title><p hwp:id="p-57">We performed clustering with five methods on gene expression, <italic toggle="yes">L</italic><sup>2</sup>, or Gene+<italic toggle="yes">L</italic><sup>2</sup> to cluster spatial locations, using SVGs (top 3000 SVGs selected by SPARK[<xref ref-type="bibr" rid="c9" hwp:id="xref-ref-9-2" hwp:rel-id="ref-9">9</xref>]) or cliff gene sets (absolute gradient at any boundary greater than 5). The five methods include BayesSpace [<xref ref-type="bibr" rid="c12" hwp:id="xref-ref-12-2" hwp:rel-id="ref-12">12</xref>], K-Means [<xref ref-type="bibr" rid="c49" hwp:id="xref-ref-49-2" hwp:rel-id="ref-49">49</xref>, <xref ref-type="bibr" rid="c50" hwp:id="xref-ref-50-2" hwp:rel-id="ref-50">50</xref>], SpaGCN [<xref ref-type="bibr" rid="c11" hwp:id="xref-ref-11-2" hwp:rel-id="ref-11">11</xref>], Stardust [<xref ref-type="bibr" rid="c14" hwp:id="xref-ref-14-2" hwp:rel-id="ref-14">14</xref>] and stLearn [<xref ref-type="bibr" rid="c13" hwp:id="xref-ref-13-2" hwp:rel-id="ref-13">13</xref>]. BayesSpace [<xref ref-type="bibr" rid="c12" hwp:id="xref-ref-12-3" hwp:rel-id="ref-12">12</xref>], a fully Bayesian method, leverages spatial neighborhoods information to enhance resolution in ST data and perform clustering. K-Means [<xref ref-type="bibr" rid="c60" hwp:id="xref-ref-60-1" hwp:rel-id="ref-60">60</xref>] partitions a dataset into K distinct, non-overlapping clusters by minimizing the variance within each cluster. SpaGCN [<xref ref-type="bibr" rid="c11" hwp:id="xref-ref-11-3" hwp:rel-id="ref-11">11</xref>] employs a graph convolutional network to identify SVGs and spatial domains. Stardust [<xref ref-type="bibr" rid="c14" hwp:id="xref-ref-14-3" hwp:rel-id="ref-14">14</xref>] integrates gene expression with spatial location to create a pairwise distance matrix for clustering using the Louvain algorithm [<xref ref-type="bibr" rid="c61" hwp:id="xref-ref-61-1" hwp:rel-id="ref-61">61</xref>]. stLearn [<xref ref-type="bibr" rid="c13" hwp:id="xref-ref-13-3" hwp:rel-id="ref-13">13</xref>] utilizes Spatial Morphological gene Expression normalization (SME normalization) based on gene expression, spatial location, and image data, followed by K-Means clustering [<xref ref-type="bibr" rid="c60" hwp:id="xref-ref-60-2" hwp:rel-id="ref-60">60</xref>]. For each method, we adopted the standard pipelines and followed their recommended parameter settings. For BayesSpace, SpaGCN, Stardust, and stLearn, we applied log-transformation and Principal Component Analysis (PCA) to all three types of data (i.e., gene expression, spatial location, and image data). We did not perform any transformations on the input for K-Means. For Gene+<italic toggle="yes">L</italic><sup>2</sup> in BayesSpace, we add rounded and scaled <italic toggle="yes">L</italic><sup>2</sup> to the gene expression matrix, as the developers recommended when the data contain values between 0 and 1. In BayesSpace, K-Means, SpaGCN, and stLearn, the number of clusters is set based on the ground truth (here we used 7 for DLPFC 151676). Stardust does not have the option to specify the number of clusters.</p></sec><sec id="s4g" hwp:id="sec-20"><title hwp:id="title-24">De-convolution analysis</title><p hwp:id="p-58">We applied RCTD for cell type de-convolution to the HER2+ breast cancer data. RCTD is a computational technique that uses single-cell RNA-seq data to decompose cell type mixtures while adjusting for technical variations. We set the doublet mode to be “full” to obtain the proportion of all the cell types in each spot.</p></sec><sec id="s4h" hwp:id="sec-21"><title hwp:id="title-25">Gene ontology analysis</title><p hwp:id="p-59">We performed gene ontology (GO) analysis using R package GO.db. We included three GO categories, molecular function (MF), biological process (BP) and cellular component (CC).</p></sec></sec><sec id="s5" hwp:id="sec-22"><title hwp:id="title-26">Data availability</title><p hwp:id="p-60">The DLPFC [<xref ref-type="bibr" rid="c16" hwp:id="xref-ref-16-7" hwp:rel-id="ref-16">16</xref>] data is obtained from <ext-link l:rel="related" l:ref-type="uri" l:ref="https://research.libd.org/globus/" ext-link-type="uri" xlink:href="https://research.libd.org/globus/" hwp:id="ext-link-2">https://research.libd.org/globus/</ext-link>. The breast cancer data are obtained from <ext-link l:rel="related" l:ref-type="uri" l:ref="https://doi.org/10.5281/zenodo.4751624" ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.4751624" hwp:id="ext-link-3">https://doi.org/10.5281/zenodo.4751624</ext-link> (ST) and <ext-link l:rel="related" l:ref-type="uri" l:ref="https://singlecell.broadinstitute.org/single_cell/study/SCP1039" ext-link-type="uri" xlink:href="https://singlecell.broadinstitute.org/single_cell/study/SCP1039" hwp:id="ext-link-4">https://singlecell.broadinstitute.org/single_cell/study/SCP1039</ext-link> (scRNA-seq reference). The CODEX [<xref ref-type="bibr" rid="c36" hwp:id="xref-ref-36-2" hwp:rel-id="ref-36">36</xref>] data is shared by the MaxFuse [<xref ref-type="bibr" rid="c37" hwp:id="xref-ref-37-2" hwp:rel-id="ref-37">37</xref>] authors.</p></sec><sec id="s6" hwp:id="sec-23"><title hwp:id="title-27">Code availability</title><p hwp:id="p-61">All code used in this study, including the StarTrail software, can be found at <ext-link l:rel="related" l:ref-type="uri" l:ref="https://github.com/JiawenChenn/StarTrail" ext-link-type="uri" xlink:href="https://github.com/JiawenChenn/StarTrail" hwp:id="ext-link-5">https://github.com/JiawenChenn/StarTrail</ext-link>.</p></sec><ref-list hwp:id="ref-list-1"><title hwp:id="title-28">References</title><ref id="c1" hwp:id="ref-1"><label>[1]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.1" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-1"><string-name name-style="western" hwp:sortable="Chen M."><surname>Chen</surname>, <given-names>M.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Maier K."><surname>Maier</surname>, <given-names>K.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Ritenour D."><surname>Ritenour</surname>, <given-names>D.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Sun C."><surname>Sun</surname>, <given-names>C.</given-names></string-name> <article-title hwp:id="article-title-2">Improving global tbi tracking and prevention: An environmental science approach</article-title>. <source hwp:id="source-1">Global Journal of Health Science</source> <volume>14</volume>, <fpage>20</fpage> (<year>2022</year>).</citation></ref><ref id="c2" hwp:id="ref-2"><label>[2]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.2" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-2"><string-name name-style="western" hwp:sortable="Zhou X."><surname>Zhou</surname>, <given-names>X.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-3">Choice of voxel-based morphometry processing pipeline drives variability in the location of neuroanatomical brain markers</article-title>. <source hwp:id="source-2">Communications Biology</source> <volume>5</volume>, <fpage>913</fpage> (<year>2022</year>).</citation></ref><ref id="c3" hwp:id="ref-3"><label>[3]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.3" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-3"><string-name name-style="western" hwp:sortable="Lewis S. M."><surname>Lewis</surname>, <given-names>S. M.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-4">Spatial omics and multiplexed imaging to explore cancer biology</article-title>. <source hwp:id="source-3">Nature methods</source> <volume>18</volume>, <fpage>997</fpage>–<lpage>1012</lpage> (<year>2021</year>).</citation></ref><ref id="c4" hwp:id="ref-4" hwp:rev-id="xref-ref-4-1"><label>[4]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.4" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-4"><string-name name-style="western" hwp:sortable="Marx V."><surname>Marx</surname>, <given-names>V.</given-names></string-name> <article-title hwp:id="article-title-5">Method of the year: spatially resolved transcriptomics</article-title>. <source hwp:id="source-4">Nature methods</source> <volume>18</volume>, <fpage>9</fpage>–<lpage>14</lpage> (<year>2021</year>).</citation></ref><ref id="c5" hwp:id="ref-5" hwp:rev-id="xref-ref-5-1 xref-ref-5-2 xref-ref-5-3"><label>[5]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.5" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-5"><string-name name-style="western" hwp:sortable="Ståhl P. L."><surname>Ståhl</surname>, <given-names>P. L.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-6">Visualization and analysis of gene expression in tissue sections by spatial transcriptomics</article-title>. <source hwp:id="source-5">Science</source> <volume>353</volume>, <fpage>78</fpage>–<lpage>82</lpage> (<year>2016</year>).</citation></ref><ref id="c6" hwp:id="ref-6" hwp:rev-id="xref-ref-6-1"><label>[6]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.6" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-6"><string-name name-style="western" hwp:sortable="Stickels R. R."><surname>Stickels</surname>, <given-names>R. R.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-7">Highly sensitive spatial transcriptomics at near-cellular resolution with slide-seqv2</article-title>. <source hwp:id="source-6">Nature biotechnology</source> <volume>39</volume>, <fpage>313</fpage>–<lpage>319</lpage> (<year>2021</year>).</citation></ref><ref id="c7" hwp:id="ref-7" hwp:rev-id="xref-ref-7-1"><label>[7]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.7" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-7"><string-name name-style="western" hwp:sortable="Chen K. H."><surname>Chen</surname>, <given-names>K. H.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Boettiger A. N."><surname>Boettiger</surname>, <given-names>A. N.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Moffitt J. R."><surname>Moffitt</surname>, <given-names>J. R.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Wang S."><surname>Wang</surname>, <given-names>S.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Zhuang X."><surname>Zhuang</surname>, <given-names>X.</given-names></string-name> <article-title hwp:id="article-title-8">Spatially resolved, highly multiplexed rna profiling in single cells</article-title>. <source hwp:id="source-7">Science</source> <volume>348</volume>, <fpage>aaa6090</fpage> (<year>2015</year>).</citation></ref><ref id="c8" hwp:id="ref-8" hwp:rev-id="xref-ref-8-1 xref-ref-8-2"><label>[8]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.8" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-8"><string-name name-style="western" hwp:sortable="Goltsev Y."><surname>Goltsev</surname>, <given-names>Y.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-9">Deep profiling of mouse splenic architecture with codex multiplexed imaging</article-title>. <source hwp:id="source-8">Cell</source> <volume>174</volume>, <fpage>968</fpage>–<lpage>981</lpage> (<year>2018</year>).</citation></ref><ref id="c9" hwp:id="ref-9" hwp:rev-id="xref-ref-9-1 xref-ref-9-2"><label>[9]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.9" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-9"><string-name name-style="western" hwp:sortable="Sun S."><surname>Sun</surname>, <given-names>S.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Zhu J."><surname>Zhu</surname>, <given-names>J.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Zhou X."><surname>Zhou</surname>, <given-names>X.</given-names></string-name> <article-title hwp:id="article-title-10">Statistical analysis of spatial expression patterns for spatially resolved transcriptomic studies</article-title>. <source hwp:id="source-9">Nature methods</source> <volume>17</volume>, <fpage>193</fpage>–<lpage>200</lpage> (<year>2020</year>).</citation></ref><ref id="c10" hwp:id="ref-10"><label>[10]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.10" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-10"><string-name name-style="western" hwp:sortable="Dries R."><surname>Dries</surname>, <given-names>R.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-11">Giotto: a toolbox for integrative analysis and visualization of spatial expression data</article-title>. <source hwp:id="source-10">Genome biology</source> <volume>22</volume>, <fpage>1</fpage>–<lpage>31</lpage> (<year>2021</year>).</citation></ref><ref id="c11" hwp:id="ref-11" hwp:rev-id="xref-ref-11-1 xref-ref-11-2 xref-ref-11-3"><label>[11]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.11" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-11"><string-name name-style="western" hwp:sortable="Hu J."><surname>Hu</surname>, <given-names>J.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-12">Spagcn: Integrating gene expression, spatial location and histology to identify spatial domains and spatially variable genes by graph convolutional network</article-title>. <source hwp:id="source-11">Nature methods</source> <volume>18</volume>, <fpage>1342</fpage>–<lpage>1351</lpage> (<year>2021</year>).</citation></ref><ref id="c12" hwp:id="ref-12" hwp:rev-id="xref-ref-12-1 xref-ref-12-2 xref-ref-12-3"><label>[12]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.12" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-12"><string-name name-style="western" hwp:sortable="Zhao E."><surname>Zhao</surname>, <given-names>E.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-13">Spatial transcriptomics at subspot resolution with bayesspace</article-title>. <source hwp:id="source-12">Nature biotechnology</source> <volume>39</volume>, <fpage>1375</fpage>–<lpage>1384</lpage> (<year>2021</year>).</citation></ref><ref id="c13" hwp:id="ref-13" hwp:rev-id="xref-ref-13-1 xref-ref-13-2 xref-ref-13-3"><label>[13]</label><citation publication-type="other" citation-type="journal" ref:id="2024.05.08.593025v1.13" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-13"><string-name name-style="western" hwp:sortable="Pham D."><surname>Pham</surname>, <given-names>D.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-14">stlearn: integrating spatial location, tissue morphology and gene expression to find cell types, cell-cell interactions and spatial trajectories within undissociated tissues</article-title>. <source hwp:id="source-13">BioRxiv</source> <fpage>2020</fpage>–<lpage>05</lpage> (<year>2020</year>).</citation></ref><ref id="c14" hwp:id="ref-14" hwp:rev-id="xref-ref-14-1 xref-ref-14-2 xref-ref-14-3"><label>[14]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.14" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-14"><string-name name-style="western" hwp:sortable="Avesani S."><surname>Avesani</surname>, <given-names>S.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-15">Stardust: improving spatial transcriptomics data analysis through space-aware modularity optimization-based clustering</article-title>. <source hwp:id="source-14">GigaScience</source> <volume>11</volume>, <fpage>giac075</fpage> (<year>2022</year>).</citation></ref><ref id="c15" hwp:id="ref-15" hwp:rev-id="xref-ref-15-1"><label>[15]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.15" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-15"><string-name name-style="western" hwp:sortable="Cable D. M."><surname>Cable</surname>, <given-names>D. M.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-16">Cell type-specific inference of differential expression in spatial transcriptomics</article-title>. <source hwp:id="source-15">Nature methods</source> <volume>19</volume>, <fpage>1076</fpage>–<lpage>1087</lpage> (<year>2022</year>).</citation></ref><ref id="c16" hwp:id="ref-16" hwp:rev-id="xref-ref-16-1 xref-ref-16-2 xref-ref-16-3 xref-ref-16-4 xref-ref-16-5 xref-ref-16-6 xref-ref-16-7"><label>[16]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.16" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-16"><string-name name-style="western" hwp:sortable="Maynard K. R."><surname>Maynard</surname>, <given-names>K. R.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-17">Transcriptome-scale spatial gene expression in the human dorsolateral prefrontal cortex</article-title>. <source hwp:id="source-16">Nature neuroscience</source> <volume>24</volume>, <fpage>425</fpage>–<lpage>436</lpage> (<year>2021</year>).</citation></ref><ref id="c17" hwp:id="ref-17"><label>[17]</label><citation publication-type="other" citation-type="journal" ref:id="2024.05.08.593025v1.17" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-17"><string-name name-style="western" hwp:sortable="Li C."><surname>Li</surname>, <given-names>C.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Thijssen J."><surname>Thijssen</surname>, <given-names>J.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Abdelaal T."><surname>Abdelaal</surname>, <given-names>T.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Hollt T."><surname>Hollt</surname>, <given-names>T.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Lelieveldt B."><surname>Lelieveldt</surname>, <given-names>B.</given-names></string-name> <article-title hwp:id="article-title-18">Spacewalker: Interactive gradient exploration for spatial transcriptomics data</article-title>. <source hwp:id="source-17">bioRxiv</source> <fpage>2023</fpage>–<lpage>03</lpage> (<year>2023</year>).</citation></ref><ref id="c18" hwp:id="ref-18"><label>[18]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.18" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-18"><string-name name-style="western" hwp:sortable="Hildebrandt F."><surname>Hildebrandt</surname>, <given-names>F.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-19">Spatial transcriptomics to define transcriptional patterns of zonation and structural components in the mouse liver</article-title>. <source hwp:id="source-18">Nature communications</source> <volume>12</volume>, <fpage>7046</fpage> (<year>2021</year>).</citation></ref><ref id="c19" hwp:id="ref-19" hwp:rev-id="xref-ref-19-1 xref-ref-19-2"><label>[19]</label><citation publication-type="other" citation-type="journal" ref:id="2024.05.08.593025v1.19" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-19"><string-name name-style="western" hwp:sortable="Chitra U."><surname>Chitra</surname>, <given-names>U.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-20">Mapping the topography of spatial gene expression with interpretable deep learning</article-title>. <source hwp:id="source-19">bioRxiv</source> (<year>2023</year>).</citation></ref><ref id="c20" hwp:id="ref-20" hwp:rev-id="xref-ref-20-1"><label>[20]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.20" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-20"><string-name name-style="western" hwp:sortable="La Manno G."><surname>La Manno</surname>, <given-names>G.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-21">Rna velocity of single cells</article-title>. <source hwp:id="source-20">Nature</source> <volume>560</volume>, <fpage>494</fpage>–<lpage>498</lpage> (<year>2018</year>).</citation></ref><ref id="c21" hwp:id="ref-21"><label>[21]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.21" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-21"><string-name name-style="western" hwp:sortable="Ren H."><surname>Ren</surname>, <given-names>H.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Walker B. L."><surname>Walker</surname>, <given-names>B. L.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Cang Z."><surname>Cang</surname>, <given-names>Z.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Nie Q."><surname>Nie</surname>, <given-names>Q.</given-names></string-name> <article-title hwp:id="article-title-22">Identifying multicellular spatiotemporal organization of cells with spaceflow</article-title>. <source hwp:id="source-21">Nature communications</source> <volume>13</volume>, <fpage>4076</fpage> (<year>2022</year>).</citation></ref><ref id="c22" hwp:id="ref-22"><label>[22]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.22" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-22"><string-name name-style="western" hwp:sortable="Cang Z."><surname>Cang</surname>, <given-names>Z.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-23">Screening cell–cell communication in spatial transcriptomics via collective optimal transport</article-title>. <source hwp:id="source-22">Nature Methods</source> <volume>20</volume>, <fpage>218</fpage>–<lpage>228</lpage> (<year>2023</year>).</citation></ref><ref id="c23" hwp:id="ref-23" hwp:rev-id="xref-ref-23-1"><label>[23]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.23" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-23"><string-name name-style="western" hwp:sortable="Hastie T."><surname>Hastie</surname>, <given-names>T.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Stuetzle W."><surname>Stuetzle</surname>, <given-names>W.</given-names></string-name> <article-title hwp:id="article-title-24">Principal curves</article-title>. <source hwp:id="source-23">Journal of the American Statistical Association</source> <volume>84</volume>, <fpage>502</fpage>–<lpage>516</lpage> (<year>1989</year>).</citation></ref><ref id="c24" hwp:id="ref-24" hwp:rev-id="xref-ref-24-1"><label>[24]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.24" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-24"><string-name name-style="western" hwp:sortable="Banerjee S."><surname>Banerjee</surname>, <given-names>S.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Gelfand A. E."><surname>Gelfand</surname>, <given-names>A. E.</given-names></string-name> <article-title hwp:id="article-title-25">Bayesian wombling: Curvilinear gradient assessment under spatial process models</article-title>. <source hwp:id="source-24">Journal of the American Statistical Association</source> <volume>101</volume>, <fpage>1487</fpage>–<lpage>1501</lpage> (<year>2006</year>).</citation></ref><ref id="c25" hwp:id="ref-25" hwp:rev-id="xref-ref-25-1 xref-ref-25-2"><label>[25]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.25" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-25"><string-name name-style="western" hwp:sortable="Banerjee S."><surname>Banerjee</surname>, <given-names>S.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Gelfand A. E."><surname>Gelfand</surname>, <given-names>A. E.</given-names></string-name> <article-title hwp:id="article-title-26">Bayesian wombling: Curvilinear gradient assessment under spatial process models</article-title>. <source hwp:id="source-25">Journal of the American Statistical Association</source> <volume>101</volume>, <fpage>1487</fpage>–<lpage>1501</lpage> (<year>2006</year>).</citation></ref><ref id="c26" hwp:id="ref-26" hwp:rev-id="xref-ref-26-1 xref-ref-26-2"><label>[26]</label><citation publication-type="other" citation-type="journal" ref:id="2024.05.08.593025v1.26" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-26"><string-name name-style="western" hwp:sortable="Halder A."><surname>Halder</surname>, <given-names>A.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Banerjee S."><surname>Banerjee</surname>, <given-names>S.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Dey D. K."><surname>Dey</surname>, <given-names>D. K.</given-names></string-name> <article-title hwp:id="article-title-27">Bayesian modeling with spatial curvature processes</article-title>. <source hwp:id="source-26">Journal of the American Statistical Association</source> <fpage>1</fpage>–<lpage>13</lpage> (<year>2023</year>).</citation></ref><ref id="c27" hwp:id="ref-27"><label>[27]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.27" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-27"><string-name name-style="western" hwp:sortable="Banerjee S."><surname>Banerjee</surname>, <given-names>S.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Gelfand A. E."><surname>Gelfand</surname>, <given-names>A. E.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Sirmans C."><surname>Sirmans</surname>, <given-names>C.</given-names></string-name> <article-title hwp:id="article-title-28">Directional rates of change under spatial process models</article-title>. <source hwp:id="source-27">Journal of the American Statistical Association</source> <volume>98</volume>, <fpage>946</fpage>–<lpage>954</lpage> (<year>2003</year>).</citation></ref><ref id="c28" hwp:id="ref-28" hwp:rev-id="xref-ref-28-1"><label>[28]</label><citation publication-type="other" citation-type="journal" ref:id="2024.05.08.593025v1.28" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-28"><string-name name-style="western" hwp:sortable="Liu Z."><surname>Liu</surname>, <given-names>Z.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Li M."><surname>Li</surname>, <given-names>M.</given-names></string-name> <source hwp:id="source-28">Optimal plug-in gaussian processes for modelling derivatives</source> (<year>2023</year>). 2210.11626.</citation></ref><ref id="c29" hwp:id="ref-29" hwp:rev-id="xref-ref-29-1 xref-ref-29-2 xref-ref-29-3 xref-ref-29-4"><label>[29]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.29" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-29"><string-name name-style="western" hwp:sortable="Datta A."><surname>Datta</surname>, <given-names>A.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Banerjee S."><surname>Banerjee</surname>, <given-names>S.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Finley A. O."><surname>Finley</surname>, <given-names>A. O.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Gelfand A. E."><surname>Gelfand</surname>, <given-names>A. E.</given-names></string-name> <article-title hwp:id="article-title-29">Hierarchical nearest-neighbor gaussian process models for large geostatistical datasets</article-title>. <source hwp:id="source-29">Journal of the American Statistical Association</source> <volume>111</volume>, <fpage>800</fpage>–<lpage>812</lpage> (<year>2016</year>).</citation></ref><ref id="c30" hwp:id="ref-30" hwp:rev-id="xref-ref-30-1 xref-ref-30-2 xref-ref-30-3 xref-ref-30-4 xref-ref-30-5"><label>[30]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.30" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-30"><string-name name-style="western" hwp:sortable="Finley A. O."><surname>Finley</surname>, <given-names>A. O.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-30">Efficient algorithms for bayesian nearest neighbor gaussian processes</article-title>. <source hwp:id="source-30">Journal of Computational and Graphical Statistics</source> <volume>28</volume>, <fpage>401</fpage>–<lpage>414</lpage> (<year>2019</year>).</citation></ref><ref id="c31" hwp:id="ref-31" hwp:rev-id="xref-ref-31-1"><label>[31]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.31" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-31"><string-name name-style="western" hwp:sortable="Wang F."><surname>Wang</surname>, <given-names>F.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Bhattacharya A."><surname>Bhattacharya</surname>, <given-names>A.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Gelfand A. E."><surname>Gelfand</surname>, <given-names>A. E.</given-names></string-name> <article-title hwp:id="article-title-31">Process modeling for slope and aspect with application to elevation data maps</article-title>. <source hwp:id="source-31">Test</source> <volume>27</volume>, <fpage>749</fpage>–<lpage>772</lpage> (<year>2018</year>).</citation></ref><ref id="c32" hwp:id="ref-32" hwp:rev-id="xref-ref-32-1 xref-ref-32-2"><label>[32]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.32" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-32"><string-name name-style="western" hwp:sortable="Andersson A."><surname>Andersson</surname>, <given-names>A.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-32">Spatial deconvolution of her2-positive breast cancer delineates tumor-associated cell type interactions</article-title>. <source hwp:id="source-32">Nature communications</source> <volume>12</volume>, <fpage>6012</fpage> (<year>2021</year>).</citation></ref><ref id="c33" hwp:id="ref-33" hwp:rev-id="xref-ref-33-1"><label>[33]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.33" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-33"><string-name name-style="western" hwp:sortable="Cable D. M."><surname>Cable</surname>, <given-names>D. M.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-33">Robust decomposition of cell type mixtures in spatial transcriptomics</article-title>. <source hwp:id="source-33">Nature biotechnology</source> <volume>40</volume>, <fpage>517</fpage>–<lpage>526</lpage> (<year>2022</year>).</citation></ref><ref id="c34" hwp:id="ref-34" hwp:rev-id="xref-ref-34-1"><label>[34]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.34" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-34"><string-name name-style="western" hwp:sortable="Andersson A."><surname>Andersson</surname>, <given-names>A.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-34">Single-cell and spatial transcriptomics enables probabilistic inference of cell type topography</article-title>. <source hwp:id="source-34">Communications biology</source> <volume>3</volume>, <fpage>565</fpage> (<year>2020</year>).</citation></ref><ref id="c35" hwp:id="ref-35" hwp:rev-id="xref-ref-35-1"><label>[35]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.35" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-35"><string-name name-style="western" hwp:sortable="Chen J."><surname>Chen</surname>, <given-names>J.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-35">Cell composition inference and identification of layer-specific spatial transcriptional profiles with polaris</article-title>. <source hwp:id="source-35">Science Advances</source> <volume>9</volume>, <fpage>eadd9818</fpage> (<year>2023</year>).</citation></ref><ref id="c36" hwp:id="ref-36" hwp:rev-id="xref-ref-36-1 xref-ref-36-2"><label>[36]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.36" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-36"><string-name name-style="western" hwp:sortable="Kennedy-Darling J."><surname>Kennedy-Darling</surname>, <given-names>J.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-36">Highly multiplexed tissue imaging using repeated oligonucleotide exchange reaction</article-title>. <source hwp:id="source-36">European Journal of Immunology</source> <volume>51</volume>, <fpage>1262</fpage>–<lpage>1277</lpage> (<year>2021</year>).</citation></ref><ref id="c37" hwp:id="ref-37" hwp:rev-id="xref-ref-37-1 xref-ref-37-2"><label>[37]</label><citation publication-type="other" citation-type="journal" ref:id="2024.05.08.593025v1.37" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-37"><string-name name-style="western" hwp:sortable="Chen S."><surname>Chen</surname>, <given-names>S.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-37">Integration of spatial and single-cell data across modalities with weakly linked features</article-title>. <source hwp:id="source-37">Nature Biotechnology</source> <fpage>1</fpage>–<lpage>11</lpage> (<year>2023</year>).</citation></ref><ref id="c38" hwp:id="ref-38" hwp:rev-id="xref-ref-38-1"><label>[38]</label><citation publication-type="other" citation-type="journal" ref:id="2024.05.08.593025v1.38" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-38"><string-name name-style="western" hwp:sortable="Caiado H."><surname>Caiado</surname>, <given-names>H.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Cancela M. L."><surname>Cancela</surname>, <given-names>M. L.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Conceição N."><surname>Conceição</surname>, <given-names>N.</given-names></string-name> <article-title hwp:id="article-title-38">Assessment of mgp gene expression in cancer and contribution to prognosis</article-title>. <source hwp:id="source-38">Biochimie</source> (<year>2023</year>).</citation></ref><ref id="c39" hwp:id="ref-39" hwp:rev-id="xref-ref-39-1"><label>[39]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.39" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-39"><string-name name-style="western" hwp:sortable="Jing X."><surname>Jing</surname>, <given-names>X.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-39">Cd24 is a potential biomarker for prognosis in human breast carcinoma</article-title>. <source hwp:id="source-39">Cellular Physiology and Biochemistry</source> <volume>48</volume>, <fpage>111</fpage>–<lpage>119</lpage> (<year>2018</year>).</citation></ref><ref id="c40" hwp:id="ref-40" hwp:rev-id="xref-ref-40-1"><label>[40]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.40" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-40"><string-name name-style="western" hwp:sortable="Zhu J."><surname>Zhu</surname>, <given-names>J.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Sun S."><surname>Sun</surname>, <given-names>S.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Zhou X."><surname>Zhou</surname>, <given-names>X.</given-names></string-name> <article-title hwp:id="article-title-40">Spark-x: non-parametric modeling enables scalable and robust detection of spatial expression patterns for large spatial transcriptomic studies</article-title>. <source hwp:id="source-40">Genome biology</source> <volume>22</volume>, <fpage>1</fpage>–<lpage>25</lpage> (<year>2021</year>).</citation></ref><ref id="c41" hwp:id="ref-41" hwp:rev-id="xref-ref-41-1"><label>[41]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.41" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-41"><string-name name-style="western" hwp:sortable="Hao M."><surname>Hao</surname>, <given-names>M.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Hua K."><surname>Hua</surname>, <given-names>K.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Zhang X."><surname>Zhang</surname>, <given-names>X.</given-names></string-name> <article-title hwp:id="article-title-41">Somde: a scalable method for identifying spatially variable genes with self-organizing map</article-title>. <source hwp:id="source-41">Bioinformatics</source> <volume>37</volume>, <fpage>4392</fpage>–<lpage>4398</lpage> (<year>2021</year>).</citation></ref><ref id="c42" hwp:id="ref-42" hwp:rev-id="xref-ref-42-1"><label>[42]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.42" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-42"><string-name name-style="western" hwp:sortable="Weber L. M."><surname>Weber</surname>, <given-names>L. M.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Saha A."><surname>Saha</surname>, <given-names>A.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Datta A."><surname>Datta</surname>, <given-names>A.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Hansen K. D."><surname>Hansen</surname>, <given-names>K. D.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Hicks S. C."><surname>Hicks</surname>, <given-names>S. C.</given-names></string-name> <article-title hwp:id="article-title-42">nnsvg for the scalable identification of spatially variable genes using nearest-neighbor gaussian processes</article-title>. <source hwp:id="source-42">Nature communications</source> <volume>14</volume>, <fpage>4059</fpage> (<year>2023</year>).</citation></ref><ref id="c43" hwp:id="ref-43" hwp:rev-id="xref-ref-43-1"><label>[43]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.43" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-43"><string-name name-style="western" hwp:sortable="Svensson V."><surname>Svensson</surname>, <given-names>V.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Teichmann S. A."><surname>Teichmann</surname>, <given-names>S. A.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Stegle O."><surname>Stegle</surname>, <given-names>O.</given-names></string-name> <article-title hwp:id="article-title-43">Spatialde: identification of spatially variable genes</article-title>. <source hwp:id="source-43">Nature methods</source> <volume>15</volume>, <fpage>343</fpage>–<lpage>346</lpage> (<year>2018</year>).</citation></ref><ref id="c44" hwp:id="ref-44" hwp:rev-id="xref-ref-44-1"><label>[44]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.44" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-44"><string-name name-style="western" hwp:sortable="Li Y."><surname>Li</surname>, <given-names>Y.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-44">Immunoglobulin superfamily genes are novel prognostic biomarkers for breast cancer</article-title>. <source hwp:id="source-44">Oncotarget</source> <volume>8</volume>, <fpage>2444</fpage> (<year>2017</year>).</citation></ref><ref id="c45" hwp:id="ref-45" hwp:rev-id="xref-ref-45-1"><label>[45]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.45" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-45"><string-name name-style="western" hwp:sortable="Madan S."><surname>Madan</surname>, <given-names>S.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-45">Pan-cancer analysis of patient tumor single-cell transcriptomes identifies promising selective and safe chimeric antigen receptor targets in head and neck cancer</article-title>. <source hwp:id="source-45">Cancers</source> <volume>15</volume>, <fpage>4885</fpage> (<year>2023</year>).</citation></ref><ref id="c46" hwp:id="ref-46" hwp:rev-id="xref-ref-46-1"><label>[46]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.46" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-46"><string-name name-style="western" hwp:sortable="Verschueren E."><surname>Verschueren</surname>, <given-names>E.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-46">The immunoglobulin superfamily receptome defines cancer-relevant networks associated with clinical outcome</article-title>. <source hwp:id="source-46">Cell</source> <volume>182</volume>, <fpage>329</fpage>–<lpage>344</lpage> (<year>2020</year>).</citation></ref><ref id="c47" hwp:id="ref-47" hwp:rev-id="xref-ref-47-1"><label>[47]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.47" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-47"><string-name name-style="western" hwp:sortable="Wu T."><surname>Wu</surname>, <given-names>T.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-47">clusterprofiler 4.0: A universal enrichment tool for interpreting omics data</article-title>. <source hwp:id="source-47">The innovation</source> <volume>2</volume> (<year>2021</year>).</citation></ref><ref id="c48" hwp:id="ref-48" hwp:rev-id="xref-ref-48-1"><label>[48]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.48" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-48"><string-name name-style="western" hwp:sortable="Perrier S."><surname>Perrier</surname>, <given-names>S.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Caldefie-Chezet F."><surname>Caldefie-Chezet</surname>, <given-names>F.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Vasson M.-P."><surname>Vasson</surname>, <given-names>M.-P.</given-names></string-name> <article-title hwp:id="article-title-48">Il-1 family in breast cancer: potential interplay with leptin and other adipocytokines</article-title>. <source hwp:id="source-48">FEBS letters</source> <volume>583</volume>, <fpage>259</fpage>–<lpage>265</lpage> (<year>2009</year>).</citation></ref><ref id="c49" hwp:id="ref-49" hwp:rev-id="xref-ref-49-1 xref-ref-49-2"><label>[49]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.49" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-49"><string-name name-style="western" hwp:sortable="Fix E."><surname>Fix</surname>, <given-names>E.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Hodges J. L."><surname>Hodges</surname>, <given-names>J. L.</given-names></string-name> <article-title hwp:id="article-title-49">Discriminatory analysis. nonparametric discrimination: Consistency properties</article-title>. <source hwp:id="source-49">International Statistical Review/Revue Internationale de Statistique</source> <volume>57</volume>, <fpage>238</fpage>–<lpage>247</lpage> (<year>1989</year>).</citation></ref><ref id="c50" hwp:id="ref-50" hwp:rev-id="xref-ref-50-1 xref-ref-50-2"><label>[50]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.50" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-50"><string-name name-style="western" hwp:sortable="Cover T."><surname>Cover</surname>, <given-names>T.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Hart P."><surname>Hart</surname>, <given-names>P.</given-names></string-name> <article-title hwp:id="article-title-50">Nearest neighbor pattern classification</article-title>. <source hwp:id="source-50">IEEE transactions on information theory</source> <volume>13</volume>, <fpage>21</fpage>–<lpage>27</lpage> (<year>1967</year>).</citation></ref><ref id="c51" hwp:id="ref-51" hwp:rev-id="xref-ref-51-1"><label>[51]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.51" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-51"><string-name name-style="western" hwp:sortable="Rand W. M."><surname>Rand</surname>, <given-names>W. M.</given-names></string-name> <article-title hwp:id="article-title-51">Objective criteria for the evaluation of clustering methods</article-title>. <source hwp:id="source-51">Journal of the American Statistical association</source> <volume>66</volume>, <fpage>846</fpage>–<lpage>850</lpage> (<year>1971</year>).</citation></ref><ref id="c52" hwp:id="ref-52" hwp:rev-id="xref-ref-52-1"><label>[52]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.52" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-52"><string-name name-style="western" hwp:sortable="Hao Y."><surname>Hao</surname>, <given-names>Y.</given-names></string-name> <etal>et al.</etal> <article-title hwp:id="article-title-52">Integrated analysis of multimodal single-cell data</article-title>. <source hwp:id="source-52">Cell</source> <volume>184</volume>, <fpage>3573</fpage>–<lpage>3587</lpage> (<year>2021</year>).</citation></ref><ref id="c53" hwp:id="ref-53" hwp:rev-id="xref-ref-53-1"><label>[53]</label><citation publication-type="book" citation-type="book" ref:id="2024.05.08.593025v1.53" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-53"><string-name name-style="western" hwp:sortable="Cressie N."><surname>Cressie</surname>, <given-names>N.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Wikle C."><surname>Wikle</surname>, <given-names>C.</given-names></string-name> <source hwp:id="source-53">Statistics for Spatio-Temporal Data CourseSmart Series</source> (<publisher-name>Wiley</publisher-name>, <year>2011</year>). URL <ext-link l:rel="related" l:ref-type="uri" l:ref="https://books.google.com/books?id=-kOC6D0DiNYC" ext-link-type="uri" xlink:href="https://books.google.com/books?id=-kOC6D0DiNYC" hwp:id="ext-link-6">https://books.google.com/books?id=-kOC6D0DiNYC</ext-link>.</citation></ref><ref id="c54" hwp:id="ref-54" hwp:rev-id="xref-ref-54-1"><label>[54]</label><citation publication-type="book" citation-type="book" ref:id="2024.05.08.593025v1.54" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-54"><string-name name-style="western" hwp:sortable="Banerjee S."><surname>Banerjee</surname>, <given-names>S.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Carlin B."><surname>Carlin</surname>, <given-names>B.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Gelfand A."><surname>Gelfand</surname>, <given-names>A.</given-names></string-name> <source hwp:id="source-54">Hierarchical Modeling and Analysis for Spatial Data Chapman &amp; Hall/CRC Monographs on Statistics &amp; Applied Probability</source> (<publisher-name>CRC Press</publisher-name>, <year>2014</year>). URL <ext-link l:rel="related" l:ref-type="uri" l:ref="https://books.google.td/books?id=WVHRBQAAQBAJ" ext-link-type="uri" xlink:href="https://books.google.td/books?id=WVHRBQAAQBAJ" hwp:id="ext-link-7">https://books.google.td/books?id=WVHRBQAAQBAJ</ext-link>.</citation></ref><ref id="c55" hwp:id="ref-55" hwp:rev-id="xref-ref-55-1 xref-ref-55-2"><label>[55]</label><citation publication-type="book" citation-type="book" ref:id="2024.05.08.593025v1.55" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-55"><string-name name-style="western" hwp:sortable="Stein M. L."><surname>Stein</surname>, <given-names>M. L.</given-names></string-name> <source hwp:id="source-55">Interpolation of spatial data: some theory for kriging</source> (<publisher-name>Springer Science &amp; Business Media</publisher-name>, <year>1999</year>).</citation></ref><ref id="c56" hwp:id="ref-56" hwp:rev-id="xref-ref-56-1"><label>[56]</label><citation publication-type="book" citation-type="book" ref:id="2024.05.08.593025v1.56" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-56"><string-name name-style="western" hwp:sortable="Wilmott P."><surname>Wilmott</surname>, <given-names>P.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Howison S."><surname>Howison</surname>, <given-names>S.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Dewynne J."><surname>Dewynne</surname>, <given-names>J.</given-names></string-name> <source hwp:id="source-56">The mathematics of financial derivatives: a student introduction</source> (<publisher-name>Cambridge university press</publisher-name>, <year>1995</year>).</citation></ref><ref id="c57" hwp:id="ref-57" hwp:rev-id="xref-ref-57-1"><label>[57]</label><citation publication-type="book" citation-type="book" ref:id="2024.05.08.593025v1.57" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-57"><string-name name-style="western" hwp:sortable="Atkinson K."><surname>Atkinson</surname>, <given-names>K.</given-names></string-name> <source hwp:id="source-57">An introduction to numerical analysis</source> (<publisher-name>John wiley &amp; sons</publisher-name>, <year>1991</year>).</citation></ref><ref id="c58" hwp:id="ref-58" hwp:rev-id="xref-ref-58-1"><label>[58]</label><citation publication-type="book" citation-type="book" ref:id="2024.05.08.593025v1.58" ref:linkable="no" ref:use-reference-as-is="yes" hwp:id="citation-58"><string-name name-style="western" hwp:sortable="Burden R."><surname>Burden</surname>, <given-names>R.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Faires J."><surname>Faires</surname>, <given-names>J.</given-names></string-name> <source hwp:id="source-58">Numerical Analysis Fourth Edition</source> (<publisher-name>PWS KENT Publishing Company Boston</publisher-name>, <year>1989</year>).</citation></ref><ref id="c59" hwp:id="ref-59" hwp:rev-id="xref-ref-59-1"><label>[59]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.59" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-59"><string-name name-style="western" hwp:sortable="Ester M."><surname>Ester</surname>, <given-names>M.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Kriegel H.-P."><surname>Kriegel</surname>, <given-names>H.-P.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Sander J."><surname>Sander</surname>, <given-names>J.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Xu X."><surname>Xu</surname>, <given-names>X.</given-names></string-name> <etal>et al.</etal> <source hwp:id="source-59">A density-based algorithm for discovering clusters in large spatial databases with noise</source>, Vol. <volume>96</volume>, <fpage>226</fpage>–<lpage>231</lpage> (<year>1996</year>).</citation></ref><ref id="c60" hwp:id="ref-60" hwp:rev-id="xref-ref-60-1 xref-ref-60-2"><label>[60]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.60" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-60"><string-name name-style="western" hwp:sortable="Lloyd S."><surname>Lloyd</surname>, <given-names>S.</given-names></string-name> <article-title hwp:id="article-title-53">Least squares quantization in pcm</article-title>. <source hwp:id="source-60">IEEE transactions on information theory</source> <volume>28</volume>, <fpage>129</fpage>–<lpage>137</lpage> (<year>1982</year>).</citation></ref><ref id="c61" hwp:id="ref-61" hwp:rev-id="xref-ref-61-1"><label>[61]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.61" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-61"><string-name name-style="western" hwp:sortable="Blondel V. D."><surname>Blondel</surname>, <given-names>V. D.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Guillaume J.-L."><surname>Guillaume</surname>, <given-names>J.-L.</given-names></string-name>, <string-name name-style="western" hwp:sortable="Lambiotte R."><surname>Lambiotte</surname>, <given-names>R.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Lefebvre E."><surname>Lefebvre</surname>, <given-names>E.</given-names></string-name> <article-title hwp:id="article-title-54">Fast unfolding of communities in large networks</article-title>. <source hwp:id="source-61">Journal of statistical mechanics: theory and experiment</source> <volume>2008</volume>, <fpage>P10008</fpage> (<year>2008</year>).</citation></ref><ref id="c62" hwp:id="ref-62" hwp:rev-id="xref-ref-62-1"><label>[62]</label><citation publication-type="journal" citation-type="journal" ref:id="2024.05.08.593025v1.62" ref:linkable="yes" ref:use-reference-as-is="yes" hwp:id="citation-62"><string-name name-style="western" hwp:sortable="Shang L."><surname>Shang</surname>, <given-names>L.</given-names></string-name> &amp; <string-name name-style="western" hwp:sortable="Zhou X."><surname>Zhou</surname>, <given-names>X.</given-names></string-name> <article-title hwp:id="article-title-55">Spatially aware dimension reduction for spatial transcriptomics</article-title>. <source hwp:id="source-62">Nature Communications</source> <volume>13</volume>, <fpage>7203</fpage> (<year>2022</year>).</citation></ref></ref-list><sec id="s7" hwp:id="sec-24"><label>5</label><title hwp:id="title-29">Extended Data</title><fig id="figED1" position="float" fig-type="figure" orientation="portrait" hwp:id="F5" hwp:rev-id="xref-fig-5-1 xref-fig-5-2 xref-fig-5-3 xref-fig-5-4 xref-fig-5-5 xref-fig-5-6"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED1</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F5</object-id><object-id pub-id-type="publisher-id">figED1</object-id><label>Extended Data Fig. 1.</label><caption hwp:id="caption-5"><p hwp:id="p-62">Simulation study. Measurements are simulated from a mixture of sine and cosine functions, streak, hot spot and regional patterns.</p></caption><graphic xlink:href="593025v1_figED1" position="float" orientation="portrait" hwp:id="graphic-17"/></fig><fig id="figED2" position="float" fig-type="figure" orientation="portrait" hwp:id="F6" hwp:rev-id="xref-fig-6-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED2</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F6</object-id><object-id pub-id-type="publisher-id">figED2</object-id><label>Extended Data Fig. 2.</label><caption hwp:id="caption-6"><p hwp:id="p-63">Estimated gradients for gene with top 10 median <italic toggle="yes">L</italic><sup>2</sup> values in DLPFC. For each gene, the left panel is the normalized gene expression, the middle panel is the smoothed and normalized gene expression, and the right panel is the estimated <italic toggle="yes">L</italic><sup>2</sup>. Spots with <italic toggle="yes">L</italic><sup>2</sup> smaller than 0.7 quantile are masked by gray color.</p></caption><graphic xlink:href="593025v1_figED2" position="float" orientation="portrait" hwp:id="graphic-18"/></fig><fig id="figED3" position="float" fig-type="figure" orientation="portrait" hwp:id="F7" hwp:rev-id="xref-fig-7-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED3</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F7</object-id><object-id pub-id-type="publisher-id">figED3</object-id><label>Extended Data Fig. 3.</label><caption hwp:id="caption-7"><p hwp:id="p-64">Estimated gradients for cell type proportion in HER2+ breast cancer data. For each gene, the left panel is the cell type proportion, the middle panel is the smoothed cell type proportion, and the right panel is the estimated <italic toggle="yes">L</italic><sup>2</sup>. Spots with <italic toggle="yes">L</italic><sup>2</sup> smaller than 0.7 quantile are masked by gray color.</p></caption><graphic xlink:href="593025v1_figED3" position="float" orientation="portrait" hwp:id="graphic-19"/></fig><fig id="figED4" position="float" fig-type="figure" orientation="portrait" hwp:id="F8" hwp:rev-id="xref-fig-8-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED4</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F8</object-id><object-id pub-id-type="publisher-id">figED4</object-id><label>Extended Data Fig. 4.</label><caption hwp:id="caption-8"><p hwp:id="p-65">Estimated gradients for selected proteins in CODEX data. For each protein, the left panel is the normalized abundance, the middle panel is the smoothed and normalized abundance, and the right panel is the estimated <italic toggle="yes">L</italic><sup>2</sup>. Spots with <italic toggle="yes">L</italic><sup>2</sup> smaller than 0.7 quantile are masked by gray color.</p></caption><graphic xlink:href="593025v1_figED4" position="float" orientation="portrait" hwp:id="graphic-20"/></fig><fig id="figED5" position="float" fig-type="figure" orientation="portrait" hwp:id="F9" hwp:rev-id="xref-fig-9-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED5</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F9</object-id><object-id pub-id-type="publisher-id">figED5</object-id><label>Extended Data Fig. 5.</label><caption hwp:id="caption-9"><p hwp:id="p-66">Cliff genes with top five positive gradient for each boundary in DLPFC.</p></caption><graphic xlink:href="593025v1_figED5" position="float" orientation="portrait" hwp:id="graphic-21"/></fig><fig id="figED6" position="float" fig-type="figure" orientation="portrait" hwp:id="F10" hwp:rev-id="xref-fig-10-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED6</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F10</object-id><object-id pub-id-type="publisher-id">figED6</object-id><label>Extended Data Fig. 6.</label><caption hwp:id="caption-10"><p hwp:id="p-67">Cliff genes with top five negative gradient for each boundary in DLPFC.</p></caption><graphic xlink:href="593025v1_figED6" position="float" orientation="portrait" hwp:id="graphic-22"/></fig><fig id="figED7" position="float" fig-type="figure" orientation="portrait" hwp:id="F11" hwp:rev-id="xref-fig-11-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED7</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F11</object-id><object-id pub-id-type="publisher-id">figED7</object-id><label>Extended Data Fig. 7.</label><caption hwp:id="caption-11"><p hwp:id="p-68">Comparison between mean expression change slope (spots in distance 0 layer to -0.05 layer) and estimated boundary gradients.</p></caption><graphic xlink:href="593025v1_figED7" position="float" orientation="portrait" hwp:id="graphic-23"/></fig><fig id="figED8" position="float" fig-type="figure" orientation="portrait" hwp:id="F12" hwp:rev-id="xref-fig-12-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED8</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F12</object-id><object-id pub-id-type="publisher-id">figED8</object-id><label>Extended Data Fig. 8.</label><caption hwp:id="caption-12"><p hwp:id="p-69">Cliff genes with top five positive gradient for each boundary pair in DLPFC.</p></caption><graphic xlink:href="593025v1_figED8" position="float" orientation="portrait" hwp:id="graphic-24"/></fig><fig id="figED9" position="float" fig-type="figure" orientation="portrait" hwp:id="F13" hwp:rev-id="xref-fig-13-1 xref-fig-13-2"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED9</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F13</object-id><object-id pub-id-type="publisher-id">figED9</object-id><label>Extended Data Fig. 9.</label><caption hwp:id="caption-13"><p hwp:id="p-70">Cliff genes with top five gradient for multi-boundaries in HER2+ breast cancer. (Row 1) top five genes with positive gradients for hand-drawn cancerous boundary based on pathologist annotation. (row 2) top five genes with negative gradients for hand-drawn cancerous boundary based on pathologist annotation. (Row 3) top five genes with negative gradients for cancel epithelial cell boundaries. (Row 4) top five genes with positive gradients for cancel epithelial cell boundaries.</p></caption><graphic xlink:href="593025v1_figED9" position="float" orientation="portrait" hwp:id="graphic-25"/></fig><fig id="figED10" position="float" fig-type="figure" orientation="portrait" hwp:id="F14" hwp:rev-id="xref-fig-14-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGED10</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F14</object-id><object-id pub-id-type="publisher-id">figED10</object-id><label>Extended Data Fig. 10.</label><caption hwp:id="caption-14"><p hwp:id="p-71">Refined clustering results for all methods with different input (gene expression, <italic toggle="yes">L</italic><sup>2</sup>, Gene+<italic toggle="yes">L</italic><sup>2</sup>) in different gene sets (HVG, cliff genes).</p></caption><graphic xlink:href="593025v1_figED10" position="float" orientation="portrait" hwp:id="graphic-26"/></fig></sec><sec id="s8" hwp:id="sec-25"><title hwp:id="title-30">Supplementary Note for “Investigating spatial dynamics in spatial omics data with StarTrail”</title><sec id="s8a" hwp:id="sec-26"><label>1</label><title hwp:id="title-31">Analysis of non-cliff SVGs</title><p hwp:id="p-72">StarTrail, used together with SVG methods, can reveal novel biological insights. For the DLPFC dataset, upon examining significant SVGs with minimal gradient values, it becomes apparent that many genes exhibit expression changes across multiple boundaries (<xref rid="figS9" ref-type="fig" hwp:id="xref-fig-23-2" hwp:rel-id="F23">Supplementary Fig. 9</xref>). Notably, genes such as <italic toggle="yes">HMGCR</italic> and <italic toggle="yes">BOLA3</italic> manifest high expression in the bottom right of the slide, indicative of potentially new subregions not yet recognized in literature. These subregions potentially account for the distinct cluster observed in the bottom right region in various clustering analysis results (as detailed in Supplementary Fig. 12 of the spatialPCA paper [<xref ref-type="bibr" rid="c62" hwp:id="xref-ref-62-1" hwp:rel-id="ref-62">62</xref>]).</p><p hwp:id="p-73">In the analysis of HER2+ breast cancer, we discovered that certain SVGs upregulated in adipose tissue and breast glands exhibit minimal gradient values at the cancer boundaries (<xref rid="figS10" ref-type="fig" hwp:id="xref-fig-24-2" hwp:rel-id="F24">Supplementary Fig. 10</xref>). These observations again suggest that utilizing SVG method alongside StarTrail has the potential to unveil genes linked to previously unidentified domains. It is important to note again that StarTrail is not a SVG method. Complementary to SVG methods, StarTrail highlights the critical role of boundary analysis in spatial omics, revealing its potential to connect omics features with distinct boundary-related changes and uncovering novel aspects of tissue structure, disease development or progression.</p></sec><sec id="s8b" hwp:id="sec-27"><label>2</label><title hwp:id="title-32">Matching predicted clusters to true layers</title><p hwp:id="p-74">We match predicted clusters and true layers by relabeling each predicted cluster to allow easier performance comparison across methods and to enhance result visualization (this method is not involved in the calculation of ARI). When the number of predicted clusters is greater than or equal to the number of true layers, our matching ensures that each true layer has a corresponding set of predicted clusters. In cases where the number of predicted clusters exceeds the number of true layers, some predicted clusters may be merged and assigned to match the same true layer. When the number of predicted clusters matches the number of true layers, a one-to-one correspondence is established.</p><p hwp:id="p-75">The matching process starts by iterating through each predicted cluster. Step1: We assign the predicted cluster to a new label based on its best-match (i.e., the highest proportion of overlap) true layer. Step2: As multiple predicted clusters can have the same best-match true layer, we also evaluate the proportion of each predicted cluster in the true layer and select the best-match predicted cluster. For instance, if true layer A is the best-match for both predicted cluster a and b in the first step and true layer B is not assigned to any predicted cluster in the first step, we will examine the proportion of predicted cluster a and b in true layer A. For example, the proportions of predicted clusters a and b in true layer A are 90% and 10%, then true layer A will be assigned to predicted cluster a and true layer B will be assigned to predicted cluster b. Step2 is repeated until all the true layers are assigned to at least one predicted cluster. In order to reduce computation time, we add a threshold <italic toggle="yes">c</italic> for Step2: we only consider predicted cluster when its number of spots is greater than <italic toggle="yes">c</italic>. In our analysis, we used <italic toggle="yes">c</italic> = 10.</p><fig id="figS1" position="float" fig-type="figure" orientation="portrait" hwp:id="F15"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS1</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F15</object-id><object-id pub-id-type="publisher-id">figS1</object-id><label>Supplementary Figs. 1.</label><caption hwp:id="caption-15"><p hwp:id="p-76">The distance between spots and DLPFC boundary.</p></caption><graphic xlink:href="593025v1_figS1" position="float" orientation="portrait" hwp:id="graphic-27"/></fig><fig id="figS2" position="float" fig-type="figure" orientation="portrait" hwp:id="F16" hwp:rev-id="xref-fig-16-1 xref-fig-16-2"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS2</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F16</object-id><object-id pub-id-type="publisher-id">figS2</object-id><label>Supplementary Figs. 2.</label><caption hwp:id="caption-16"><p hwp:id="p-77">Comparison of mean expression change across each boundary with the gradients in DLPFC analysis. The lines are colored by estimated boundary gradients.</p></caption><graphic xlink:href="593025v1_figS2" position="float" orientation="portrait" hwp:id="graphic-28"/></fig><fig id="figS3" position="float" fig-type="figure" orientation="portrait" hwp:id="F17" hwp:rev-id="xref-fig-17-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS3</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F17</object-id><object-id pub-id-type="publisher-id">figS3</object-id><label>Supplementary Figs. 3.</label><caption hwp:id="caption-17"><p hwp:id="p-78">Upset plot for cliff genes and SVGs discovery in DLPFC 151676 analysis. Cliff genes were detected along each of the six boundaries indicated in <xref rid="fig2" ref-type="fig" hwp:id="xref-fig-2-10" hwp:rel-id="F2">Fig. 2e</xref>.</p></caption><graphic xlink:href="593025v1_figS3" position="float" orientation="portrait" hwp:id="graphic-29"/></fig><fig id="figS4" position="float" fig-type="figure" orientation="portrait" hwp:id="F18" hwp:rev-id="xref-fig-18-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS4</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F18</object-id><object-id pub-id-type="publisher-id">figS4</object-id><label>Supplementary Figs. 4.</label><caption hwp:id="caption-18"><p hwp:id="p-79">The unique cliff genes detected by StarTrail in DLPFC analysis. For each gene, the left panel is the raw gene expression, the right panel is the smoothed and normalized gene expression.</p></caption><graphic xlink:href="593025v1_figS4" position="float" orientation="portrait" hwp:id="graphic-30"/></fig><fig id="figS5" position="float" fig-type="figure" orientation="portrait" hwp:id="F19" hwp:rev-id="xref-fig-19-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS5</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F19</object-id><object-id pub-id-type="publisher-id">figS5</object-id><label>Supplementary Figs. 5.</label><caption hwp:id="caption-19"><p hwp:id="p-80">The unique cliff genes detected by StarTrail in HER2+ analysis with top 10 largest absolute gradient on any cancer epithelial boundary. For each gene, the left panel is the raw gene expression, the right panel is the smoothed and normalized gene expression.</p></caption><graphic xlink:href="593025v1_figS5" position="float" orientation="portrait" hwp:id="graphic-31"/></fig><fig id="figS6" position="float" fig-type="figure" orientation="portrait" hwp:id="F20" hwp:rev-id="xref-fig-20-1 xref-fig-20-2"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS6</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F20</object-id><object-id pub-id-type="publisher-id">figS6</object-id><label>Supplementary Figs. 6.</label><caption hwp:id="caption-20"><p hwp:id="p-81">The unique cliff genes detected by StarTrail in HER2+ analysis with top 10 largest absolute gradient on the cancer epithelial multi-boundaries. For each gene, the left panel is the raw gene expression, the right panel is the smoothed and normalized gene expression.</p></caption><graphic xlink:href="593025v1_figS6" position="float" orientation="portrait" hwp:id="graphic-32"/></fig><fig id="figS7" position="float" fig-type="figure" orientation="portrait" hwp:id="F21" hwp:rev-id="xref-fig-21-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS7</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F21</object-id><object-id pub-id-type="publisher-id">figS7</object-id><label>Supplementary Figs. 7.</label><caption hwp:id="caption-21"><p hwp:id="p-82">StarTrail detected cliff genes associated with GO:0071347 in HER2+ analysis. For each gene, the left panel shows the raw gene expression, while the right panel shows the smoothed and normalized gene expression.</p></caption><graphic xlink:href="593025v1_figS7" position="float" orientation="portrait" hwp:id="graphic-33"/></fig><fig id="figS8" position="float" fig-type="figure" orientation="portrait" hwp:id="F22" hwp:rev-id="xref-fig-22-1"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS8</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F22</object-id><object-id pub-id-type="publisher-id">figS8</object-id><label>Supplementary Figs. 8.</label><caption hwp:id="caption-22"><p hwp:id="p-83">StarTrail detected cliff genes associated with GO:0002715 in HER2+ analysis. For each gene, the left panel shows the raw gene expression, while the right panel shows the smoothed and normalized gene expression.</p></caption><graphic xlink:href="593025v1_figS8" position="float" orientation="portrait" hwp:id="graphic-34"/></fig><fig id="figS9" position="float" fig-type="figure" orientation="portrait" hwp:id="F23" hwp:rev-id="xref-fig-23-1 xref-fig-23-2"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS9</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F23</object-id><object-id pub-id-type="publisher-id">figS9</object-id><label>Supplementary Figs. 9.</label><caption hwp:id="caption-23"><p hwp:id="p-84">The SVGs genes that are not cliff genes in DLPFC analysis with top 10 smallest absolute gradient on any boundary. For each gene, the left panel is the raw gene expression, the right panel is the smoothed and normalized gene expression.</p></caption><graphic xlink:href="593025v1_figS9" position="float" orientation="portrait" hwp:id="graphic-35"/></fig><fig id="figS10" position="float" fig-type="figure" orientation="portrait" hwp:id="F24" hwp:rev-id="xref-fig-24-1 xref-fig-24-2"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS10</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F24</object-id><object-id pub-id-type="publisher-id">figS10</object-id><label>Supplementary Figs. 10.</label><caption hwp:id="caption-24"><p hwp:id="p-85">The SVGs genes that are not cliff genes in HER2+ analysis with top 10 smallest absolute gradient on any cancer epithelial boundary. For each gene, the left panel is the raw gene expression, the right panel is the smoothed and normalized gene expression.</p></caption><graphic xlink:href="593025v1_figS10" position="float" orientation="portrait" hwp:id="graphic-36"/></fig><fig id="figS11" position="float" fig-type="figure" orientation="portrait" hwp:id="F25" hwp:rev-id="xref-fig-25-1 xref-fig-25-2 xref-fig-25-3"><object-id pub-id-type="other" hwp:sub-type="pisa">biorxiv;2024.05.08.593025v1/FIGS11</object-id><object-id pub-id-type="other" hwp:sub-type="slug">F25</object-id><object-id pub-id-type="publisher-id">figS11</object-id><label>Supplementary Figs. 11.</label><caption hwp:id="caption-25"><p hwp:id="p-86">Raw clustering results for all methods with different input (gene expression, <italic toggle="yes">L</italic><sup>2</sup>, Gene+<italic toggle="yes">L</italic><sup>2</sup>) in different gene sets (HVG, cliff genes).</p></caption><graphic xlink:href="593025v1_figS11" position="float" orientation="portrait" hwp:id="graphic-37"/></fig></sec></sec></back></article>
