<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing with OASIS Tables v3.0 20080202//EN" "journalpub-oasis3.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:oasis="http://docs.oasis-open.org/ns/oasis-exchange/table" xml:lang="en" dtd-version="3.0" article-type="research-article"><?xmltex \bartext{Model description paper}?>
  <front>
    <journal-meta><journal-id journal-id-type="publisher">GMD</journal-id><journal-title-group>
    <journal-title>Geoscientific Model Development</journal-title>
    <abbrev-journal-title abbrev-type="publisher">GMD</abbrev-journal-title><abbrev-journal-title abbrev-type="nlm-ta">Geosci. Model Dev.</abbrev-journal-title>
  </journal-title-group><issn pub-type="epub">1991-9603</issn><publisher>
    <publisher-name>Copernicus Publications</publisher-name>
    <publisher-loc>Göttingen, Germany</publisher-loc>
  </publisher></journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.5194/gmd-15-5739-2022</article-id><title-group><article-title>swNEMO_v4.0: an ocean model based on NEMO4 for the new-generation Sunway supercomputer</article-title><alt-title>swNEMO_v4.0: an ocean model NEMO for the new-generation Sunway supercomputer</alt-title>
      </title-group><?xmltex \runningtitle{swNEMO\_v4.0: an ocean model NEMO for the new-generation Sunway supercomputer}?><?xmltex \runningauthor{Y. Ye et al.}?>
      <contrib-group>
        <contrib contrib-type="author" corresp="no" rid="aff1 aff2 aff3 aff4">
          <name><surname>Ye</surname><given-names>Yuejin</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff2 aff3 aff4">
          <name><surname>Song</surname><given-names>Zhenya</given-names></name>
          
        <ext-link>https://orcid.org/0000-0002-8098-5529</ext-link></contrib>
        <contrib contrib-type="author" corresp="no" rid="aff2 aff4 aff5">
          <name><surname>Zhou</surname><given-names>Shengchang</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff6">
          <name><surname>Liu</surname><given-names>Yao</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff2 aff3 aff4">
          <name><surname>Shu</surname><given-names>Qi</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff1">
          <name><surname>Wang</surname><given-names>Bingzhuo</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="no" rid="aff4 aff5">
          <name><surname>Liu</surname><given-names>Weiguo</given-names></name>
          
        </contrib>
        <contrib contrib-type="author" corresp="yes" rid="aff2 aff3 aff4">
          <name><surname>Qiao</surname><given-names>Fangli</given-names></name>
          <email>qiaofl@fio.org.cn</email>
        </contrib>
        <contrib contrib-type="author" corresp="yes" rid="aff4 aff7">
          <name><surname>Wang</surname><given-names>Lanning</given-names></name>
          <email>wangln@bnu.edu.cn</email>
        </contrib>
        <aff id="aff1"><label>1</label><institution>National Supercomputing Center, Wuxi 214000, China</institution>
        </aff>
        <aff id="aff2"><label>2</label><institution>First Institute of Oceanography, and Key Laboratory of Marine Science and Numerical Modeling, <?xmltex \hack{\break}?> Ministry of Natural Resources, Qingdao 266061, China</institution>
        </aff>
        <aff id="aff3"><label>3</label><institution>Shandong Key Laboratory of Marine Science and Numerical Modeling, Qingdao 266061, China</institution>
        </aff>
        <aff id="aff4"><label>4</label><institution>Laboratory for Regional Oceanography and Numerical Modeling,
Pilot National Laboratory for Marine Science and Technology, Qingdao 266237, China</institution>
        </aff>
        <aff id="aff5"><label>5</label><institution>School of Software, Shandong University, Jinan 250101, China</institution>
        </aff>
        <aff id="aff6"><label>6</label><institution>School of Data Science and Engineering, East China Normal University, Shanghai 200062, China</institution>
        </aff>
        <aff id="aff7"><label>7</label><institution>College of Global Change and Earth System Science, Beijing Normal University, Beijing 100875, China</institution>
        </aff>
      </contrib-group>
      <author-notes><corresp id="corr1">Fangli Qiao (qiaofl@fio.org.cn) and Lanning Wang (wangln@bnu.edu.cn)</corresp></author-notes><pub-date><day>25</day><month>July</month><year>2022</year></pub-date>
      
      <volume>15</volume>
      <issue>14</issue>
      <fpage>5739</fpage><lpage>5756</lpage>
      <history>
        <date date-type="received"><day>5</day><month>February</month><year>2022</year></date>
           <date date-type="rev-request"><day>2</day><month>March</month><year>2022</year></date>
           <date date-type="rev-recd"><day>30</day><month>June</month><year>2022</year></date>
           <date date-type="accepted"><day>4</day><month>July</month><year>2022</year></date>
      </history>
      <permissions>
        <copyright-statement>Copyright: © 2022 Yuejin Ye et al.</copyright-statement>
        <copyright-year>2022</copyright-year>
      <license license-type="open-access"><license-p>This work is licensed under the Creative Commons Attribution 4.0 International License. To view a copy of this licence, visit <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link></license-p></license></permissions><self-uri xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022.html">This article is available from https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022.html</self-uri><self-uri xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022.pdf">The full text article is available as a PDF file from https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022.pdf</self-uri>
      <abstract><title>Abstract</title>

      <p id="d1e201">The current large-scale parallel barrier of ocean general circulation models (OGCMs) makes it difficult to meet the computing demand of high resolution. Fully considering both the computational characteristics of OGCMs and the heterogeneous many-core architecture of the new Sunway supercomputer, swNEMO_v4.0, based on NEMO4 (Nucleus for European Modelling of the Ocean version 4), is developed with ultrahigh scalability. Three innovations and breakthroughs are shown in our work: (1) a highly adaptive, efficient four-level parallelization framework for OGCMs is proposed to release a new level of parallelism along the compute-dependency column dimension. (2) A many-core optimization method using blocking by remote memory access (RMA) and a dynamic cache scheduling strategy is applied, effectively utilizing the temporal and spatial locality of data. The test shows that the actual direct memory access (DMA) bandwidth is greater than 90 % of the ideal bandwidth after optimization, and the maximum is up to 95 %. (3) A mixed-precision optimization method with half, single and double precision is explored, which can effectively improve the computation performance while maintaining the simulated accuracy of OGCMs. The results demonstrate that swNEMO_v4.0 has ultrahigh scalability, achieving up to 99.29 % parallel efficiency with a resolution of 500 m using 27 988 480 cores, reaching the peak performance with 1.97 PFLOPS.</p>
  </abstract>
    </article-meta>
  </front>
<body>
      

<sec id="Ch1.S1" sec-type="intro">
  <label>1</label><title>Introduction</title>
      <p id="d1e213">Ocean general circulation models (OGCMs) are numerical models focusing on the properties of oceans based on the Navier–Stokes equations on the rotating sphere with thermodynamic terms for various energy sources <xref ref-type="bibr" rid="bib1.bibx6" id="paren.1"/>. OGCMs are the most powerful tools for predicting the ocean and climate states. As shown in Fig. <xref ref-type="fig" rid="Ch1.F1"/>, recent studies indicate that the horizontal resolution of OGCMs used for ocean research has been improved from 5<inline-formula><mml:math id="M1" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> (approximately 500 km) <xref ref-type="bibr" rid="bib1.bibx5 bib1.bibx4" id="paren.2"/> to <inline-formula><mml:math id="M2" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">48</mml:mn></mml:mrow></mml:math></inline-formula><inline-formula><mml:math id="M3" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> (approximately 2 km) <xref ref-type="bibr" rid="bib1.bibx20 bib1.bibx25 bib1.bibx8 bib1.bibx18 bib1.bibx19" id="paren.3"/>. However, the small-scale processes in the ocean <xref ref-type="bibr" rid="bib1.bibx6 bib1.bibx14" id="paren.4"/>, which are critical for further reducing the ocean simulation and prediction biases, still cannot be resolved within a 2 km resolution. Therefore, improving the spatial resolution is one of the most important directions of OGCM development.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F1"><?xmltex \currentcnt{1}?><?xmltex \def\figurename{Figure}?><label>Figure 1</label><caption><p id="d1e262">Peak performance of the supercomputers (blue bar) and the total number of OGCM grid points (red line) over the past 60 years.</p></caption>
        <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f01.png"/>

      </fig>

      <p id="d1e271"><?xmltex \hack{\newpage}?>The increase in OGCMs' resolution results in an exponential increase in the demand for computing capabilities. With the doubled horizontal resolution, the amount of calculation increases about 10 times accordingly. However, the enhanced performance of supercomputers makes it feasible to simulate oceans at higher resolutions. Compared to homogeneous systems, which cannot afford the high-power cost of the transition from petaFLOPS supercomputers to exaFLOPS supercomputers, heterogeneous many-core systems reduce the power loss from the perspective of system design, thus becoming mainstream.</p>

<?xmltex \floatpos{p}?><table-wrap id="Ch1.T1" orientation="landscape"><?xmltex \currentcnt{1}?><label>Table 1</label><caption><p id="d1e279">Research on models based on heterogeneous architectures. The bold font means the work of this paper.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="7">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:colspec colnum="3" colname="col3" align="left"/>
     <oasis:colspec colnum="4" colname="col4" align="left"/>
     <oasis:colspec colnum="5" colname="col5" align="left"/>
     <oasis:colspec colnum="6" colname="col6" align="left"/>
     <oasis:colspec colnum="7" colname="col7" align="left"/>
     <oasis:thead>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">Year</oasis:entry>
         <oasis:entry colname="col2">Models</oasis:entry>
         <oasis:entry colname="col3">Maximum resolutions</oasis:entry>
         <oasis:entry colname="col4">Platforms</oasis:entry>
         <oasis:entry colname="col5">Maximum scales</oasis:entry>
         <oasis:entry colname="col6">Parallel programming</oasis:entry>
         <oasis:entry colname="col7">Results</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">2015</oasis:entry>
         <oasis:entry colname="col2">POM.gpu</oasis:entry>
         <oasis:entry colname="col3">1922 <inline-formula><mml:math id="M4" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 1442 <inline-formula><mml:math id="M5" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 51</oasis:entry>
         <oasis:entry colname="col4">Nvidia TESLA K20X</oasis:entry>
         <oasis:entry colname="col5">4</oasis:entry>
         <oasis:entry colname="col6">MPI <inline-formula><mml:math id="M6" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> CUDA</oasis:entry>
         <oasis:entry colname="col7">Equivalent to 408 Westmere cores' performance</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">2019</oasis:entry>
         <oasis:entry colname="col2">LICOM2</oasis:entry>
         <oasis:entry colname="col3">360 <inline-formula><mml:math id="M7" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 218 <inline-formula><mml:math id="M8" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 30</oasis:entry>
         <oasis:entry colname="col4">Nvidia Tesla K80</oasis:entry>
         <oasis:entry colname="col5">4</oasis:entry>
         <oasis:entry colname="col6">MPI <inline-formula><mml:math id="M9" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> OpenACC</oasis:entry>
         <oasis:entry colname="col7">6.6<inline-formula><mml:math id="M10" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> the four Intel Xeon CPU E5-2690 v2 GPUs.</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">2020</oasis:entry>
         <oasis:entry colname="col2">LICOM3</oasis:entry>
         <oasis:entry colname="col3">7200 <inline-formula><mml:math id="M11" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 3920 <inline-formula><mml:math id="M12" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 55</oasis:entry>
         <oasis:entry colname="col4">AMD GFX906 GPU</oasis:entry>
         <oasis:entry colname="col5">26 200</oasis:entry>
         <oasis:entry colname="col6">MPI <inline-formula><mml:math id="M13" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> HIP</oasis:entry>
         <oasis:entry colname="col7">2.72 SYPD</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">2020</oasis:entry>
         <oasis:entry colname="col2">POP2</oasis:entry>
         <oasis:entry colname="col3">3600 <inline-formula><mml:math id="M14" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 2400 <inline-formula><mml:math id="M15" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 65</oasis:entry>
         <oasis:entry colname="col4">TaihuLight</oasis:entry>
         <oasis:entry colname="col5">1 189 500 cores</oasis:entry>
         <oasis:entry colname="col6">MPI <inline-formula><mml:math id="M16" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> Athread</oasis:entry>
         <oasis:entry colname="col7">3.8 <inline-formula><mml:math id="M17" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> speedup</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"><bold>2022</bold></oasis:entry>
         <oasis:entry colname="col2"><bold>swNEMO_v4.0</bold></oasis:entry>
         <oasis:entry colname="col3"><bold>82 500</bold> <inline-formula><mml:math id="M18" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> <bold>55 000</bold> <inline-formula><mml:math id="M19" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> <bold>128</bold></oasis:entry>
         <oasis:entry colname="col4"><bold>New-generation Sunway</bold></oasis:entry>
         <oasis:entry colname="col5"><bold>27 988 480 cores</bold></oasis:entry>
         <oasis:entry colname="col6"><bold>MPI</bold> <inline-formula><mml:math id="M20" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> <bold>Athread</bold></oasis:entry>
         <oasis:entry colname="col7"><inline-formula><mml:math id="M21" display="inline"><mml:mo>∼</mml:mo></mml:math></inline-formula><bold>8</bold> <inline-formula><mml:math id="M22" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula><bold>speedup, 99.29 % parallel efficiency</bold></oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <p id="d1e608">In the last few years, heterogeneous architectures (i.e., CPU <inline-formula><mml:math id="M23" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> GPU, <xref ref-type="bibr" rid="bib1.bibx24" id="altparen.5"/>; CPU <inline-formula><mml:math id="M24" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> FPGA, <xref ref-type="bibr" rid="bib1.bibx17" id="altparen.6"/>; CPU <inline-formula><mml:math id="M25" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> MIC, <xref ref-type="bibr" rid="bib1.bibx15" id="altparen.7"/>; and MPE <inline-formula><mml:math id="M26" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> CPE (computing processing element), <xref ref-type="bibr" rid="bib1.bibx10" id="altparen.8"/>) have been widely applied to speed up model computation (Table <xref ref-type="table" rid="Ch1.T1"/>). GPUs are increasingly important in high-performance computing due to their high-density computing. Using GPUs to accelerate computing speed shows great potential for the ocean and climate modeling. For example, based on mpiPOM, <xref ref-type="bibr" rid="bib1.bibx28" id="text.9"/> developed a POM (Princeton Ocean Model) GPU solution, which reduced power consumption by a factor of 6.8 times and achieved the equivalent performance of 408 Intel Westmere cores using four K20 GPUs <xref ref-type="bibr" rid="bib1.bibx28" id="paren.10"/>.
LICOM (LASG/IAP Climate system Ocean Model) was successfully ported and optimized to run on Nvidia and AMD GPUs, which showed great potential compared with CPUs <xref ref-type="bibr" rid="bib1.bibx13 bib1.bibx27" id="paren.11"/>. Compared with GPUs, the results of ocean models published with FPGA are relatively few, and most are in the program porting stage.</p>
      <p id="d1e664">On the platform of Sunway supercomputers, <xref ref-type="bibr" rid="bib1.bibx33" id="text.12"/> and <xref ref-type="bibr" rid="bib1.bibx32" id="text.13"/> studied the porting of the ocean component POP2 (Parallel Ocean Program version 2) and successfully scaled it to 1 million and 4 million cores, respectively, speeding up computation by a factor of 3 to 4. To use heterogeneous architectures, parallel programming of MPI <inline-formula><mml:math id="M27" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> OpenMP/CUDA/OpenACC/Athread was developed to improve the parallel efficiency of the model using finer-grained parallelism <xref ref-type="bibr" rid="bib1.bibx1" id="paren.14"/>. The coordination between the master and slave cores (involving master–slave cores, slave–slave cores, memory bandwidth and register communication) is the key to parallel efficiency.</p>
      <p id="d1e683">Due to the limitation of memory bandwidth, large-scale parallelism usually cannot maintain high efficiency <xref ref-type="bibr" rid="bib1.bibx14 bib1.bibx21 bib1.bibx6" id="paren.15"/>. The classic parallelization method is the “longitude–latitude 2D decomposition based on the MPI of OGCMs. To further improve parallelism, there are many efforts in parallel decomposition schemes and algorithm improvements of physical processes. To achieve better load balance, several decomposition schemes based on curve filling have been introduced in OGCMs (POP, <xref ref-type="bibr" rid="bib1.bibx22" id="altparen.16"/> and NEMO (Nucleus for European Modelling of the Ocean), <xref ref-type="bibr" rid="bib1.bibx16" id="altparen.17"/>). The parallel scale can reach <inline-formula><mml:math id="M28" display="inline"><mml:mi>O</mml:mi></mml:math></inline-formula> (100 km) in practice with a resolution of <inline-formula><mml:math id="M29" display="inline"><mml:mi>O</mml:mi></mml:math></inline-formula> (10 km) <xref ref-type="bibr" rid="bib1.bibx12 bib1.bibx29" id="paren.18"/>. However, due to the communication barrier, the large-scale parallel efficiency is still below 50 %. Therefore, we need to explore new schemes that can further improve the scalability to the next level. At the current stage, developing a large-scale parallel algorithm for the OGCMs is still in two-dimensional parallelism along with the longitudinal and latitudinal horizontal directions. Therefore, integrating the vertical direction in the three-dimensional parallelism scheme is still challenging.</p>
      <p id="d1e713">As most emerging Exascale systems provide support for mixed-precision arithmetic, a mixed-precision computing scheme becomes an important step to reduce the computational and memory pressure further, as well as to improve the computing performance. However, due to the weak support of mixed precision from computer systems and the difficulty of balancing computing precision and simulation accuracy, many efforts are still at an early stage. <xref ref-type="bibr" rid="bib1.bibx7" id="text.19"/> used a reduced-precision emulator (RPE) to study the shallow water equation (SWE) model using mixed precision and found that the error caused by iterations with low precision can be solved by mixed precision. Then, based on the RPE, <xref ref-type="bibr" rid="bib1.bibx23" id="text.20"/> investigated the application of a mixed scheme of double precision (DP) and single precision (SP) on NEMO and verified the feasibility of half precision (HP) in the regional ocean model of ROMS (Regional Ocean Modeling System) <xref ref-type="bibr" rid="bib1.bibx23" id="paren.21"/>. Previous studies have shown that mixed precision can improve computational efficiency. However, RPE can only verify the feasibility of mixed precision in theoretical models. The new-generation Sunway supercomputer, supporting HP, can lay a solid foundation for applying mixed-precision OGCMs.</p>
      <p id="d1e725">The architecture of the new-generation of Sunway processors (SW26010 Pro) adopts a more advanced DDR4 compared with the original SW26010. It not only expands the capacity, but also greatly improves the direct memory access (DMA) bandwidth of the processor. The upgrade of the on-chip communication mechanism of many-core arrays makes the interconnection between CPEs more convenient and builds a more efficient global network on the entire supercomputer. The upgrade of these key technologies provides a solid foundation for the parallelization of OGCMs on the new generation of Sunway supercomputers. In this work, we design and implement a four-level parallel algorithm on three-dimensional space based on hardware–software co-design. We also resolve the problem of memory bandwidth through fine-grained data reuse technology, thus paving the way for the ultrahigh scalability of NEMO. Furthermore, based on Sunway's heterogeneous many-core architecture, a composite block algorithm and a dynamic scheduling algorithm based on LDCache are proposed to fully exploit the performance of many-core acceleration. Finally, half precision is introduced to further release the memory pressure of NEMO under simulation with ultrahigh resolution. We develop a highly scalable swNEMO_v4.0 based on the Nucleus for European Modelling of the Ocean version 4 (NEMO4) with the GYRE-PISCES benchmark <xref ref-type="bibr" rid="bib1.bibx16" id="paren.22"/>, which is the benchmark abbreviation of the Gyre Pelagic Interactions Scheme for Carbon and Ecosystem Studies, where PISCES is short for Pelagic Interactions Scheme for Carbon and Ecosystem Studies <xref ref-type="bibr" rid="bib1.bibx2" id="paren.23"/>. The resolution is equivalent to the horizontal resolution of 500 m on a global scale.</p>
      <p id="d1e735">The following section briefly introduces the new generation of Sunway heterogeneous many-core supercomputing platforms. In Sect. <xref ref-type="sec" rid="Ch1.S3"/>, we briefly review the basics of NEMO4 and describe our optimization methods in detail. The performance results are discussed in Sect. <xref ref-type="sec" rid="Ch1.S4"/>, and Sect. <xref ref-type="sec" rid="Ch1.S5"/> concludes this paper with discussions.</p>
</sec>
<sec id="Ch1.S2">
  <label>2</label><title>The new-generation heterogeneous many-core supercomputing platform and NEMO4</title>
      <p id="d1e752">Succeeded by the architecture of Sunway TaihuLight, the new Sunway generation, as shown in Fig. <xref ref-type="fig" rid="Ch1.F2"/>, which is driven by SW26010 Pro (Table <xref ref-type="table" rid="Ch1.T2"/>), consists of six core groups, with one management processing element (MPE) and one <inline-formula><mml:math id="M30" display="inline"><mml:mrow><mml:mn mathvariant="normal">8</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">8</mml:mn></mml:mrow></mml:math></inline-formula> computing processing element (CPE) cluster in each. The core groups are connected within the loop network. Data can be transferred between CPEs via remote memory access (RMA), which significantly increases the efficiency of CPE co-working.</p>
      <p id="d1e771">CPE is also based on the SW64 instruction set, with the 512 bit single-instruction multiple-data (SIMD) vector, where double precision, single precision, half precision and integers are all supported. Furthermore, double and single precision share the same computation speed, whereas half precision performs twice faster. There are also a separate instruction cache and a scratchpad memory (SPM) in each CPE. The SPM allocates local data memory (LDM) for users, part of which can be set as a local data cache, automatically administrated by hardware. The data are transmitted between LDM and main memory via either direct memory access (DMA) or the general load and store instructions.</p>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T2"><?xmltex \currentcnt{2}?><label>Table 2</label><caption><p id="d1e777">Basic information about SW26010 Pro. Note: CG refers to core group.</p></caption><oasis:table frame="topbot"><oasis:tgroup cols="2">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="left"/>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">CPU</oasis:entry>
         <oasis:entry colname="col2">SW26010 Pro</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Number of CGs</oasis:entry>
         <oasis:entry colname="col2">6</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Number of processors of one CG</oasis:entry>
         <oasis:entry colname="col2">65 (1 MPE <inline-formula><mml:math id="M31" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> 64 CPE)</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Programming language</oasis:entry>
         <oasis:entry colname="col2">C/C<inline-formula><mml:math id="M32" display="inline"><mml:mrow><mml:mo>+</mml:mo><mml:mo>+</mml:mo></mml:mrow></mml:math></inline-formula>, Fortran, Python</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Parallel programming environment</oasis:entry>
         <oasis:entry colname="col2">MPI, Athread/OpenACC</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Memory size</oasis:entry>
         <oasis:entry colname="col2">96 GB</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Instruction set</oasis:entry>
         <oasis:entry colname="col2">sw64</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">L1 instruction cache</oasis:entry>
         <oasis:entry colname="col2">32 KB</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">L1 data cache</oasis:entry>
         <oasis:entry colname="col2">32 KB</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">L2 data cache of MPE</oasis:entry>
         <oasis:entry colname="col2">512 KB</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SIMD register of MPE</oasis:entry>
         <oasis:entry colname="col2">256 bit</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">SIMD register of CPE</oasis:entry>
         <oasis:entry colname="col2">512 bit</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Precision support</oasis:entry>
         <oasis:entry colname="col2">Double precision</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Single precision</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1"/>
         <oasis:entry colname="col2">Half precision</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">Scratchpad memory</oasis:entry>
         <oasis:entry colname="col2">256 KB</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup></oasis:table></table-wrap>

      <?xmltex \floatpos{t}?><fig id="Ch1.F2" specific-use="star"><?xmltex \currentcnt{2}?><?xmltex \def\figurename{Figure}?><label>Figure 2</label><caption><p id="d1e956">Architecture of the new-generation Sunway supercomputer.</p></caption>
        <?xmltex \igopts{width=341.433071pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f02.png"/>

      </fig>

</sec>
<sec id="Ch1.S3">
  <label>3</label><title>Porting and optimizing NEMO4</title>
      <p id="d1e973">NEMO is a state-of-the-art modeling framework for research activities and forecasting services in the ocean and climate sciences developed in a sustainable way by a European consortium since 2008. It has been widely used in marine science, climate change studies, ocean forecasting systems and climate models. For ocean forecasting, NEMO has been applied by many operational forecasting systems, such as the Mercator Ocean monitoring and forecasting systems <xref ref-type="bibr" rid="bib1.bibx14" id="paren.24"/>, as well as the ocean forecasting system in the National Marine Environmental Forecasting Center of China with a <inline-formula><mml:math id="M33" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">12</mml:mn></mml:mrow></mml:math></inline-formula><inline-formula><mml:math id="M34" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> resolution <xref ref-type="bibr" rid="bib1.bibx26" id="paren.25"/>. For the climate simulation and projections, approximately one-third of the climate models in the latest phase of the Coupled Model Intercomparison Project Phase 6 (CMIP6) <xref ref-type="bibr" rid="bib1.bibx9" id="paren.26"/> use the ocean component models of NEMO. The breakthroughs made in this study are based on NEMO, and, thus, this work provides useful methods and ideas that can directly contribute to the NEMO community for optimizing their models and improving their simulation speeds.</p>
      <p id="d1e1005">Fully considering the characteristics of the three-dimensional spatial computation of OGCMs and the heterogeneous many-core architecture of the new-generation Sunway supercomputer, we develop a highly scalable NEMO4 named swNEMO_v4.0, with the following three major contributions based on the concepts of hardware–software co-design:
<list list-type="bullet"><list-item>
      <p id="d1e1010">A highly efficient four-level parallelization framework is proposed for OGCMs to release a new level of parallelism along the column dimension that was originally not parallelism-friendly due to the computational dependency.</p></list-item><list-item>
      <p id="d1e1014">A raised many-core optimization method that uses an effective dynamic cache scheduling strategy utilizes the temporal and spatial locality of data effectively.</p></list-item><list-item>
      <p id="d1e1018">A multi-level mixed-precision optimization method that uses half, single and double precision is explored, which can effectively improve the computation performance while maintaining the same level of accuracy.</p></list-item></list>
We then elaborate on the above algorithms in the following.</p>
<sec id="Ch1.S3.SS1">
  <label>3.1</label><title>An adaptive four-level parallelization framework</title>
      <p id="d1e1029">To utilize the many heterogeneous cores of the new-generation Sunway supercomputer, the load of the expanded C-grid computation should be assigned in a balanced way. Considering the characteristics of the three-dimensional spatial computation of NEMO and the heterogeneous many-core architecture of a new-generation Sunway supercomputer, we propose an adaptive four-level parallel framework that can realize computation load dispatching among different levels. Figure <xref ref-type="fig" rid="Ch1.F3"/> demonstrates our adaptive four-level parallel framework.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F3"><?xmltex \currentcnt{3}?><?xmltex \def\figurename{Figure}?><label>Figure 3</label><caption><p id="d1e1036">The adaptive four-level parallel framework, where <inline-formula><mml:math id="M35" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="M36" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M37" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> axes indicate latitude, longitude and depth, respectively. Level 1 is the load-balanced longitude–latitude decomposition among MPEs, level 2 is the asynchronous parallel between a MPE and a CPE cluster, level 3 is the latitude–depth decomposition in a CPE clusters and level 4 is the vector reconstruction in a CPE.</p></caption>
          <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f03.png"/>

        </fig>

<sec id="Ch1.S3.SS1.SSS1">
  <label>3.1.1</label><title>Level 1: load-balanced longitude–latitude decomposition among MPEs</title>
      <p id="d1e1073">With land grids eliminated in NEMO, keeping the horizontal data in a traditional way benefits the load balance among processes. The variables are stored as 3-D arrays, and the <inline-formula><mml:math id="M38" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="M39" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M40" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> axes represent the longitude, latitude and depth, respectively. Since the scales of the <inline-formula><mml:math id="M41" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and <inline-formula><mml:math id="M42" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis are considerably larger than the scale of the <inline-formula><mml:math id="M43" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> axis, we decompose the data along the <inline-formula><mml:math id="M44" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and <inline-formula><mml:math id="M45" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis and dispatch each data block to different MPI processes (first level of the four-level parallelization framework). Our major strategy is to guarantee that (a) the grid sizes of the subdomain in each process are similar, achieving a good load balance; and (b) the <inline-formula><mml:math id="M46" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and <inline-formula><mml:math id="M47" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis dimension sizes are close to each other. The <inline-formula><mml:math id="M48" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and <inline-formula><mml:math id="M49" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis of each process are the closest in all options, thus minimizing the halo areas that require communication among processes.</p>
</sec>
<sec id="Ch1.S3.SS1.SSS2">
  <label>3.1.2</label><title>Level 2: asynchronous parallelization between a MPE and a CPE cluster</title>
      <p id="d1e1171"><xref ref-type="bibr" rid="bib1.bibx11" id="text.27"/> proposed a multi-level optimization method to use the heterogeneous architecture. In the first level of the method, they designed pre-communication, communication and post-communication in the code based on MPE–CPE asynchronous parallelism architecture. As the default barotropic solver in NEMO4 is an explicit method with a small time step, which is more accurate and more suitable for high resolution without a filter, instead of implicit or split-explicit methods (e.g., preconditioned conjugate gradients – PCGs), there is no global communication. Therefore, it is only necessary to update the information of the halo region between different processes. Moreover, in the explicit method, most boundary information exchanges are independent of the partition data in the process and can be used for asynchronous parallelization. So boundary data exchange can be parallelized aside from the computation. Considering that MPE is asynchronous with a CPE cluster, we propose an asynchronous parallelization design between the MPE and the CPE cluster using efficient DMA (second level of the four-level parallelization framework). The MPE is in charge of the boundary data exchange, I/O and a small amount of computation, while the CPE cluster performs the computation of most kernels. Such an asynchronous communication pattern can make full use of the asynchronous parallelism between the MPE and the CPE cluster, thus improving the parallel efficiency.</p>
</sec>
<sec id="Ch1.S3.SS1.SSS3">
  <label>3.1.3</label><title>Level 3: “latitude–depth” decomposition in the CPE cluster</title>
      <p id="d1e1184">Furthermore, we design the third level of parallelism in the CPE cluster by utilizing the fine-grained data-sharing features within the CPE cluster, releasing a new level of parallelism along the column dimension that was originally not parallelism-friendly due to the computational dependency. RMA is a unique on-chip communication mechanism on the new-generation Sunway supercomputer, which enables high-speed communication among CPEs. We realize the latitude–depth decomposition with LDM and the data exchange with RMA. By utilizing the row and column communication features of RMA to achieve fine-grained data sharing among the CPE threads, we can accomplish an efficient parallelization on the column (i.e., the depth) dimension and release a new level of parallelism for the underlying hardware.</p>
      <p id="d1e1187">In the data structure of swNEMO, data items along the longitude axis are stored in a continuous way. Following such a memory layout, we divide the data block along the latitude and depth into smaller blocks and copy them into the LDM. More details are shown in Sect. <xref ref-type="sec" rid="Ch1.S3.SS2"/>.</p><?xmltex \hack{\newpage}?>
</sec>
<sec id="Ch1.S3.SS1.SSS4">
  <label>3.1.4</label><title>Level 4: data layout reconstruction for vectorization</title>
      <p id="d1e1201">To further improve the computational efficiency within each CPE, we design the fourth level of parallelism for vectorization. SW26010 Pro has a 512 bit SIMD instruction set, one of which can compute eight double-precision digits simultaneously. To adapt our computation pattern for the SIMD instruction, we perform a data layout reconstruction to achieve the most suitable vectorization arrangement. Moreover, methods such as instruction rearrangement, branch prediction and cycle expansion are adopted to improve the execution efficiency of the instruction pipeline.</p>
</sec>
</sec>
<sec id="Ch1.S3.SS2">
  <label>3.2</label><title>Performance optimization for the many-core architecture</title>
<sec id="Ch1.S3.SS2.SSS1">
  <label>3.2.1</label><title>A composite blocking algorithm based on RMA</title>
      <p id="d1e1220">To further improve the parallel scale and efficiency, we implement the fine-grained latitude–depth decomposition. There are many stencils and temporal dependency computations existing in NEMO, which results in high demand for DMA bandwidth. A stencil computation is a class of algorithms that updates elements in a multidimensional grid based on neighboring values using a fixed pattern (hereafter called stencil). In a stencil operation, each point in a multidimensional grid is updated with the weighted contributions from a subset of its neighbors in both time and space, thereby representing the coefficients of the partial differential equation (PDE) for that data element. Therefore, we take advantage of the RMA provided by SW26010 Pro to relieve the pressure of the DMA bandwidth. RMA is an on-chip communication mechanism with superior bisection bandwidth within a CPE cluster. RMA enables direct remote LDM access among different CPEs within one CPE cluster. The efficient batch communication mechanism of RMA is highly adaptable for solving typical <inline-formula><mml:math id="M50" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> pointer problems. For example, in the diffusion process of the <italic>tracer</italic> in NEMO, upstream points are needed for data exchange, which includes horizontal unidirectional grid information exchange (three-pointer stencil) and grid information exchange along with longitude and latitude directions (five-pointer stencil). Figure <xref ref-type="fig" rid="Ch1.F4"/> represents the grid communication process between CPE #0 and CPE #1. Each point represents a multidimensional tensor composed of different variables that are irrelevant to each other. When updating <inline-formula><mml:math id="M51" display="inline"><mml:mrow><mml:mi>A</mml:mi><mml:mo>(</mml:mo><mml:mi>u</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, variables <inline-formula><mml:math id="M52" display="inline"><mml:mi>v</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M53" display="inline"><mml:mi>w</mml:mi></mml:math></inline-formula> from the surrounding eight points are required to participate in the calculation. We first send <inline-formula><mml:math id="M54" display="inline"><mml:mi>v</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="M55" display="inline"><mml:mi>w</mml:mi></mml:math></inline-formula> required by adjacent CPEs to the buffer in the corresponding CPEs while updating the local variables whose surrounding points are on the same CPE. After all the needed variables in the halo are transferred into the buffer, the remaining variables in the points located at the edge of the CPE can be updated.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F4"><?xmltex \currentcnt{4}?><?xmltex \def\figurename{Figure}?><label>Figure 4</label><caption><p id="d1e1280">Tensor distribution in different CPEs <bold>(a)</bold> and data transportation between CPEs <bold>(b)</bold>.</p></caption>
            <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f04.png"/>

          </fig>

      <p id="d1e1295">Based on RMA, we design different parallel blocking algorithms for computing kernels with different characteristics to maintain an efficient performance in various application scenarios. In the following,  <inline-formula><mml:math id="M56" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="italic">α</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M57" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="italic">β</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M58" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="italic">β</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> are only the coefficient, <inline-formula><mml:math id="M59" display="inline"><mml:mi>f</mml:mi></mml:math></inline-formula> is the mathematical formula for the loop segment and <inline-formula><mml:math id="M60" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> means the <inline-formula><mml:math id="M61" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis (longitude axis) in the coordinate system.</p>
      <p id="d1e1354"><list list-type="bullet">
              <list-item>

      <p id="d1e1359"><italic>Temporal dependency in computing along the</italic> <inline-formula><mml:math id="M62" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> <italic>axis</italic>. We suppose <inline-formula><mml:math id="M63" display="inline"><mml:mrow><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> is an original value of grid <inline-formula><mml:math id="M64" display="inline"><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> in the coordinate system <inline-formula><mml:math id="M65" display="inline"><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>z</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> (where the <inline-formula><mml:math id="M66" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis is the most continuous one, and the <inline-formula><mml:math id="M67" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> axis is the least), and <inline-formula><mml:math id="M68" display="inline"><mml:mrow><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup></mml:mrow></mml:math></inline-formula> is computed as follows:
                    <disp-formula id="Ch1.E1" content-type="numbered"><label>1</label><mml:math id="M69" display="block"><mml:mrow><mml:mfenced open="{" close=""><mml:mtable class="array" columnalign="left left"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mo>(</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>f</mml:mi></mml:mtd><mml:mtd><mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="italic">α</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:msub><mml:mo>×</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi mathvariant="italic">β</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mo>×</mml:mo><mml:mo>(</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>±</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>±</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mfenced></mml:mrow></mml:math></disp-formula>
                  where <inline-formula><mml:math id="M70" display="inline"><mml:mrow><mml:mn mathvariant="normal">0</mml:mn><mml:mo>&lt;</mml:mo><mml:mi>i</mml:mi><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mi>x</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M71" display="inline"><mml:mrow><mml:mn mathvariant="normal">0</mml:mn><mml:mo>&lt;</mml:mo><mml:mi>j</mml:mi><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mi>y</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M72" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>&lt;</mml:mo><mml:mi>k</mml:mi><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>. In this equation, <inline-formula><mml:math id="M73" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="italic">α</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M74" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="italic">β</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> are coefficients that indicate the linear characteristic of this mapping <inline-formula><mml:math id="M75" display="inline"><mml:mi>f</mml:mi></mml:math></inline-formula>. To fully utilize the DMA bandwidth, we decompose data along the <inline-formula><mml:math id="M76" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis and send them into different CPEs. Then, continuous data along the <inline-formula><mml:math id="M77" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis are copied into the LDM at one time.</p>
              </list-item>
              <list-item>

      <p id="d1e1738"><italic>Temporal dependency in computation along the</italic> <inline-formula><mml:math id="M78" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> <italic>axis</italic>. In this case, the computation is as follows:
                    <disp-formula id="Ch1.E2" content-type="numbered"><label>2</label><mml:math id="M79" display="block"><mml:mrow><mml:mfenced open="{" close=""><mml:mtable class="array" columnalign="left left"><mml:mtr><mml:mtd><mml:mrow><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mo>(</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>f</mml:mi></mml:mtd><mml:mtd><mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="italic">α</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msub><mml:mo>×</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi mathvariant="italic">β</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mrow><mml:mo>×</mml:mo><mml:mo>(</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>±</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>u</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi><mml:mo>±</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi mathvariant="normal">iter</mml:mi></mml:msubsup><mml:mo>)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mfenced></mml:mrow></mml:math></disp-formula>
                  Here <inline-formula><mml:math id="M80" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="italic">α</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M81" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="italic">β</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msub></mml:mrow></mml:math></inline-formula> are also coefficients that indicate the linear characteristic of this mapping <inline-formula><mml:math id="M82" display="inline"><mml:mi>f</mml:mi></mml:math></inline-formula>. Restricted by the size of the LDM, we decompose data along the <inline-formula><mml:math id="M83" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> axis and then decompose data along the <inline-formula><mml:math id="M84" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis into the proper size <inline-formula><mml:math id="M85" display="inline"><mml:mi>m</mml:mi></mml:math></inline-formula> and dispatch each data block into different CPEs as follows:
                    <disp-formula id="Ch1.E3" content-type="numbered"><label>3</label><mml:math id="M86" display="block"><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:mi mathvariant="normal">size</mml:mi><mml:mo>(</mml:mo><mml:mi mathvariant="normal">LDM</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="normal">size</mml:mi><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
                  Therefore, blocks on the same <inline-formula><mml:math id="M87" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> layer are dispatched to the same CPE one by one.</p>
              </list-item>
              <list-item>

      <p id="d1e1999"><italic>Nontemporal dependency in computation</italic>. In this case, we decompose the data along with the <inline-formula><mml:math id="M88" display="inline"><mml:mi>z</mml:mi></mml:math></inline-formula> axis and <inline-formula><mml:math id="M89" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis. Due to the elaborate design of size (<inline-formula><mml:math id="M90" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula>), each block can be copied into LDM at once to reduce redundant halo transfer and improve the asynchronous parallel of stencil calculations.</p>
              </list-item>
            </list></p>
      <p id="d1e2027">To increase the utility of the RMA bandwidth, we pack the data before sending them for better utilization of the bandwidth in an aggregated manner. While sending, CPEs compute, and in this way, we can realize the overlap between RMA communication and computation and, thus, make efficient utilization of the bandwidth by reducing redundant DMA references. By utilizing RMA and the composite blocking strategy, for nontemporal-dependency and temporal-dependency computation, we finally achieve over 90 % effective utilization of the RMA bandwidth and over 80 % utilization of the DDR4 memory bandwidth. With such an efficient memory scheme, the average speedup of most computing kernels (comparing the performance of a CPE cluster to an MPE) can be up to 40 times.</p>
</sec>
<sec id="Ch1.S3.SS2.SSS2">
  <label>3.2.2</label><title>A dynamic LDCache scheduling algorithm</title>
      <p id="d1e2038">To access discrete data items in swNEMO, the traditional global load/global store (gld/gst) method becomes a major performance hindrance. LDCache, provided by SW26010 Pro, stores data with a cache_line of 256 bytes in CPE. In one CPE, LDM and LDCache share an SPM with a total capacity of 256 KB, and the size of LDM and LDCache can be manually adjusted. Since the amount of data required for a round of computation in different kernels in NEMO is varied, if the cache space is fixed, the utilization of the LDM becomes less efficient. Therefore, we design a dynamic LDCache scheduling algorithm that can realize efficient and fine-grained memory access. One feature of this algorithm is to dynamically adjust the size of LDCache to achieve a balance between LDM and LDCache. Furthermore, the algorithm has a time-division update technique since LDCache cannot guarantee data consistency with memory. We regularly update the stored data and eliminate the outdated data in the LDM simultaneously. As shown in Fig. <xref ref-type="fig" rid="Ch1.F5"/>, the data that need to be refreshed are packed on the CPE and sent to the designated buffer of the MPE. The MPE then uses the MPE–CPE message mechanism to find the buffers that need to be updated in a round-robin way and update them, while the CPE eliminates the corresponding data in the cache. By applying the dynamic LDCache scheduling algorithm with both an adjustable cache and a manual time-division update technique, we can improve the memory bandwidth utilization rate to approximately 88.7 % for DDR4, with a speedup of 88 times (comparing the performance of a CPE cluster to an MPE) for most computing kernels, which is a substantial improvement compared with the speedup of 5.1 times when using the traditional gld/gst method.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F5"><?xmltex \currentcnt{5}?><?xmltex \def\figurename{Figure}?><label>Figure 5</label><caption><p id="d1e2045">Time-division refresh technology for LDM.</p></caption>
            <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f05.png"/>

          </fig>

</sec>
</sec>
<sec id="Ch1.S3.SS3">
  <label>3.3</label><title>Mixed-precision optimization</title>
      <p id="d1e2064">Because the NEMO, as one of the OGCMs, is memory-intensive, the memory bandwidth limits the computational efficiency of NEMO to a large extent. A reduced-precision method can be a promising solution. However, using low-precision data is a double-edged sword.
On the one hand, it can effectively resolve the performance obstacles; on the other hand, the reduced precision brings errors and uncertainties. According to <xref ref-type="bibr" rid="bib1.bibx7" id="text.28"/>, 95 % of NEMO variables support the single-precision floating-point format (SP). Therefore, we specifically reconstruct the data structure and introduce a new three-level mixed precision scheme in NEMO. To achieve a higher performance (Fig. <xref ref-type="fig" rid="Ch1.F6"/>), we reconstruct two “half-precision <inline-formula><mml:math id="M91" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> single-precision” (HP <inline-formula><mml:math id="M92" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> SP) computing kernels <italic>tracer_fct</italic> and <italic>tracer_iso</italic> of NEMO on CPE, which account for 50 % of the hotspot runtime in total. As HP is only supported on the CPEs of SW26010 Pro, we store HP data in the format of char or short types of memory on the MPE and use HP format to read them to LDM by DMA to ensure the correctness on the CPE.</p>
      <p id="d1e2093">By analyzing the calculation characteristics of NEMO, we find that the HP format subtracts operations between adjacent grids, which makes the final results diverge due to the round error of low precision. Therefore, a high-precision format must be used when calculating adjacent grids. Based on the above analysis in other calculations, we adopt the SP floating-point format for calculations between adjacent grids and the HP floating-point format with BF16 (consisting of 1 sign bit, 8 exponent bits and 7 mantissa bits). After optimization, we achieve a speedup of close to 2 times compared with NEMO using DP, while the maximum biases of temperature, salinity and velocity are within 0.05 % (figures not shown). Therefore, utilizing mixed precision can effectively increase the computational intensity and improve the scalability while maintaining the simulated accuracy of NEMO.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F6"><?xmltex \currentcnt{6}?><?xmltex \def\figurename{Figure}?><label>Figure 6</label><caption><p id="d1e2098">Data reconstruction for three-level mixed precision.</p></caption>
          <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f06.png"/>

        </fig>

      <p id="d1e2108">In order to validate the results of optimized swNEMO_v4.0, we carried out the perturbation experiments for the temperature results and analyzed the RMSZ (root-mean-square <inline-formula><mml:math id="M93" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> score) proposed by <xref ref-type="bibr" rid="bib1.bibx3" id="text.29"/>. The experimental sample set can be represented as an ensemble <inline-formula><mml:math id="M94" display="inline"><mml:mrow><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mo mathvariant="italic">{</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mn mathvariant="normal">2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mn mathvariant="normal">3</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>m</mml:mi></mml:msub><mml:mo mathvariant="italic">}</mml:mo></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M95" display="inline"><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mn mathvariant="normal">101</mml:mn></mml:mrow></mml:math></inline-formula>, where each sample <inline-formula><mml:math id="M96" display="inline"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>m</mml:mi></mml:msub><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> represents a series of results at a given point <inline-formula><mml:math id="M97" display="inline"><mml:mi>j</mml:mi></mml:math></inline-formula> in all months. <inline-formula><mml:math id="M98" display="inline"><mml:mrow><mml:mi mathvariant="italic">μ</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M99" display="inline"><mml:mrow><mml:mi mathvariant="italic">δ</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> represent the mean and standard deviations of this series at the given point <inline-formula><mml:math id="M100" display="inline"><mml:mi>j</mml:mi></mml:math></inline-formula>.
            <disp-formula id="Ch1.E4" content-type="numbered"><label>4</label><mml:math id="M101" display="block"><mml:mrow><mml:mi mathvariant="normal">RMSZ</mml:mi><mml:mo>(</mml:mo><mml:mi>X</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">1</mml:mn><mml:mi>n</mml:mi></mml:mfrac></mml:mstyle><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:munderover><mml:msup><mml:mfenced close=")" open="("><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mrow><mml:mi>X</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi mathvariant="italic">μ</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi mathvariant="italic">δ</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mfrac></mml:mstyle></mml:mfenced><mml:mn mathvariant="normal">2</mml:mn></mml:msup></mml:mrow></mml:msqrt></mml:mrow></mml:math></disp-formula></p>
      <p id="d1e2317">Since the experiment in this work is the ocean basin benchmark, it is not feasible to directly compare with observed values. Hence random disturbance conditions for the initial temperature are configured within the NEMO program. At the same time, we also selected 101 sets of data as experimental simulation results to validate the results of optimized NEMO_v4.0, of which 100 sets are temperature results with perturbation conditions, and the remaining one is the mixed-precision temperature field experimental results of swNEMO_v4.0.</p>
      <p id="d1e2320">At first, the perturbation coefficient is selected as <inline-formula><mml:math id="M102" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">14</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. However, we found that the RMSZ obtained by mixed-precision simulation was completely out of the shadow area formed by 100 simulations with double precision (Fig. <xref ref-type="fig" rid="Ch1.F7"/>a). Therefore, we increased the perturbation coefficient, and the mixed-precision <inline-formula><mml:math id="M103" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> score gradually approached the shadow area. We found that the disturbance had a greater influence in the first few years, then gradually decreased and tended to be stable after several years (Fig. <xref ref-type="fig" rid="Ch1.F7"/>a and b).  As shown in Fig. <xref ref-type="fig" rid="Ch1.F7"/>b, when the perturbation coefficient is <inline-formula><mml:math id="M104" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">11</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, the <inline-formula><mml:math id="M105" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> score of mixed-precision simulation falls partially in the region formed by double-precision simulations. When the perturbation coefficient equals <inline-formula><mml:math id="M106" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">10</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, the <inline-formula><mml:math id="M107" display="inline"><mml:mi>Z</mml:mi></mml:math></inline-formula> score completely falls in the shadow area (Fig. <xref ref-type="fig" rid="Ch1.F7"/>c), which indicates that the effects of mixed precision are similar to these of the perturbation coefficient of <inline-formula><mml:math id="M108" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">10</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. It shows that the mixed-precision affects the results, but the effects are small (the perturbation coefficient is around <inline-formula><mml:math id="M109" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">10</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>) and can be accepted (Fig. <xref ref-type="fig" rid="Ch1.F7"/>c).</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F7" specific-use="star"><?xmltex \currentcnt{7}?><?xmltex \def\figurename{Figure}?><label>Figure 7</label><caption><p id="d1e2458">RMSZ biases of the sea surface temperature in ensemble experiments (gray lines) with perturbation coefficient of <bold>(a)</bold> <inline-formula><mml:math id="M110" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">14</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, <bold>(b)</bold> <inline-formula><mml:math id="M111" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">11</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> and <bold>(c)</bold> <inline-formula><mml:math id="M112" display="inline"><mml:mrow><mml:mi>O</mml:mi><mml:mo>(</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">10</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> and in the mixed-precision experiment (red line).</p></caption>
          <?xmltex \igopts{width=497.923228pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f07.png"/>

        </fig>

</sec>
</sec>
<sec id="Ch1.S4">
  <label>4</label><title>Performance results</title>
      <p id="d1e2546">We choose the benchmark named GYRE-PISCES to test the swNEMO_v4.0 performance. The domain geometry is a closed rectangular basin on the <inline-formula><mml:math id="M113" display="inline"><mml:mi mathvariant="italic">β</mml:mi></mml:math></inline-formula> plane centered at <inline-formula><mml:math id="M114" display="inline"><mml:mrow><mml:mo>∼</mml:mo><mml:mn mathvariant="normal">30</mml:mn></mml:mrow></mml:math></inline-formula><inline-formula><mml:math id="M115" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> N and rotated by 45<inline-formula><mml:math id="M116" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>, 3180 km long, 2120 km wide and 4 km deep. The circulation is forced by analytical profiles of wind and buoyancy fluxes. This benchmark represents an idealized North Atlantic or North Pacific basin <xref ref-type="bibr" rid="bib1.bibx16" id="paren.30"/>. In addition, the east–west periodic conditions and the North Pole folding of the global ocean with a tripolar grid have a large impact on performance. Therefore, we activate the BENCH option to include these periodicity conditions and reproduce the communication pattern of the global ocean with a tripolar grid between two North Pole subdomains. It is equivalent to a global ocean with a tripolar grid with the same number of grid points from the perspective of computational cost and computational characteristics, although the physical meaning is limited. In the following content, the resolution of the benchmark is equivalent to that of the global ocean.</p>
      <p id="d1e2586">In this part, we have added a more specific description. We designed three groups of experiments with different resolutions of 2 km, 1 km and 500 m for strong expansibility. In addition, we also carried out weak-scalability experiments. The corresponding relationship among resolution, computing grid points and data scale of weak scalability can also be seen from the information in Table <xref ref-type="table" rid="Ch1.T3"/>. The speedup is equal to the clock time in different scales divided by the baseline record of the minimum scale with 2 129 920 cores. For the weak-scalability analysis, we design the experiment with eight resolutions (Table <xref ref-type="table" rid="Ch1.T3"/>). All experiments are run for 1 model day without I/O. We adopt the following methods to perform real-time statistics and floating points to ensure measurement accuracy.
<list list-type="bullet"><list-item>
      <p id="d1e2595">For time statistics, we use two methods to proofread:
<list list-type="bullet"><list-item>
      <p id="d1e2600">We use the MPI_Wtime() function provided by MPI to obtain the wall-clock time.</p></list-item><list-item>
      <p id="d1e2604">We use the assembly instructions to count the cycle time.</p></list-item></list></p></list-item><list-item>
      <p id="d1e2608">Similarly, two methods are used in the statistics of floating-point operations:
<list list-type="bullet"><list-item>
      <p id="d1e2613">We use the loader to count the floating-point operation of the program when submitting the job.</p></list-item><list-item>
      <p id="d1e2617">We use performance interface functions to perform the program instrumentation to count the operations.</p></list-item></list></p></list-item></list></p>

<?xmltex \floatpos{t}?><table-wrap id="Ch1.T3"><?xmltex \currentcnt{3}?><label>Table 3</label><caption><p id="d1e2623">Eight different scales used in weak scalability, with the conversion between resolution in degrees and resolution in kilometers. The total number of computed grid points used in weak scalability is also shown.</p></caption><oasis:table frame="topbot"><?xmltex \begin{scaleboxenv}{.95}[.95]?><oasis:tgroup cols="4">
     <oasis:colspec colnum="1" colname="col1" align="left"/>
     <oasis:colspec colnum="2" colname="col2" align="right"/>
     <oasis:colspec colnum="3" colname="col3" align="right"/>
     <oasis:colspec colnum="4" colname="col4" align="right"/>
     <oasis:thead>
       <oasis:row>
         <oasis:entry colname="col1">Scales in weak</oasis:entry>
         <oasis:entry colname="col2">Resolution</oasis:entry>
         <oasis:entry colname="col3">Resolution</oasis:entry>
         <oasis:entry colname="col4">Computed grid</oasis:entry>
       </oasis:row>
       <oasis:row rowsep="1">
         <oasis:entry colname="col1">scalability (cores)</oasis:entry>
         <oasis:entry colname="col2">(<inline-formula><mml:math id="M117" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula>)</oasis:entry>
         <oasis:entry colname="col3">(km)</oasis:entry>
         <oasis:entry colname="col4">points (horizontal)</oasis:entry>
       </oasis:row>
     </oasis:thead>
     <oasis:tbody>
       <oasis:row>
         <oasis:entry colname="col1">2 129 920</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M118" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">12</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">9.0</oasis:entry>
         <oasis:entry colname="col4">13 515 004</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">4 259 840</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M119" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">16</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">7.0</oasis:entry>
         <oasis:entry colname="col4">24 020 004</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">8 519 680</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M120" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">24</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">4.5</oasis:entry>
         <oasis:entry colname="col4">54 030 004</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">12 779 520</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M121" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">32</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">3.5</oasis:entry>
         <oasis:entry colname="col4">96 040 004</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">17 039 360</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M122" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">44</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">2.5</oasis:entry>
         <oasis:entry colname="col4">181 555 004</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">21 299 200</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M123" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">64</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">2.0</oasis:entry>
         <oasis:entry colname="col4">384 080 004</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">25 559 040</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M124" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">96</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">1.2</oasis:entry>
         <oasis:entry colname="col4">864 120 004</oasis:entry>
       </oasis:row>
       <oasis:row>
         <oasis:entry colname="col1">27 988 480</oasis:entry>
         <oasis:entry colname="col2"><inline-formula><mml:math id="M125" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">116</mml:mn></mml:mrow></mml:math></inline-formula></oasis:entry>
         <oasis:entry colname="col3">1.0</oasis:entry>
         <oasis:entry colname="col4">1 261 645 004</oasis:entry>
       </oasis:row>
     </oasis:tbody>
   </oasis:tgroup><?xmltex \end{scaleboxenv}?></oasis:table></table-wrap>

<sec id="Ch1.S4.SS1">
  <label>4.1</label><title>Parallel-working performance on a many-core architecture</title>
      <p id="d1e2894">The cycle time and speedup ratio of the hotspots are shown in Fig. <xref ref-type="fig" rid="Ch1.F8"/>, where different parallel methods are applied for comparison with the original method. While the CPEs parallel method (level 3 of four-level parallel framework), which introduces all 64 slave kernels to help with acceleration, is 12 times faster than the original method, we find the master–slave asynchronous parallelization mode (MPE–CPE multilevel parallel, level 2 of four-level parallel framework) shows quite satisfying results with a speedup of 65 times. The four-level parallel framework fits well in Sunway architecture, and the master–slave asynchronous parallelization strategy combines data parallelism with task parallelism. By making great use of the Sunway architecture, this parallelization strategy overlaps the computation part with the communication and I/O part, thus considerably improving hotspot efficiency.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F8"><?xmltex \currentcnt{8}?><?xmltex \def\figurename{Figure}?><label>Figure 8</label><caption><p id="d1e2901">Performance of MPE–CPE asynchronous parallelization.</p></caption>
          <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f08.png"/>

        </fig>

      <p id="d1e2910">The Sunway architecture suggests storing data in slave kernels before further computation. Therefore, the performance of the many-core architecture is related to the actual amount of memory access, the bandwidth of DMA and the FLOPS performance of slave kernels. We assume the total amount of time of the hotspot part in slave kernels is represented by <inline-formula><mml:math id="M126" display="inline"><mml:mrow><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="M127" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> is the time cost of the actual floating-point operation in this part, and <inline-formula><mml:math id="M128" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> is the time cost of direct memory access. To speed up the many-core architecture, we can seek methods for both DMA optimization and floating-point operation optimization.</p>
      <p id="d1e2970">For stencil computation, due to the low ratio of computation to memory access, we have <inline-formula><mml:math id="M129" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>&lt;</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula>. We can minimize <inline-formula><mml:math id="M130" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> using computation–communication overlap, yet <inline-formula><mml:math id="M131" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> is related to the actual amount of memory access and the bandwidth of DMA <inline-formula><mml:math id="M132" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>≥</mml:mo><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mi>M</mml:mi><mml:mrow><mml:msubsup><mml:mi mathvariant="normal">BW</mml:mi><mml:mi mathvariant="normal">DMA</mml:mi><mml:mo>′</mml:mo></mml:msubsup></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="M133" display="inline"><mml:mrow><mml:msub><mml:mi mathvariant="normal">BW</mml:mi><mml:mi mathvariant="normal">DMA</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> is the theoretical value of DMA bandwidth, and <inline-formula><mml:math id="M134" display="inline"><mml:mi>M</mml:mi></mml:math></inline-formula> is the total valid amount of memory access. In the real NEMO4 case, we have <inline-formula><mml:math id="M135" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>≪</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula>. Assuming the following equation holds true, <inline-formula><mml:math id="M136" display="inline"><mml:mrow><mml:msubsup><mml:mi mathvariant="normal">BW</mml:mi><mml:mi mathvariant="normal">DMA</mml:mi><mml:mo>′</mml:mo></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mi>M</mml:mi><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></inline-formula>, then the ratio of actual DMA bandwidth to theoretical bandwidth should be
            <disp-formula id="Ch1.E5" content-type="numbered"><label>5</label><mml:math id="M137" display="block"><mml:mrow><mml:mi mathvariant="italic">α</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mi>M</mml:mi><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:mfrac></mml:mstyle><mml:mrow><mml:msub><mml:mi mathvariant="normal">BW</mml:mi><mml:mi mathvariant="normal">DMA</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mi>M</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>)</mml:mo><mml:msub><mml:mi mathvariant="normal">BW</mml:mi><mml:mi mathvariant="normal">DMA</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
      <p id="d1e3210">The theoretical time cost of DMA when <inline-formula><mml:math id="M138" display="inline"><mml:mi>M</mml:mi></mml:math></inline-formula> is fixed is <inline-formula><mml:math id="M139" display="inline"><mml:mrow><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="false"><mml:mfrac style="text"><mml:mi>M</mml:mi><mml:mrow><mml:msub><mml:mi mathvariant="normal">BW</mml:mi><mml:mi mathvariant="normal">DMA</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></inline-formula>. Since <inline-formula><mml:math id="M140" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> is affected by the frequency and size of DMA, <inline-formula><mml:math id="M141" display="inline"><mml:mrow><mml:mi>T</mml:mi><mml:mo>&lt;</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M142" display="inline"><mml:mrow><mml:mi>T</mml:mi><mml:mo>&lt;</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula> hold. Therefore, we have
            <disp-formula id="Ch1.E6" content-type="numbered"><label>6</label><mml:math id="M143" display="block"><mml:mrow><mml:mi mathvariant="italic">α</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mi>M</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>)</mml:mo><mml:msub><mml:mi mathvariant="normal">BW</mml:mi><mml:mi mathvariant="normal">DMA</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mi>T</mml:mi><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>&lt;</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></disp-formula>
          when <inline-formula><mml:math id="M144" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>→</mml:mo><mml:mn mathvariant="normal">0</mml:mn></mml:mrow></mml:math></inline-formula> or <inline-formula><mml:math id="M145" display="inline"><mml:mrow><mml:msup><mml:msub><mml:mi>T</mml:mi><mml:mi mathvariant="normal">c</mml:mi></mml:msub><mml:mo>′</mml:mo></mml:msup><mml:mo>→</mml:mo><mml:mi>T</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M146" display="inline"><mml:mrow><mml:mi mathvariant="italic">α</mml:mi><mml:mo>→</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula>. The more the ratio of actual memory bandwidth to theoretical bandwidth approaches 1, the faster the whole architecture works and, thus, the better the performance becomes.</p>
      <p id="d1e3420">To analyze the performance of the algorithm mentioned in Sect. <xref ref-type="sec" rid="Ch1.S3.SS2.SSS1"/>, we use five kernels to simulate the test, with <inline-formula><mml:math id="M147" display="inline"><mml:mrow><mml:mn mathvariant="normal">49</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">65</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">128</mml:mn></mml:mrow></mml:math></inline-formula> grids on average per kernel. The five kernels are the most time-consuming chunks, which are only a fraction of the physical process. However, they are all specific implementations of Stencil computing. We calculate the average clock cycle of running one simulation, and the results are shown in Fig. <xref ref-type="fig" rid="Ch1.F9"/> (see the Appendix for the detailed codes of Algorithms A1–A5), where the red bars are the original results, and the blue bars are the optimized results using this algorithm. It is clearly shown that the time cost is largely reduced via this method, as with the third kernel, the clock cycle is reduced from <inline-formula><mml:math id="M148" display="inline"><mml:mrow><mml:mn mathvariant="normal">61</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">3</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> to <inline-formula><mml:math id="M149" display="inline"><mml:mrow><mml:mn mathvariant="normal">0.6</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn mathvariant="normal">10</mml:mn><mml:mrow><mml:mo>-</mml:mo><mml:mn mathvariant="normal">3</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> s, which means the result becomes 70.9 times faster. Meanwhile, according to the simulation test, this algorithm raises the ratio of actual DMA bandwidth to theoretical bandwidth up to above 92 %, as shown in the third kernel, and even reaches the theoretical limit of 95 % as a whole.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F9"><?xmltex \currentcnt{9}?><?xmltex \def\figurename{Figure}?><label>Figure 9</label><caption><p id="d1e3481">The performance of the five-kernel simulation test and the five kernels are shown in the Appendix.</p></caption>
          <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f09.png"/>

        </fig>

</sec>
<sec id="Ch1.S4.SS2">
  <label>4.2</label><title>Mixed-precision optimization</title>
      <p id="d1e3498">Introducing low-precision format data helps reduce memory use. Since the LDM in the slave kernel has limited storage, lowering the data precision could better use the given space and accelerate communication with the same actual DMA bandwidth. In terms of NEMO, due to its memory constraint, DMA data communication contributes to greater than 95 % of the total running time on the many-core architecture, and, as a result, we can enhance the overall performance by lowering the time cost of data communication. Focusing on tracer_fct in NEMO, we simulate a four-kernel test with a <inline-formula><mml:math id="M150" display="inline"><mml:mrow><mml:mn mathvariant="normal">49</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">65</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">128</mml:mn></mml:mrow></mml:math></inline-formula> s grid on average per kernel, and the results, which include the optimizations of four-level parallelization framework and mixed-precision approaches, are shown in Fig. <xref ref-type="fig" rid="Ch1.F10"/>, where the red and orange bars represent the average wall time of SP and SP <inline-formula><mml:math id="M151" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> HP in one simulation. Since NEMO is sensitive to precision, we cannot use HP only. We can see that the SP <inline-formula><mml:math id="M152" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> HP method shortens the total time by half compared to the original DP method. To further explore this topic, we simulate another test comparing the bandwidth usage and the ratio in both SP and SP <inline-formula><mml:math id="M153" display="inline"><mml:mo>+</mml:mo></mml:math></inline-formula> HP precision, which is shown in Fig. <xref ref-type="fig" rid="Ch1.F11"/>. The result shows that both methods surpass the 90 % ratio threshold, with the fourth kernel performing best with a ratio of 95 %.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F10"><?xmltex \currentcnt{10}?><?xmltex \def\figurename{Figure}?><label>Figure 10</label><caption><p id="d1e3545">Time cost of the tracer_fct process.</p></caption>
          <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f10.png"/>

        </fig>

      <?xmltex \floatpos{t}?><fig id="Ch1.F11"><?xmltex \currentcnt{11}?><?xmltex \def\figurename{Figure}?><label>Figure 11</label><caption><p id="d1e3556">Bandwidth ratio of the tracer_fct process.</p></caption>
          <?xmltex \igopts{width=241.848425pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f11.png"/>

        </fig>

</sec>
<sec id="Ch1.S4.SS3">
  <label>4.3</label><title>Strong scaling</title>
      <p id="d1e3573">Figure <xref ref-type="fig" rid="Ch1.F12"/> shows the results of strong scaling. We conduct experiments with resolutions of 2 km, 1 km and 500 m, increasing the number of cores from 2 129 920 to 27 988 480. The numbers of grids at resolutions of 2 km, 1 km and 500 m are 24 002 <inline-formula><mml:math id="M154" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 16 002 <inline-formula><mml:math id="M155" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 128, 43 502 <inline-formula><mml:math id="M156" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 29 002 <inline-formula><mml:math id="M157" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 128 and 82 502 <inline-formula><mml:math id="M158" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 55 002 <inline-formula><mml:math id="M159" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 128, respectively. We set 2 129 920 cores as the baseline of strong scaling, and the final parallel efficiencies are 74.18 %, 83.40 % and 99.29 %, respectively.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F12"><?xmltex \currentcnt{12}?><?xmltex \def\figurename{Figure}?><label>Figure 12</label><caption><p id="d1e3623">The strong scalability results, scaling from 2 129 920 to 27 988 480 cores with 2 km, 1 km and 500 m resolutions.</p></caption>
          <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f12.png"/>

        </fig>

      <p id="d1e3632">When the number of cores is 2 129 920, the average computing task of each process (i.e., CG) is approximately 325 <inline-formula><mml:math id="M160" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 432, including a halo in the
horizontal direction and 128 grids in the vertical direction. The mixed-precision optimization proposed in this paper considerably speeds up computations and reduces memory overhead. However, to guarantee the accuracy of the results, there are still some unavoidable double-precision floating-point arithmetics in the key steps in swNEMO_v4.0. During computing, the size of a four-dimensional variable with double precision has exceeded 1 GB. To maintain a reasonable efficiency of the memory system, we select 2 129 920 cores as the minimum available baseline.</p>
      <p id="d1e3643">For two-dimensional stencil computations, a grid with similar sizes of the <inline-formula><mml:math id="M161" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and <inline-formula><mml:math id="M162" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis can effectively reduce the amount of communication and speed up the stencil computation. When the grid sizes of the <inline-formula><mml:math id="M163" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and the <inline-formula><mml:math id="M164" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis are the same, the amount of communication touches the bottom. According to the four-level parallel architecture, the process-level task assignment does not involve grid partitioning in the vertical direction. For the strong scalability test, with a resolution of 500 m, the size of the horizontal grid is 82 502 <inline-formula><mml:math id="M165" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 55 002, and the size of the vertical grid is 128. When using 21 299 200 cores, the process division in the horizontal direction is 640 <inline-formula><mml:math id="M166" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 512. At this time, each process (i.e. CG) computes 128 <inline-formula><mml:math id="M167" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 108 grids, and the grid sizes of the <inline-formula><mml:math id="M168" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and the <inline-formula><mml:math id="M169" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis of each process are the closest in all options. Therefore, the best parallel efficiency, 104 %, is achieved when the number of cores is 21 299 200.</p>
      <p id="d1e3710">In summary, the strong scalability of the three resolutions has always maintained high performance with 10 million cores, and the speedup is still nearly linear at an ultra-large scale. Using 27 988 480 cores, swNEMO_v4.0 achieves up to 99.29 % parallel efficiency with a high resolution of 500 m.</p>
</sec>
<sec id="Ch1.S4.SS4">
  <label>4.4</label><title>Weak scaling</title>
      <p id="d1e3721">The choice of resolution of the GYRE needs to be followed with strict discipline. Therefore, to ensure that the workload in a single process is fixed, the number of grids of the <inline-formula><mml:math id="M170" display="inline"><mml:mi>x</mml:mi></mml:math></inline-formula> axis and <inline-formula><mml:math id="M171" display="inline"><mml:mi>y</mml:mi></mml:math></inline-formula> axis is proportional to the number of cores, and the automatic scheme of domain decomposition in swNEMO_v4.0 is replaced with a manual scheme at the same time, which is shown in Fig. <xref ref-type="fig" rid="Ch1.F13"/> and Table <xref ref-type="table" rid="Ch1.T3"/>.</p>
      <p id="d1e3742">As we build roofline models on NEMO, we find that when the horizontal grid surpasses <inline-formula><mml:math id="M172" display="inline"><mml:mrow><mml:mn mathvariant="normal">49</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">65</mml:mn></mml:mrow></mml:math></inline-formula> and the vertical grid surpasses 128, the computation efficiency of the floating point for a single core group performs the best. Therefore, our following experiments on weak scaling are conducted on a single core group with <inline-formula><mml:math id="M173" display="inline"><mml:mrow><mml:mn mathvariant="normal">49</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">65</mml:mn><mml:mo>×</mml:mo><mml:mn mathvariant="normal">128</mml:mn></mml:mrow></mml:math></inline-formula> grid size each. Based on the above-mentioned principle, we choose resolutions of 9, 7, 4.5, 3.5, 2.5, 2.0, 1.2 and 1.0 km. According to the expansion of the workload of each process (i.e., CG), the number of cores increases from 299 520 to 27 988 480. With a resolution of 1.0 km, the total number of grids is 43 502 <inline-formula><mml:math id="M174" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 29 002 <inline-formula><mml:math id="M175" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 128. As shown in Fig. <xref ref-type="fig" rid="Ch1.F13"/>, the performance is stable with different resolutions, and the computation efficiency of the floating point is still 1.99 ‰ at a resolution of 1 km, which is very close to the baseline. The nearly linear trend indicates that the model has good weak scalability.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F13"><?xmltex \currentcnt{13}?><?xmltex \def\figurename{Figure}?><label>Figure 13</label><caption><p id="d1e3791">The weak-scalability results, scaling from 299 520 to 27 988 480 cores with eight different resolutions.</p></caption>
          <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f13.png"/>

        </fig>

</sec>
<sec id="Ch1.S4.SS5">
  <label>4.5</label><title>Peak performance</title>
      <p id="d1e3808">Figure <xref ref-type="fig" rid="Ch1.F14"/> shows the peak performance of swNEMO_v4.0 with a resolution of 1 km. When the total grid size is 43 502 <inline-formula><mml:math id="M176" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 29 002 <inline-formula><mml:math id="M177" display="inline"><mml:mo>×</mml:mo></mml:math></inline-formula> 128, the number of cores increases from 2 129 920 to 27 988 480. It is obvious that the optimal performance is 1.97 PFLOPS using 27 988 480 cores, and the performance of the optimized version is 10.25 times faster than that of the original version.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F14"><?xmltex \currentcnt{14}?><?xmltex \def\figurename{Figure}?><label>Figure 14</label><caption><p id="d1e3829">The peak performance results for simulations with 1 km resolution. Numbers along the graph line are the peak performances for simulations with different cores. The peak performance with 27 988 480 cores reaches 1.973 PFLOPS.</p></caption>
          <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f14.png"/>

        </fig>

      <p id="d1e3838">In swNEMO_v4.0, we fully parallelize 70.58 % of the hotspots according to the performance profiling tools. The remaining 29.42 % of hotspots are mainly serial, which can hardly be parallelized. Therefore, mixed-precision optimization is used to optimize the serial part. According to Amdahl's law,
            <disp-formula id="Ch1.E7" content-type="numbered"><label>7</label><mml:math id="M178" display="block"><mml:mrow><mml:mi mathvariant="normal">Speedup</mml:mi><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mo>=</mml:mo><mml:mspace width="0.125em" linebreak="nobreak"/><mml:mstyle displaystyle="true"><mml:mfrac style="display"><mml:mn mathvariant="normal">1</mml:mn><mml:mrow><mml:mo>(</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>-</mml:mo><mml:mi>P</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>P</mml:mi><mml:mo>/</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mstyle><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
          where the theoretical speedup is approximately by a factor of 6.8.</p>
      <p id="d1e3879">Combining the breakthroughs in an adaptive four-level parallelization design, CPE cluster optimization and mixed-precision optimization, the performance of our optimized version surpasses the theoretical values after further refactoring the code to reduce conditional judgment, instruction branching and the complexity of the code.</p>
      <p id="d1e3882">Additionally, as shown in Fig. <xref ref-type="fig" rid="Ch1.F15"/>, 1.43 SYPD (simulated years per day) with a resolution of 1 km using 27 988 480 cores is achieved, which is 7.53 times better compared with 0.19 SYPD of the original version, exceeding the ideal-performance upper bound.</p>

      <?xmltex \floatpos{t}?><fig id="Ch1.F15"><?xmltex \currentcnt{15}?><?xmltex \def\figurename{Figure}?><label>Figure 15</label><caption><p id="d1e3889">The computing throughput results using 27 988 480 cores. The SYPD (simulated years per day) increases from 0.19 to 1.43 with the 1 km resolution. The optimized performance is 7.53 times better than that of the original version.</p></caption>
          <?xmltex \igopts{width=236.157874pt}?><graphic xlink:href="https://gmd.copernicus.org/articles/15/5739/2022/gmd-15-5739-2022-f15.png"/>

        </fig>

</sec>
</sec>
<sec id="Ch1.S5" sec-type="conclusions">
  <label>5</label><title>Conclusions and discussions</title>
      <p id="d1e3908">This paper presents a successful solution for global ocean simulations with ultrahigh resolution using NEMO on a new-generation Sunway supercomputer. Three breakthroughs, including an adaptive four-level parallelization design, many-core optimization and mixed-precision optimization, are designed and tested. The simulations achieve 71.48 %, 83.40 % and 99.29 % parallel efficiency with resolutions of 2 km, 1 km and 500 m using 27 988 480 cores, respectively.</p>
      <p id="d1e3911">Current resolutions of ocean models in ocean forecasting systems and climate models cannot resolve the mesoscale and sub-mesoscale eddies in the real ocean well, not to mention the other smaller-scale processes such as internal waves. However, these sub-mesoscale and small-scale processes are not only important to navigation safety in themselves, but also have notable influences on the large-scale simulations of the global/regional ocean and circulations through interacting with different scales. Improving resolution is one of the best ways to resolve these processes and improve simulations, forecasts and predictions. The highest resolution used in this study is 500 m, and this resolution can resolve the sub-mesoscale eddies well and partly resolve the internal waves, which are very important for the safety of offshore structures. Therefore, the breakthroughs made in this study make the direct simulations of these important sub-mesoscale and small-scale processes at the global scale possible. This will substantially improve the forecast and prediction accuracy of the ocean and climate while strongly supporting the key outcome of a predicted ocean of the UN Ocean Decade.</p>
      <p id="d1e3914">This study is conducted in the new generation of Sunway supercomputers. The breakthroughs in this paper, such as the four-level parallelization design and the method for efficient data transportation inside CPEs, provide novel ideas for other applications in this series of Sunway supercomputers. Moreover, the proposed new optimization approaches, such as a four-level parallel framework with longitude–latitude–depth decomposition, a multi-level mixed-precision optimization method that uses half, single and double precision, are the methods of general applicability. We test these optimization approaches in the NEMO, but these can be incorporated into other global/regional ocean general circulation models (e.g., MOM, POP and ROMS). Moreover, the optimizations on the stencil computation can be applied to any model with stencil computations.</p>
      <p id="d1e3917">High-resolution ocean simulations are crucial for navigation safety, weather forecasting and global climate change prediction. We believe that these approaches are also suitable for other OGCMs and supercomputers. In this paper, three innovative algorithms proposed based on the new generation of Sunway supercomputers provide an important reference for ultrahigh-resolution ocean circulation forecasting. However, only benchmarks are tested in the work, and real application data have not been used. In the future, we will build an ultrahigh-resolution ocean circulation forecasting model under real scenarios and conduct in-depth research to provide efficient solutions to predict ocean circulation and climate change accurately.</p>
      <p id="d1e3921">From the view of software and hardware co-design, the following should be focused on in the future.
The first is the decomposition and load balance. For the model design, we should find the proper decomposition scheme to utilize the computer architecture fully. Besides the time dimension, solving an ocean general circulation model is a 3-dimensional problem, with longitude, latitude and depth. Usually, only the longitude–latitude domain is decomposed. In our work, driven by the RMA technology, we achieved the decomposition of the longitude–latitude–depth domain, which enables better large-scale scalability. Meanwhile, keeping a good load balance is also important for scalability. For the computer design, the RMA technology is a good example, which enables the decomposition of the longitude–latitude–depth domain. In other words, the high communication bandwidth between different cores or nodes will help achieve large-scale scalability.
The second is communications. With the increasing processes used for model simulation, the ratio of communication time to computational time will increase. For the model design, the first thing is to avoid the global operator, such as ALLREDUCE and BCAST, which will take more time with an increase in processes. Otherwise, it will be the crucial bottleneck. Meanwhile, we also should pack the exchanged data between different processes as much as possible. For the computer design, the low latency will help in saving the communication time.
The third is reduced precision. Our work's results demonstrate that there is a great potential to save computational time by incorporating the mixed double, single and half precision into the model. For the model design, we should understand the minimum computational precision requirements essential for successful ocean simulations and then revise or develop arithmetic. For the computer design, the support for half precision should be considered in the future.</p>
      <p id="d1e3924">Overall, the above are only several examples for further improving the performance of ocean modeling from perspectives of model development and computer design. Furthermore, other aspects such as I/O efficiency and the trade-off between precision and energy consumption should also be considered. Moreover, it should be noted that these suggestions are from different aspects of the model and computer development and need to be considered based on the software and hardware co-design ideology.</p>
</sec>

      
      </body>
    <back><app-group>

<app id="App1.Ch1.S1">
  <?xmltex \currentcnt{A}?><label>Appendix A</label><title>Supplementary code in Fortran format</title>
      <p id="d1e3938">The following five chunks of code correspond to the five kernels stated in our previous experimental section. The codes listed below are exactly from NEMO for experimental validation.</p><?xmltex \hack{\clearpage}?><?xmltex \floatpos{htbp}?><boxed-text content-type="algorithm" position="float" id="App1.Ch1.S1.Prog1" specific-use="star"><?xmltex \currentcnt{A1}?><label>Algorithm A1</label><caption><p id="d1e3943"> </p></caption><disp-quote content-type="algorithmic" specific-use="numbering{0}"><list>

    <list-item>

      <p id="d1e3950" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M179" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e3983" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M180" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4017" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold> <inline-formula><mml:math id="M181" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4064" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M182" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>u</mml:mi><mml:mo>=</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mo>(</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4202" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M183" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1.0</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4297" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M184" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi><mml:mo>=</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mo>(</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4435" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M185" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1.0</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4529" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M186" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4625" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M187" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4723" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M188" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4819" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M189" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e4918" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M190" display="inline"><mml:mrow><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5048" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M191" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5142" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e5150" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e5158" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>
        </list></disp-quote></boxed-text><?xmltex \floatpos{htbp}?><boxed-text content-type="algorithm" position="float" id="App1.Ch1.S1.Prog2" specific-use="star"><?xmltex \currentcnt{A2}?><label>Algorithm A2</label><caption><p id="d1e5165"> </p></caption><disp-quote content-type="algorithmic" specific-use="numbering{0}"><list>

    <list-item>

      <p id="d1e5172" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M192" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5205" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M193" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5252" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><inline-formula><mml:math id="M194" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>t</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>d</mml:mi><mml:mi>t</mml:mi><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>t</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5415" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e5423" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e5431" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M195" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5464" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M196" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5498" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M197" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5545" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M198" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>r</mml:mi><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>t</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>n</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>d</mml:mi><mml:mi>t</mml:mi><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>t</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5706" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M199" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>z</mml:mi><mml:mi>r</mml:mi><mml:mi>h</mml:mi><mml:mi>s</mml:mi><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5850" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e5859" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e5867" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>
        </list></disp-quote></boxed-text><?xmltex \floatpos{htbp}?><boxed-text content-type="algorithm" position="float" id="App1.Ch1.S1.Prog3" specific-use="star"><?xmltex \currentcnt{A3}?><label>Algorithm A3</label><caption><p id="d1e5874"> </p></caption><disp-quote content-type="algorithmic" specific-use="numbering{0}"><list>

    <list-item>

      <p id="d1e5881" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M200" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5914" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M201" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e5961" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><inline-formula><mml:math id="M202" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>d</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6024" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e6032" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e6040" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M203" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6073" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M204" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6107" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M205" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6154" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M206" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>d</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6317" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e6325" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e6334" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e6341" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M207" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6374" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M208" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6408" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M209" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6455" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M210" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>d</mml:mi><mml:mi>t</mml:mi><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>w</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6568" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M211" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>d</mml:mi><mml:mi>t</mml:mi><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>w</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6690" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M212" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>d</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>t</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6820" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e6828" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e6836" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>
        </list></disp-quote></boxed-text><?xmltex \hack{\clearpage}?><?xmltex \floatpos{p}?><boxed-text content-type="algorithm" position="float" id="App1.Ch1.S1.Prog4" specific-use="star"><?xmltex \currentcnt{A4}?><label>Algorithm A4</label><caption><p id="d1e6844"> </p></caption><disp-quote content-type="algorithmic" specific-use="numbering{0}"><list>

    <list-item>

      <p id="d1e6851" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M213" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6884" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M214" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6918" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M215" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e6959" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M216" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7099" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M217" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>j</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7240" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e7248" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e7256" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e7263" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M218" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7296" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><inline-formula><mml:math id="M219" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7419" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>IF</bold> <inline-formula><mml:math id="M220" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula>  <bold>THEN</bold>  <inline-formula><mml:math id="M221" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7494" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>ELSE</bold></p>
          </list-item>

    <list-item>

      <p id="d1e7502" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M222" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7532" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><bold>DO</bold>  <inline-formula><mml:math id="M223" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

            <?xmltex \hack{\hspace*{6mm}}?>

      <p id="d1e7564" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><inline-formula><mml:math id="M224" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7698" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e7706" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e7715" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END IF</bold></p>
          </list-item>

    <list-item>

      <p id="d1e7723" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M225" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7757" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M226" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7798" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M227" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mo>=</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi mathvariant="italic">_</mml:mi><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>u</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e7913" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M228" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>=</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi mathvariant="italic">_</mml:mi><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>v</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8028" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M229" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>u</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>.</mml:mo><mml:mo>/</mml:mo><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mo>(</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8140" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M230" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>.</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8236" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M231" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>.</mml:mo><mml:mo>/</mml:mo><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mo>(</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8347" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M232" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>.</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8443" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M233" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>u</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8564" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M234" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>v</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8685" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M235" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8772" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M236" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mo>*</mml:mo><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8856" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M237" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e8963" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M238" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>j</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9049" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M239" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>*</mml:mo><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9133" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M240" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>k</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9240" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e9249" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e9257" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M241" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9291" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M242" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9338" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M243" display="inline"><mml:mrow><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>n</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>n</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>n</mml:mi><mml:mo>*</mml:mo><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9499" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M244" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9581" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M245" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>*</mml:mo><mml:mi>r</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi mathvariant="italic">_</mml:mi><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>t</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9661" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e9669" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e9677" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>
        </list></disp-quote></boxed-text><?xmltex \hack{\clearpage}?><?xmltex \floatpos{htbp}?><boxed-text content-type="algorithm" position="float" id="App1.Ch1.S1.Prog5" specific-use="star"><?xmltex \currentcnt{A5}?><label>Algorithm A5</label><caption><p id="d1e9686"> </p></caption><disp-quote content-type="algorithmic" specific-use="numbering{0}"><list>

    <list-item>

      <p id="d1e9693" specific-use="STATE"><bold>DO</bold> <inline-formula><mml:math id="M246" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9726" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>DO</bold> <inline-formula><mml:math id="M247" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9760" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>DO</bold>   <inline-formula><mml:math id="M248" display="inline"><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mi>s</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>j</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mn mathvariant="normal">1</mml:mn></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9807" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M249" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>u</mml:mi><mml:mo>=</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mo>(</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e9945" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M250" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1.0</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10040" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M251" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi><mml:mo>=</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mo>(</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10178" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M252" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>v</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>,</mml:mo><mml:mn mathvariant="normal">1.0</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10272" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M253" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10368" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M254" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10466" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M255" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>=</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10562" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M256" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10661" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M257" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>u</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>u</mml:mi><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10770" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M258" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">4</mml:mn><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>z</mml:mi><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi>v</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mo>*</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>z</mml:mi><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>v</mml:mi><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mi>j</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e10879" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M259" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>f</mml:mi><mml:mi>w</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mo>*</mml:mo><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e11007" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M260" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e11093" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M261" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mn mathvariant="normal">4</mml:mn><mml:mo>*</mml:mo><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>j</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>j</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e11193" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M262" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>j</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>z</mml:mi><mml:mi>d</mml:mi><mml:mi>j</mml:mi><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e11280" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M263" display="inline"><mml:mrow><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>f</mml:mi><mml:mi>w</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mi>z</mml:mi><mml:mi>t</mml:mi><mml:mi>f</mml:mi><mml:mi>w</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>+</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">1</mml:mn><mml:mi>e</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mi>t</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>)</mml:mo><mml:mo>/</mml:mo><mml:mi>e</mml:mi><mml:mn mathvariant="normal">3</mml:mn><mml:mi>w</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>n</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>*</mml:mo><mml:mi>w</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e11445" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M264" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>*</mml:mo><mml:mo>(</mml:mo><mml:mi>a</mml:mi><mml:mi>h</mml:mi><mml:mi mathvariant="italic">_</mml:mi><mml:mi>w</mml:mi><mml:mi>s</mml:mi><mml:mi>l</mml:mi><mml:mi>p</mml:mi><mml:mn mathvariant="normal">2</mml:mn><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>a</mml:mi><mml:mi>k</mml:mi><mml:mi>z</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e11531" specific-use="STATE"><?xmltex \hack{\hspace*{9mm}}?><inline-formula><mml:math id="M265" display="inline"><mml:mrow><mml:mi mathvariant="italic">&amp;</mml:mi><mml:mo>*</mml:mo><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn mathvariant="normal">1</mml:mn><mml:mo>)</mml:mo><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>b</mml:mi><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p>
          </list-item>

    <list-item>

      <p id="d1e11608" specific-use="STATE"><?xmltex \hack{\hspace*{6mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e11616" specific-use="STATE"><?xmltex \hack{\hspace*{3mm}}?><bold>END DO</bold></p>
          </list-item>

    <list-item>

      <p id="d1e11624" specific-use="STATE"><bold>END DO</bold></p>
          </list-item>
        </list></disp-quote></boxed-text><?xmltex \hack{\clearpage}?>
</app>
  </app-group><notes notes-type="codeavailability"><title>Code availability</title>

      <p id="d1e11635">The swNEMO_v4.0 source codes with the user manual are available at <uri>https://doi.org/10.5281/zenodo.5976033</uri> <xref ref-type="bibr" rid="bib1.bibx30" id="paren.31"/>.</p>
  </notes><notes notes-type="dataavailability"><title>Data availability</title>

      <p id="d1e11647">Processed data used in the production of figures in this paper are available via <uri>https://doi.org/10.5281/zenodo.6834799</uri> <xref ref-type="bibr" rid="bib1.bibx31" id="paren.32"/>.</p>
  </notes><notes notes-type="authorcontribution"><title>Author contributions</title>

      <p id="d1e11659">YY, ZS and SZ were involved in methodology, coding, numerical experiment, analysis and writing original draft preparation. YL, QS and BW were involved in algorithm description and validation. YL and WL were involved in the suggestion of the algorithm and numerical experiments. FQ and LW carried out the conceptualization, supervision, funding acquisition and editing. All authors discussed, read, edited and approved the article. All authors have read and agreed to the published version of the paper.</p>
  </notes><notes notes-type="competinginterests"><title>Competing interests</title>

      <p id="d1e11665">The contact author has declared that none of the authors has any competing interests.</p>
  </notes><notes notes-type="disclaimer"><title>Disclaimer</title>

      <p id="d1e11671">Publisher’s note: Copernicus Publications remains neutral with regard to jurisdictional claims in published maps and institutional affiliations.</p>
  </notes><ack><title>Acknowledgements</title><p id="d1e11677">The authors would like to thank the two anonymous reviewers and the editor,  for their constructive comments. All the numerical simulations were carried out at the Qingdao Supercomputing and Big Data Center.</p></ack><notes notes-type="financialsupport"><title>Financial support</title>

      <p id="d1e11682">This work is supported by the National Natural Science Foundation of China (grant nos. 41821004, U1806205, 42022042), the Marine S&amp;T Fund of Shandong Province for Pilot National Laboratory for Marine Science and Technology (Qingdao) (grant no. 2022QNLM010202),  the China–Korea Cooperation Project on Northwest Pacific Marine Ecosystem Simulation under the Climate Change and the CAS Interdisciplinary Innovation Team (grant no. JCTD-2020-12).</p>
  </notes><notes notes-type="reviewstatement"><title>Review statement</title>

      <p id="d1e11688">This paper was edited by Xiaomeng Huang and reviewed by two anonymous referees.</p>
  </notes><ref-list>
    <title>References</title>

      <ref id="bib1.bibx1"><?xmltex \def\ref@label{Afzal et al.(2017)}?><label>Afzal et al.(2017)</label><?label AA2017?><mixed-citation>Afzal, A., Ansari, Z., Faizabadi, A. R., and Ramis, M.: Parallelization strategies for computational fluid dynamics software: state of the art review, Arch. Computat. Methods Eng.,  24,  337–363, <ext-link xlink:href="https://doi.org/10.1007/s11831-016-9165-4" ext-link-type="DOI">10.1007/s11831-016-9165-4</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bibx2"><?xmltex \def\ref@label{Aumont et al.(2015)}?><label>Aumont et al.(2015)</label><?label AO2015?><mixed-citation>Aumont, O., Ethé, C., Tagliabue, A., Bopp, L., and Gehlen, M.: PISCES-v2: an ocean biogeochemical model for carbon and ecosystem studies, Geosci. Model Dev., 8, 2465–2513, <ext-link xlink:href="https://doi.org/10.5194/gmd-8-2465-2015" ext-link-type="DOI">10.5194/gmd-8-2465-2015</ext-link>, 2015.</mixed-citation></ref>
      <ref id="bib1.bibx3"><?xmltex \def\ref@label{Baker et al.(2016)}?><label>Baker et al.(2016)</label><?label AB2016?><mixed-citation>Baker, A. H., Hu, Y., Hammerling, D. M., Tseng, Y.-H., Xu, H., Huang, X., Bryan, F. O., and Yang, G.: Evaluating statistical consistency in the ocean model component of the Community Earth System Model (pyCECT v2.0), Geosci. Model Dev., 9, 2391–2406, <ext-link xlink:href="https://doi.org/10.5194/gmd-9-2391-2016" ext-link-type="DOI">10.5194/gmd-9-2391-2016</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx4"><?xmltex \def\ref@label{Bryan(1969)}?><label>Bryan(1969)</label><?label KB1969?><mixed-citation>
Bryan, K.: A numerical method for the study of the circulation of the world ocean, J. Comput. Phys., 4,  347–376, 1969.</mixed-citation></ref>
      <ref id="bib1.bibx5"><?xmltex \def\ref@label{Bryan et al.(1967)}?><label>Bryan et al.(1967)</label><?label KB1967?><mixed-citation>Bryan, K. and Cox, M. D.: A numerical investigation of the oceanic general circulation, Tellus,  19,  54–80, <ext-link xlink:href="https://doi.org/10.1111/j.2153-3490.1967.tb01459.x" ext-link-type="DOI">10.1111/j.2153-3490.1967.tb01459.x</ext-link>, 1967.</mixed-citation></ref>
      <ref id="bib1.bibx6"><?xmltex \def\ref@label{Chassignet et al.(2019)}?><label>Chassignet et al.(2019)</label><?label EPC2019?><mixed-citation>Chassignet, E. P., LeSommer, J., and Wallcraft, A. J.: Generalcirculation models, in: Encyclopedia of Ocean Sciences, 3rd Edn., edited by: Cochran, K. J., Bokuniewicz, H. J., and Yager, P. L., Elsevier, 5, 486–490, <ext-link xlink:href="https://doi.org/10.1016/B978-0-12-409548-9.11410-1" ext-link-type="DOI">10.1016/B978-0-12-409548-9.11410-1</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bibx7"><?xmltex \def\ref@label{Dawson and Düben(2017)}?><label>Dawson and Düben(2017)</label><?label AD2017?><mixed-citation>Dawson, A. and Düben, P. D.: rpe v5: an emulator for reduced floating-point precision in large numerical simulations, Geosci. Model Dev., 10, 2221–2230, <ext-link xlink:href="https://doi.org/10.5194/gmd-10-2221-2017" ext-link-type="DOI">10.5194/gmd-10-2221-2017</ext-link>, 2017.</mixed-citation></ref>
      <ref id="bib1.bibx8"><?xmltex \def\ref@label{Dong et al.(2020)}?><label>Dong et al.(2020)</label><?label JD2020?><mixed-citation>Dong, J., Fox-Kemper, B., Zhang, H., and Dong, C.: The seasonality of submesoscale energy production, content, and cascade, Geophys. Res. Lett., 7, e2020GL087388, <ext-link xlink:href="https://doi.org/10.1029/2020GL087388" ext-link-type="DOI">10.1029/2020GL087388</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bibx9"><?xmltex \def\ref@label{Eyring et al.(2016)}?><label>Eyring et al.(2016)</label><?label VE2016?><mixed-citation>Eyring, V., Bony, S., Meehl, G. A., Senior, C. A., Stevens, B., Stouffer, R. J., and Taylor, K. E.: Overview of the Coupled Model Intercomparison Project Phase 6 (CMIP6) experimental design and organization, Geosci. Model Dev., 9, 1937–1958, <ext-link xlink:href="https://doi.org/10.5194/gmd-9-1937-2016" ext-link-type="DOI">10.5194/gmd-9-1937-2016</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx10"><?xmltex \def\ref@label{Fu et al.(2016)}?><label>Fu et al.(2016)</label><?label HF20161?><mixed-citation>Fu, H., Liao, J., Yang, J., Wang, L., Song, Z., Huang, X., Yang, C., Xue, W., Liu, F., Qiao, F., Zhao, W., Yin, X., Hou, C., Zhang, C., Ge, W., Zhang, J., Wang, Y., Zhou, C., and Yang, G.: The sunway taihulight supercomputer: system and applications, Sci. China Inform. Sci., 59,  1–16, <ext-link xlink:href="https://doi.org/10.1007/s11432-016-5588-7" ext-link-type="DOI">10.1007/s11432-016-5588-7</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx11"><?xmltex \def\ref@label{Gu et al.(2022)}?><label>Gu et al.(2022)</label><?label JG2022?><mixed-citation>Gu, J., Feng, F., Hao, X., Fang, T., Zhao, C., An, H., Chen, J., Xu, M., Li, J., Han, W., Yang, C., Li, F., and Chen, D.: Establishing a non-hydrostatic global atmospheric modeling system at 3-km horizontal resolution with aerosol feedbacks on the Sunway supercomputer of China, Sci. Bull., 67, 1170–1181, <ext-link xlink:href="https://doi.org/10.1016/j.scib.2022.03.009" ext-link-type="DOI">10.1016/j.scib.2022.03.009</ext-link>, 2022.</mixed-citation></ref>
      <ref id="bib1.bibx12"><?xmltex \def\ref@label{Hu et al.(2013)}?><label>Hu et al.(2013)</label><?label YH2013?><mixed-citation>Hu, Y., Huang, X., Wang,  X., Fu,  H., Xu, S., Ruan, H., Xue, W., and Yang, G.: A scalable barotropic mode solver for the parallel ocean program, in: European Conference on Parallel Processing, edited by: Wolf, F., Mohr, B., and Mey, D., Springer,  739–750, <ext-link xlink:href="https://doi.org/10.1007/978-3-642-40047-6_74" ext-link-type="DOI">10.1007/978-3-642-40047-6_74</ext-link>, 2013.</mixed-citation></ref>
      <ref id="bib1.bibx13"><?xmltex \def\ref@label{Jiang et al.(2019)}?><label>Jiang et al.(2019)</label><?label JJ2019?><mixed-citation>Jiang, J., Lin, P., Wang, J., Liu, H., Chi, X., Hao, H., Wang, Y., Wang, W., and Zhang, L.: Porting LASG/IAP climate system ocean model to GPUs using OpenAcc, IEEE Access, 7, 154490–154501, <ext-link xlink:href="https://doi.org/10.1109/ACCESS.2019.2932443" ext-link-type="DOI">10.1109/ACCESS.2019.2932443</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bibx14"><?xmltex \def\ref@label{Lellouche et al.(2018)}?><label>Lellouche et al.(2018)</label><?label JML2018?><mixed-citation>Lellouche, J.-M., Greiner, E., Le Galloudec, O., Garric, G., Regnier, C., Drevillon, M., Benkiran, M., Testut, C.-E., Bourdalle-Badie, R., Gasparin, F., Hernandez, O., Levier, B., Drillet, Y., Remy, E., and Le Traon, P.-Y.: Recent updates to the Copernicus Marine Service global ocean monitoring and forecasting real-time <inline-formula><mml:math id="M266" display="inline"><mml:mrow><mml:mn mathvariant="normal">1</mml:mn><mml:mo>/</mml:mo><mml:mn mathvariant="normal">12</mml:mn></mml:mrow></mml:math></inline-formula><inline-formula><mml:math id="M267" display="inline"><mml:msup><mml:mi/><mml:mo>∘</mml:mo></mml:msup></mml:math></inline-formula> high-resolution system, Ocean Sci., 14, 1093–1126, <ext-link xlink:href="https://doi.org/10.5194/os-14-1093-2018" ext-link-type="DOI">10.5194/os-14-1093-2018</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bibx15"><?xmltex \def\ref@label{Liao et al.(2014)}?><label>Liao et al.(2014)</label><?label XL2014?><mixed-citation>Liao, X., Xiao, L., Yang, C., and Lu, Y.: Milkyway-2 supercomputer: system and application, Front. Comput. Sci., 8,  345–356, <ext-link xlink:href="https://doi.org/10.1007/s11704-014-3501-3" ext-link-type="DOI">10.1007/s11704-014-3501-3</ext-link>, 2014.</mixed-citation></ref>
      <ref id="bib1.bibx16"><?xmltex \def\ref@label{Madec and the NEMO team(2016)}?><label>Madec and the NEMO team(2016)</label><?label GM2016?><mixed-citation>Madec, G. and the NEMO team: NEMO ocean engine, Zenodo, <ext-link xlink:href="https://doi.org/10.5281/zenodo.3248739" ext-link-type="DOI">10.5281/zenodo.3248739</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx17"><?xmltex \def\ref@label{Putnam et al.(2015)}?><label>Putnam et al.(2015)</label><?label AP2014?><mixed-citation>Putnam, A., Caulfield, A. M., Chung, E. S., Chiou, D., Constantinides, K., Demme, J., Esmaeilzadeh, H., Fowers, J., Gopal, G. P., Gray, J., Haselman, M., Hauck, S., Heil, S., Hormati, A., Kim, J.-Y., Lanka, S., Larus, J., Peterson, E., Pope, S., Smith, A., Thong, J., Xiao, P. Y., and Burger, D.: A reconfigurable fabric for accelerating large-scale datacenter services, IEEE Micro., 35, 10–22, <ext-link xlink:href="https://doi.org/10.1109/MM.2015.42" ext-link-type="DOI">10.1109/MM.2015.42</ext-link>, 2015.</mixed-citation></ref>
      <ref id="bib1.bibx18"><?xmltex \def\ref@label{Qiu et al.(2018)}?><label>Qiu et al.(2018)</label><?label BQ2018?><mixed-citation>Qiu, B., Chen, S., Klein, P., Wang, J., Torres, H., Fu, L. L., and Menemenlis, D.: Seasonality in transition scale from balanced to unbalanced motions in the world ocean, J. Phys. Oceanogr.,  48,  591–605, <ext-link xlink:href="https://doi.org/10.1175/JPO-D-17-0169.1" ext-link-type="DOI">10.1175/JPO-D-17-0169.1</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bibx19"><?xmltex \def\ref@label{Qiu et al.(2020)}?><label>Qiu et al.(2020)</label><?label BQ2020?><mixed-citation>Qiu, B., Chen, S., Klein, P., Torres, H., Wang, J., Fu, L. L., and Menemenlis, D.: Reconstructing upper-ocean vertical velocity field from sea surface height in the presence of unbalanced motion, J. Phys. Oceanogr.,  50, 55–79, <ext-link xlink:href="https://doi.org/10.1175/JPO-D-19-0172.1" ext-link-type="DOI">10.1175/JPO-D-19-0172.1</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bibx20"><?xmltex \def\ref@label{Rocha et al.(2016)}?><label>Rocha et al.(2016)</label><?label CBR2016?><mixed-citation>Rocha, C. B., Gille, S. T., Chereskin, T. K., and Menemenlis, D.: Seasonality of submesoscale dynamics in the kuroshio extension, Geophys. Res. Lett., 43, 11–304, <ext-link xlink:href="https://doi.org/10.1002/2016GL071349" ext-link-type="DOI">10.1002/2016GL071349</ext-link>, 2016.</mixed-citation></ref>
      <ref id="bib1.bibx21"><?xmltex \def\ref@label{Ruston(2019)}?><label>Ruston(2019)</label><?label BR2019?><mixed-citation>Ruston, B.: Validation test report for the dtic, <uri>https://apps.dtic.mil/sti/pdfs/AD1090615.pdf</uri> (last access: 13 July 2022), 2019.</mixed-citation></ref>
      <ref id="bib1.bibx22"><?xmltex \def\ref@label{Smith et al.(2010)}?><label>Smith et al.(2010)</label><?label RS2010?><mixed-citation>Smith, R., Jones, P., Briegleb, B.,Bryan, F., Danabasoglu, G., Dennis, J., Dukowicz, J., Eden, C., Fox-Kemper, B., Gent, P., Hecht, M., Jayne, S., Jochum, M., Large, W., Lindsay, K., Maltrud, M., Norton, N., Peacock, S., Vertenstein, M., and Yeager, S.: The parallel ocean program (pop) reference manual ocean component of the community climate system model (ccsm) and community earth system model (cesm), LAUR-01853,  141,  1–140, <uri>https://www.cesm.ucar.edu/models/cesm1.0/pop2/doc/sci/POPRefManual.pdf</uri> (last access: 13 July 2022), 2010.</mixed-citation></ref>
      <ref id="bib1.bibx23"><?xmltex \def\ref@label{Tintó Prims et al.(2019)}?><label>Tintó Prims et al.(2019)</label><?label OTP2019?><mixed-citation>Tintó Prims, O., Acosta, M. C., Moore, A. M., Castrillo, M., Serradell, K., Cortés, A., and Doblas-Reyes, F. J.: How to use mixed precision in ocean models: exploring a potential reduction of numerical precision in NEMO 4.0 and ROMS 3.6, Geosci. Model Dev., 12, 3135–3148, <ext-link xlink:href="https://doi.org/10.5194/gmd-12-3135-2019" ext-link-type="DOI">10.5194/gmd-12-3135-2019</ext-link>, 2019.</mixed-citation></ref>
      <ref id="bib1.bibx24"><?xmltex \def\ref@label{Vazhkudai et al.(2018)}?><label>Vazhkudai et al.(2018)</label><?label SSV2018?><mixed-citation>Vazhkudai, S. S., deSupinski, B. R., Bland,  A. S., Geist, A., Sexton, J., Kahle, J., Zimmer, C. J., Atchley, S., Oral, S., Maxwell, D. E., Vergara Larrea, V. G., Bertsch, A., Goldstone, R., Joubert, W., Chambreau, C., Appelhans, D., Blackmore, R., Casses, B., Chochia, G., Davision, G., Ezell, M. A., Gooding, T., Gonsiorowski, E., Grinberg, L., Hanson, B., Hartner, B., Karlin, I., Leininger, M. L., Leverman, D., Marroquin, C., Moody, A., Ohmacht, M., Pankajakshan, R., Pizzano, F., Rogers, J. H., Rosenburg, B., Schmidt, D., Shankar, M., Wang, F., Watson, P., Walkup, B., Weems, L. D., and Yin, J.: The design, deployment, and evaluation of the coral pre-exascale systems, in: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis, IEEE,  661–672, <ext-link xlink:href="https://doi.org/10.1109/SC.2018.00055" ext-link-type="DOI">10.1109/SC.2018.00055</ext-link>, 2018.
</mixed-citation></ref><?xmltex \hack{\newpage}?>
      <ref id="bib1.bibx25"><?xmltex \def\ref@label{Viglione et al.(2018)}?><label>Viglione et al.(2018)</label><?label GAV2018?><mixed-citation>Viglione, G. A., Thompson, A. F., Flexas, M. M., Sprintall, J., and Swart, S.: Abrupt transitions in submesoscale structure in southern drake passage: Glider observations and model results, J. Phys. Oceanogr.,  48,  2011–2027, <ext-link xlink:href="https://doi.org/10.1175/JPO-D-17-0192.1" ext-link-type="DOI">10.1175/JPO-D-17-0192.1</ext-link>, 2018.</mixed-citation></ref>
      <ref id="bib1.bibx26"><?xmltex \def\ref@label{Wan(2020)}?><label>Wan(2020)</label><?label WL2020?><mixed-citation>Wan, L.: The high resolution global ocean forecasting system in the nmefc and its intercomparison with the godae oceanview iv-tt class 4 metrics,
<ext-link xlink:href="https://www.godae.org/~godae-data/OceanView/Events/DA-OSEval-TT-2017/2.3-NMEFC-High-Resolution-Global-Ocean-Forecasting-System-and-Validation_v3.pdf">https://www.godae.org/$∼$godae-data/OceanView/Events/DA-OSEval-TT-2017/2.3-NMEFC-High-Resolution-Global-Ocean-Forecasting-System-and-Validation_v3.pdf</ext-link> (last access: 13 July 2022), 2020.</mixed-citation></ref>
      <ref id="bib1.bibx27"><?xmltex \def\ref@label{Wang et al.(2021)}?><label>Wang et al.(2021)</label><?label PW2020?><mixed-citation>Wang, P., Jiang, J., Lin, P., Ding, M., Wei, J., Zhang, F., Zhao, L., Li, Y., Yu, Z., Zheng, W., Yu, Y., Chi, X., and Liu, H.: The GPU version of LASG/IAP Climate System Ocean Model version 3 (LICOM3) under the heterogeneous-compute interface for portability (HIP) framework and its large-scale application , Geosci. Model Dev., 14, 2781–2799, <ext-link xlink:href="https://doi.org/10.5194/gmd-14-2781-2021" ext-link-type="DOI">10.5194/gmd-14-2781-2021</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bibx28"><?xmltex \def\ref@label{Xu et al.(2015)}?><label>Xu et al.(2015)</label><?label SX2015?><mixed-citation>Xu, S., Huang, X., Oey, L.-Y., Xu, F., Fu, H., Zhang, Y., and Yang, G.: POM.gpu-v1.0: a GPU-based Princeton Ocean Model, Geosci. Model Dev., 8, 2815–2827, <ext-link xlink:href="https://doi.org/10.5194/gmd-8-2815-2015" ext-link-type="DOI">10.5194/gmd-8-2815-2015</ext-link>, 2015.</mixed-citation></ref>
      <ref id="bib1.bibx29"><?xmltex \def\ref@label{Yang et al.(2021)}?><label>Yang et al.(2021)</label><?label XY2021?><mixed-citation>Yang, X., Zhou, S., Zhou, S., Song, Z., and Liu, W.: A barotropic solver for high-resolution ocean general circulation models, J. Mar. Sci. Eng., 9,  421, <ext-link xlink:href="https://doi.org/10.3390/jmse9040421" ext-link-type="DOI">10.3390/jmse9040421</ext-link>, 2021.</mixed-citation></ref>
      <ref id="bib1.bibx30"><?xmltex \def\ref@label{Ye et al.(2022a)}?><label>Ye et al.(2022a)</label><?label YY2022?><mixed-citation>Ye, Y., Song, Z., Zhou, S., Liu, Y., Shu, Q., Wang, B., Liu, W., Qiao, F.,  and Wang, L.: swNEMO(4.0), Zenodo [code], <ext-link xlink:href="https://doi.org/10.5281/zenodo.5976033" ext-link-type="DOI">10.5281/zenodo.5976033</ext-link>, 2022a.</mixed-citation></ref>
      <ref id="bib1.bibx31"><?xmltex \def\ref@label{Ye et al.(2022b)}?><label>Ye et al.(2022b)</label><?label YY2022b?><mixed-citation>Ye, Y., Song, Z., Zhou, S., Liu, Y., Shu, Q., Wang, B., Liu, W., Qiao, F., and Wang, L.: Data for swNEMO_v4.0 in GMD, Zenodo [data set], <ext-link xlink:href="https://doi.org/10.5281/zenodo.6834799" ext-link-type="DOI">10.5281/zenodo.6834799</ext-link>, 2022b.</mixed-citation></ref>
      <ref id="bib1.bibx32"><?xmltex \def\ref@label{Zeng et al.(2020)}?><label>Zeng et al.(2020)</label><?label YZ2020?><mixed-citation>Zeng, Y., Wang, L., Zhang, J., Zhu, G., Zhuang, Y., and Guo, Q.: Redistributing and optimizing high-resolution ocean model pop2 to million sunway cores, in: International Conference on Algorithms and Architectures for Parallel Processing, edited by: Qiu, M., Springer, <ext-link xlink:href="https://doi.org/10.1007/978-3-030-60245-1_19" ext-link-type="DOI">10.1007/978-3-030-60245-1_19</ext-link>, 2020.</mixed-citation></ref>
      <ref id="bib1.bibx33"><?xmltex \def\ref@label{Zhang et al.(2020)}?><label>Zhang et al.(2020)</label><?label SZ2020?><mixed-citation>Zhang, S., Fu, H., Wu, L., Li, Y., Wang, H., Zeng, Y., Duan, X., Wan, W., Wang, L., Zhuang, Y., Meng, H., Xu, K., Xu, P., Gan, L., Liu, Z., Wu, S., Chen, Y., Yu, H., Shi, S., Wang, L., Xu, S., Xue, W., Liu, W., Guo, Q., Zhang, J., Zhu, G., Tu, Y., Edwards, J., Baker, A., Yong, J., Yuan, M., Yu, Y., Zhang, Q., Liu, Z., Li, M., Jia, D., Yang, G., Wei, Z., Pan, J., Chang, P., Danabasoglu, G., Yeager, S., Rosenbloom, N., and Guo, Y.: Optimizing high-resolution Community Earth System Model on a heterogeneous many-core supercomputing platform, Geosci. Model Dev., 13, 4809–4829, <ext-link xlink:href="https://doi.org/10.5194/gmd-13-4809-2020" ext-link-type="DOI">10.5194/gmd-13-4809-2020</ext-link>, 2020.</mixed-citation></ref>

  </ref-list></back>
    <!--<article-title-html>swNEMO_v4.0: an ocean model based on NEMO4 for the new-generation Sunway supercomputer</article-title-html>
<abstract-html/>
<ref-html id="bib1.bib1"><label>Afzal et al.(2017)</label><mixed-citation>
Afzal, A., Ansari, Z., Faizabadi, A. R., and Ramis, M.: Parallelization strategies for computational fluid dynamics software: state of the art review, Arch. Computat. Methods Eng.,  24,  337–363, <a href="https://doi.org/10.1007/s11831-016-9165-4" target="_blank">https://doi.org/10.1007/s11831-016-9165-4</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib2"><label>Aumont et al.(2015)</label><mixed-citation>
Aumont, O., Ethé, C., Tagliabue, A., Bopp, L., and Gehlen, M.: PISCES-v2: an ocean biogeochemical model for carbon and ecosystem studies, Geosci. Model Dev., 8, 2465–2513, <a href="https://doi.org/10.5194/gmd-8-2465-2015" target="_blank">https://doi.org/10.5194/gmd-8-2465-2015</a>, 2015.
</mixed-citation></ref-html>
<ref-html id="bib1.bib3"><label>Baker et al.(2016)</label><mixed-citation>
Baker, A. H., Hu, Y., Hammerling, D. M., Tseng, Y.-H., Xu, H., Huang, X., Bryan, F. O., and Yang, G.: Evaluating statistical consistency in the ocean model component of the Community Earth System Model (pyCECT v2.0), Geosci. Model Dev., 9, 2391–2406, <a href="https://doi.org/10.5194/gmd-9-2391-2016" target="_blank">https://doi.org/10.5194/gmd-9-2391-2016</a>, 2016.
</mixed-citation></ref-html>
<ref-html id="bib1.bib4"><label>Bryan(1969)</label><mixed-citation>
Bryan, K.: A numerical method for the study of the circulation of the world ocean, J. Comput. Phys., 4,  347–376, 1969.
</mixed-citation></ref-html>
<ref-html id="bib1.bib5"><label>Bryan et al.(1967)</label><mixed-citation>
Bryan, K. and Cox, M. D.: A numerical investigation of the oceanic general circulation, Tellus,  19,  54–80, <a href="https://doi.org/10.1111/j.2153-3490.1967.tb01459.x" target="_blank">https://doi.org/10.1111/j.2153-3490.1967.tb01459.x</a>, 1967.
</mixed-citation></ref-html>
<ref-html id="bib1.bib6"><label>Chassignet et al.(2019)</label><mixed-citation>
Chassignet, E. P., LeSommer, J., and Wallcraft, A. J.: Generalcirculation models, in: Encyclopedia of Ocean Sciences, 3rd Edn., edited by: Cochran, K. J., Bokuniewicz, H. J., and Yager, P. L., Elsevier, 5, 486–490, <a href="https://doi.org/10.1016/B978-0-12-409548-9.11410-1" target="_blank">https://doi.org/10.1016/B978-0-12-409548-9.11410-1</a>, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib7"><label>Dawson and Düben(2017)</label><mixed-citation>
Dawson, A. and Düben, P. D.: rpe v5: an emulator for reduced floating-point precision in large numerical simulations, Geosci. Model Dev., 10, 2221–2230, <a href="https://doi.org/10.5194/gmd-10-2221-2017" target="_blank">https://doi.org/10.5194/gmd-10-2221-2017</a>, 2017.
</mixed-citation></ref-html>
<ref-html id="bib1.bib8"><label>Dong et al.(2020)</label><mixed-citation>
Dong, J., Fox-Kemper, B., Zhang, H., and Dong, C.: The seasonality of submesoscale energy production, content, and cascade, Geophys. Res. Lett., 7, e2020GL087388, <a href="https://doi.org/10.1029/2020GL087388" target="_blank">https://doi.org/10.1029/2020GL087388</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib9"><label>Eyring et al.(2016)</label><mixed-citation>
Eyring, V., Bony, S., Meehl, G. A., Senior, C. A., Stevens, B., Stouffer, R. J., and Taylor, K. E.: Overview of the Coupled Model Intercomparison Project Phase 6 (CMIP6) experimental design and organization, Geosci. Model Dev., 9, 1937–1958, <a href="https://doi.org/10.5194/gmd-9-1937-2016" target="_blank">https://doi.org/10.5194/gmd-9-1937-2016</a>, 2016.
</mixed-citation></ref-html>
<ref-html id="bib1.bib10"><label>Fu et al.(2016)</label><mixed-citation>
Fu, H., Liao, J., Yang, J., Wang, L., Song, Z., Huang, X., Yang, C., Xue, W., Liu, F., Qiao, F., Zhao, W., Yin, X., Hou, C., Zhang, C., Ge, W., Zhang, J., Wang, Y., Zhou, C., and Yang, G.: The sunway taihulight supercomputer: system and applications, Sci. China Inform. Sci., 59,  1–16, <a href="https://doi.org/10.1007/s11432-016-5588-7" target="_blank">https://doi.org/10.1007/s11432-016-5588-7</a>, 2016.
</mixed-citation></ref-html>
<ref-html id="bib1.bib11"><label>Gu et al.(2022)</label><mixed-citation>
Gu, J., Feng, F., Hao, X., Fang, T., Zhao, C., An, H., Chen, J., Xu, M., Li, J., Han, W., Yang, C., Li, F., and Chen, D.: Establishing a non-hydrostatic global atmospheric modeling system at 3-km horizontal resolution with aerosol feedbacks on the Sunway supercomputer of China, Sci. Bull., 67, 1170–1181, <a href="https://doi.org/10.1016/j.scib.2022.03.009" target="_blank">https://doi.org/10.1016/j.scib.2022.03.009</a>, 2022.
</mixed-citation></ref-html>
<ref-html id="bib1.bib12"><label>Hu et al.(2013)</label><mixed-citation>
Hu, Y., Huang, X., Wang,  X., Fu,  H., Xu, S., Ruan, H., Xue, W., and Yang, G.: A scalable barotropic mode solver for the parallel ocean program, in: European Conference on Parallel Processing, edited by: Wolf, F., Mohr, B., and Mey, D., Springer,  739–750, <a href="https://doi.org/10.1007/978-3-642-40047-6_74" target="_blank">https://doi.org/10.1007/978-3-642-40047-6_74</a>, 2013.
</mixed-citation></ref-html>
<ref-html id="bib1.bib13"><label>Jiang et al.(2019)</label><mixed-citation>
Jiang, J., Lin, P., Wang, J., Liu, H., Chi, X., Hao, H., Wang, Y., Wang, W., and Zhang, L.: Porting LASG/IAP climate system ocean model to GPUs using OpenAcc, IEEE Access, 7, 154490–154501, <a href="https://doi.org/10.1109/ACCESS.2019.2932443" target="_blank">https://doi.org/10.1109/ACCESS.2019.2932443</a>, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib14"><label>Lellouche et al.(2018)</label><mixed-citation>
Lellouche, J.-M., Greiner, E., Le Galloudec, O., Garric, G., Regnier, C., Drevillon, M., Benkiran, M., Testut, C.-E., Bourdalle-Badie, R., Gasparin, F., Hernandez, O., Levier, B., Drillet, Y., Remy, E., and Le Traon, P.-Y.: Recent updates to the Copernicus Marine Service global ocean monitoring and forecasting real-time 1∕12° high-resolution system, Ocean Sci., 14, 1093–1126, <a href="https://doi.org/10.5194/os-14-1093-2018" target="_blank">https://doi.org/10.5194/os-14-1093-2018</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib15"><label>Liao et al.(2014)</label><mixed-citation>
Liao, X., Xiao, L., Yang, C., and Lu, Y.: Milkyway-2 supercomputer: system and application, Front. Comput. Sci., 8,  345–356, <a href="https://doi.org/10.1007/s11704-014-3501-3" target="_blank">https://doi.org/10.1007/s11704-014-3501-3</a>, 2014.
</mixed-citation></ref-html>
<ref-html id="bib1.bib16"><label>Madec and the NEMO team(2016)</label><mixed-citation>
Madec, G. and the NEMO team: NEMO ocean engine, Zenodo, <a href="https://doi.org/10.5281/zenodo.3248739" target="_blank">https://doi.org/10.5281/zenodo.3248739</a>, 2016.
</mixed-citation></ref-html>
<ref-html id="bib1.bib17"><label>Putnam et al.(2015)</label><mixed-citation>
Putnam, A., Caulfield, A. M., Chung, E. S., Chiou, D., Constantinides, K., Demme, J., Esmaeilzadeh, H., Fowers, J., Gopal, G. P., Gray, J., Haselman, M., Hauck, S., Heil, S., Hormati, A., Kim, J.-Y., Lanka, S., Larus, J., Peterson, E., Pope, S., Smith, A., Thong, J., Xiao, P. Y., and Burger, D.: A reconfigurable fabric for accelerating large-scale datacenter services, IEEE Micro., 35, 10–22, <a href="https://doi.org/10.1109/MM.2015.42" target="_blank">https://doi.org/10.1109/MM.2015.42</a>, 2015.
</mixed-citation></ref-html>
<ref-html id="bib1.bib18"><label>Qiu et al.(2018)</label><mixed-citation>
Qiu, B., Chen, S., Klein, P., Wang, J., Torres, H., Fu, L. L., and Menemenlis, D.: Seasonality in transition scale from balanced to unbalanced motions in the world ocean, J. Phys. Oceanogr.,  48,  591–605, <a href="https://doi.org/10.1175/JPO-D-17-0169.1" target="_blank">https://doi.org/10.1175/JPO-D-17-0169.1</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib19"><label>Qiu et al.(2020)</label><mixed-citation>
Qiu, B., Chen, S., Klein, P., Torres, H., Wang, J., Fu, L. L., and Menemenlis, D.: Reconstructing upper-ocean vertical velocity field from sea surface height in the presence of unbalanced motion, J. Phys. Oceanogr.,  50, 55–79, <a href="https://doi.org/10.1175/JPO-D-19-0172.1" target="_blank">https://doi.org/10.1175/JPO-D-19-0172.1</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib20"><label>Rocha et al.(2016)</label><mixed-citation>
Rocha, C. B., Gille, S. T., Chereskin, T. K., and Menemenlis, D.: Seasonality of submesoscale dynamics in the kuroshio extension, Geophys. Res. Lett., 43, 11–304, <a href="https://doi.org/10.1002/2016GL071349" target="_blank">https://doi.org/10.1002/2016GL071349</a>, 2016.
</mixed-citation></ref-html>
<ref-html id="bib1.bib21"><label>Ruston(2019)</label><mixed-citation>
Ruston, B.: Validation test report for the dtic, <a href="https://apps.dtic.mil/sti/pdfs/AD1090615.pdf" target="_blank"/> (last access: 13 July 2022), 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib22"><label>Smith et al.(2010)</label><mixed-citation>
Smith, R., Jones, P., Briegleb, B.,Bryan, F., Danabasoglu, G., Dennis, J., Dukowicz, J., Eden, C., Fox-Kemper, B., Gent, P., Hecht, M., Jayne, S., Jochum, M., Large, W., Lindsay, K., Maltrud, M., Norton, N., Peacock, S., Vertenstein, M., and Yeager, S.: The parallel ocean program (pop) reference manual ocean component of the community climate system model (ccsm) and community earth system model (cesm), LAUR-01853,  141,  1–140, <a href="https://www.cesm.ucar.edu/models/cesm1.0/pop2/doc/sci/POPRefManual.pdf" target="_blank"/> (last access: 13 July 2022), 2010.
</mixed-citation></ref-html>
<ref-html id="bib1.bib23"><label>Tintó Prims et al.(2019)</label><mixed-citation>
Tintó Prims, O., Acosta, M. C., Moore, A. M., Castrillo, M., Serradell, K., Cortés, A., and Doblas-Reyes, F. J.: How to use mixed precision in ocean models: exploring a potential reduction of numerical precision in NEMO 4.0 and ROMS 3.6, Geosci. Model Dev., 12, 3135–3148, <a href="https://doi.org/10.5194/gmd-12-3135-2019" target="_blank">https://doi.org/10.5194/gmd-12-3135-2019</a>, 2019.
</mixed-citation></ref-html>
<ref-html id="bib1.bib24"><label>Vazhkudai et al.(2018)</label><mixed-citation>
Vazhkudai, S. S., deSupinski, B. R., Bland,  A. S., Geist, A., Sexton, J., Kahle, J., Zimmer, C. J., Atchley, S., Oral, S., Maxwell, D. E., Vergara Larrea, V. G., Bertsch, A., Goldstone, R., Joubert, W., Chambreau, C., Appelhans, D., Blackmore, R., Casses, B., Chochia, G., Davision, G., Ezell, M. A., Gooding, T., Gonsiorowski, E., Grinberg, L., Hanson, B., Hartner, B., Karlin, I., Leininger, M. L., Leverman, D., Marroquin, C., Moody, A., Ohmacht, M., Pankajakshan, R., Pizzano, F., Rogers, J. H., Rosenburg, B., Schmidt, D., Shankar, M., Wang, F., Watson, P., Walkup, B., Weems, L. D., and Yin, J.: The design, deployment, and evaluation of the coral pre-exascale systems, in: SC18: International Conference for High Performance Computing, Networking, Storage and Analysis, IEEE,  661–672, <a href="https://doi.org/10.1109/SC.2018.00055" target="_blank">https://doi.org/10.1109/SC.2018.00055</a>, 2018.

</mixed-citation></ref-html>
<ref-html id="bib1.bib25"><label>Viglione et al.(2018)</label><mixed-citation>
Viglione, G. A., Thompson, A. F., Flexas, M. M., Sprintall, J., and Swart, S.: Abrupt transitions in submesoscale structure in southern drake passage: Glider observations and model results, J. Phys. Oceanogr.,  48,  2011–2027, <a href="https://doi.org/10.1175/JPO-D-17-0192.1" target="_blank">https://doi.org/10.1175/JPO-D-17-0192.1</a>, 2018.
</mixed-citation></ref-html>
<ref-html id="bib1.bib26"><label>Wan(2020)</label><mixed-citation>
Wan, L.: The high resolution global ocean forecasting system in the nmefc and its intercomparison with the godae oceanview iv-tt class 4 metrics,
<a href="https://www.godae.org/~godae-data/OceanView/Events/DA-OSEval-TT-2017/2.3-NMEFC-High-Resolution-Global-Ocean-Forecasting-System-and-Validation_v3.pdf" target="_blank">https://www.godae.org/$∼$godae-data/OceanView/Events/DA-OSEval-TT-2017/2.3-NMEFC-High-Resolution-Global-Ocean-Forecasting-System-and-Validation_v3.pdf</a> (last access: 13 July 2022), 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib27"><label>Wang et al.(2021)</label><mixed-citation>
Wang, P., Jiang, J., Lin, P., Ding, M., Wei, J., Zhang, F., Zhao, L., Li, Y., Yu, Z., Zheng, W., Yu, Y., Chi, X., and Liu, H.: The GPU version of LASG/IAP Climate System Ocean Model version 3 (LICOM3) under the heterogeneous-compute interface for portability (HIP) framework and its large-scale application , Geosci. Model Dev., 14, 2781–2799, <a href="https://doi.org/10.5194/gmd-14-2781-2021" target="_blank">https://doi.org/10.5194/gmd-14-2781-2021</a>, 2021.
</mixed-citation></ref-html>
<ref-html id="bib1.bib28"><label>Xu et al.(2015)</label><mixed-citation>
Xu, S., Huang, X., Oey, L.-Y., Xu, F., Fu, H., Zhang, Y., and Yang, G.: POM.gpu-v1.0: a GPU-based Princeton Ocean Model, Geosci. Model Dev., 8, 2815–2827, <a href="https://doi.org/10.5194/gmd-8-2815-2015" target="_blank">https://doi.org/10.5194/gmd-8-2815-2015</a>, 2015.
</mixed-citation></ref-html>
<ref-html id="bib1.bib29"><label>Yang et al.(2021)</label><mixed-citation>
Yang, X., Zhou, S., Zhou, S., Song, Z., and Liu, W.: A barotropic solver for high-resolution ocean general circulation models, J. Mar. Sci. Eng., 9,  421, <a href="https://doi.org/10.3390/jmse9040421" target="_blank">https://doi.org/10.3390/jmse9040421</a>, 2021.
</mixed-citation></ref-html>
<ref-html id="bib1.bib30"><label>Ye et al.(2022a)</label><mixed-citation>
Ye, Y., Song, Z., Zhou, S., Liu, Y., Shu, Q., Wang, B., Liu, W., Qiao, F.,  and Wang, L.: swNEMO(4.0), Zenodo [code], <a href="https://doi.org/10.5281/zenodo.5976033" target="_blank">https://doi.org/10.5281/zenodo.5976033</a>, 2022a.
</mixed-citation></ref-html>
<ref-html id="bib1.bib31"><label>Ye et al.(2022b)</label><mixed-citation>
Ye, Y., Song, Z., Zhou, S., Liu, Y., Shu, Q., Wang, B., Liu, W., Qiao, F., and Wang, L.: Data for swNEMO_v4.0 in GMD, Zenodo [data set], <a href="https://doi.org/10.5281/zenodo.6834799" target="_blank">https://doi.org/10.5281/zenodo.6834799</a>, 2022b.
</mixed-citation></ref-html>
<ref-html id="bib1.bib32"><label>Zeng et al.(2020)</label><mixed-citation>
Zeng, Y., Wang, L., Zhang, J., Zhu, G., Zhuang, Y., and Guo, Q.: Redistributing and optimizing high-resolution ocean model pop2 to million sunway cores, in: International Conference on Algorithms and Architectures for Parallel Processing, edited by: Qiu, M., Springer, <a href="https://doi.org/10.1007/978-3-030-60245-1_19" target="_blank">https://doi.org/10.1007/978-3-030-60245-1_19</a>, 2020.
</mixed-citation></ref-html>
<ref-html id="bib1.bib33"><label>Zhang et al.(2020)</label><mixed-citation>
Zhang, S., Fu, H., Wu, L., Li, Y., Wang, H., Zeng, Y., Duan, X., Wan, W., Wang, L., Zhuang, Y., Meng, H., Xu, K., Xu, P., Gan, L., Liu, Z., Wu, S., Chen, Y., Yu, H., Shi, S., Wang, L., Xu, S., Xue, W., Liu, W., Guo, Q., Zhang, J., Zhu, G., Tu, Y., Edwards, J., Baker, A., Yong, J., Yuan, M., Yu, Y., Zhang, Q., Liu, Z., Li, M., Jia, D., Yang, G., Wei, Z., Pan, J., Chang, P., Danabasoglu, G., Yeager, S., Rosenbloom, N., and Guo, Y.: Optimizing high-resolution Community Earth System Model on a heterogeneous many-core supercomputing platform, Geosci. Model Dev., 13, 4809–4829, <a href="https://doi.org/10.5194/gmd-13-4809-2020" target="_blank">https://doi.org/10.5194/gmd-13-4809-2020</a>, 2020.
</mixed-citation></ref-html>--></article>
