﻿<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.0 20120330//EN" "http://jats.nlm.nih.gov/publishing/1.0/JATS-journalpublishing1.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="nlm-ta">J. Mater. Inf.</journal-id>
      <journal-id journal-id-type="publisher-id">JMI</journal-id>
      <journal-title-group>
        <journal-title>Journal of Materials Informatics</journal-title>
      </journal-title-group>
      <issn pub-type="epub">2770-372X</issn>
      <publisher>
        <publisher-name>OAE Publishing Inc.</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.20517/jmi.2026.21</article-id>
      <article-categories>
        <subj-group>
          <subject>Research Article</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Multi-activity derivative automated labeling and margin-sampling active learning for efficient determination of phase boundaries</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <name>
            <surname>Chen</surname>
            <given-names>Dingding</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <name>
            <surname>Xu</surname>
            <given-names>Guanglong</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I*">
            <sup>*</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1" />
          <contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-5367-7918</contrib-id>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Deng</surname>
            <given-names>Bincan</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Chen</surname>
            <given-names>Fuwen</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Fang</surname>
            <given-names>Jiheng</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Wang</surname>
            <given-names>Zhuo</given-names>
          </name>
          <xref ref-type="aff" rid="I5">
            <sup>5</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Xuan</surname>
            <given-names>Chen</given-names>
          </name>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <name>
            <surname>Cui</surname>
            <given-names>Yuwen</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I*">
            <sup>*</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1" />
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Zhang</surname>
            <given-names>Aimin</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
        </contrib>
      </contrib-group>
      <aff id="I1">
        <sup>1</sup>Sino-Spain Joint Laboratory on Biomedical Materials (S2LBM) &amp; College of Materials Science and Engineering, Nanjing Tech University, Nanjing 211816, Jiangsu, China.</aff>
      <aff id="I2">
        <sup>2</sup>School of Mathematics and Physics, Xi’an Jiaotong-Liverpool University, Suzhou 215123, Jiangsu, China.</aff>
      <aff id="I3">
        <sup>3</sup>State Key Laboratory of Precious Metal Functional Materials, Kunming Institute of Precious Metals, Kunming 650106, Yunnan, China.</aff>
      <aff id="I4">
        <sup>4</sup>Yunnan Precious Metals Laboratory Co., Ltd., Kunming 650106, Yunnan, China.</aff>
      <aff id="I5">
        <sup>5</sup>MatAi Co. Ltd., Chengdu 610213, Sichuan, China.</aff>
      <author-notes>
        <corresp id="cor1"><sup>*</sup>Correspondence to: Dr. Guanglong Xu, Prof. Yuwen Cui, Sino-Spain Joint Laboratory on Biomedical Materials (S2LBM) &amp; College of Materials Science and Engineering, Nanjing Tech University, Nanjing 211816, Jiangsu, China. E-mail: <email>guanglongxu@njtech.edu.cn</email>; <email>ycui@njtech.edu.cn</email></corresp>
        <fn fn-type="other">
          <p>
            <bold>Received:</bold> 16 Apr 2026 | <bold>First Decision:</bold> 8 Jul 2026 | <bold>Revised:</bold> 13 Aug 2026 | <bold>Accepted:</bold> 1 Sep 2026 | <bold>Published:</bold> 29 Sep 2026</p>
        </fn>
        <fn fn-type="other">
          <p>
            <bold>Academic Editor:</bold> Siqi Shi | <bold>Copy Editor:</bold> Pei-Yun Wang | <bold>Production Editor:</bold> Pei-Yun Wang</p>
        </fn>
      </author-notes>
      <pub-date pub-type="ppub">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>29</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>6</volume>
	  <issue>3</issue>
      <elocation-id>46</elocation-id>
      <permissions>
        <copyright-statement>© The Author(s) 2026.</copyright-statement>
        <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
          <license-p>© The Author(s) 2026. <bold>Open Access</bold> This article is licensed under a Creative Commons Attribution 4.0 International License (<uri xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</uri>), which permits unrestricted use, sharing, adaptation, distribution and reproduction in any medium or format, for any purpose, even commercially, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons license, and indicate if changes were made.</license-p>
        </license>
      </permissions>
      <abstract>
        <p>To realize the exploration of unknown diagrams directly from measurable physical/chemical properties, we improve the property derivatives-based machine learning (ML) framework of phase boundary prediction to overcome prior limitations in signal enhancement, labeling efficiency, uncertainty quantification, and potential applicability to multicomponent composition spaces. An automated phase-region labeling approach combining multiple activity derivatives is introduced to amplify the differentiability signal near phase boundaries, enabling precise identification of distinct phase regions. Three uncertainty sampling (US) strategies, i.e., least confidence (LC), margin sampling (MS), and entropy-based acquisition (EA), are employed to guide query selection in active learning. MS generally promoted a more uniform distribution of queries along phase boundaries and reduced redundant sampling near triple junctions, thereby facilitating efficient convergence. The truncated ML phase boundaries based on multi-layer perceptron + MS achieved Macro-F1 scores of 0.9994 and 0.9814 using only 120 and 186 labeled samples for the Pd-Pt and Au-Ag-Ge systems, respectively. The mean boundary displacements of ML-predicted phase boundaries against the Thermo-Calc-calculated results were 0.0059 for the Pd-Pt system and 0.0778 for the Au-Ag-Ge system.</p>
      </abstract>
      <kwd-group>
        <kwd>Labeling</kwd>
        <kwd>sampling</kwd>
        <kwd>activity derivatives</kwd>
        <kwd>uncertainty</kwd>
        <kwd>phase boundary</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec1">
      <title>INTRODUCTION</title>
      <p>Phase diagrams are crucial for designing the chemical compositions and phase structures of new alloys<sup>[<xref ref-type="bibr" rid="B1">1</xref>]</sup>. In recent years, machine learning (ML) has been widely applied to predict phase equilibria and phase boundaries<sup>[<xref ref-type="bibr" rid="B2">2</xref>-<xref ref-type="bibr" rid="B4">4</xref>]</sup> by creating ML energy/topological functionals<sup>[<xref ref-type="bibr" rid="B5">5</xref>-<xref ref-type="bibr" rid="B8">8</xref>]</sup> and performing automatic parameter optimization<sup>[<xref ref-type="bibr" rid="B9">9</xref>,<xref ref-type="bibr" rid="B10">10</xref>]</sup>, or by statistically analyzing geometric, physical, and chemical properties<sup>[<xref ref-type="bibr" rid="B11">11</xref>-<xref ref-type="bibr" rid="B16">16</xref>]</sup>. These methods have significantly sped up phase diagram analysis. However, most concentrate on reconstructing known phase boundaries or diagrams with greater accuracy. Few strategies are reported for using ML to investigate an unknown phase diagram based only on experimentally measurable physical or chemical properties.</p>
      <p>A property-derivatives-based sampling and classification prototype was previously proposed to estimate phase boundaries directly from activity derivatives<sup>[<xref ref-type="bibr" rid="B17">17</xref>]</sup>. It fully leveraged the universal principle that the derivability of the system’s physical and chemical properties changes abruptly in the neighborhood of the phase boundary and developed a virtual scenario that fully simulates the experimental process of determining an unknown phase diagram using assessed activity data. This method directly couples physical and chemical properties to phase boundaries, thereby circumventing the need to model and minimize Gibbs free energy<sup>[<xref ref-type="bibr" rid="B18">18</xref>,<xref ref-type="bibr" rid="B19">19</xref>]</sup>, making it particularly useful for users primarily interested in phase boundary and composition data at specific chemical points but without the experience in CALPHAD database development. Nevertheless, the prior study had four limitations: (1) it employed a single-component activity derivative to define the phase region descriptor, which failed to capture the different differential coefficients in the left and right neighborhoods, leading to insufficient signal enhancement at phase boundaries in a multi-component system; (2) it required additional manual assistance to complete the labeling process; (3) the prediction confidence score was assessed on a predefined testing grid, rather than on probability-based uncertainty measures derived from model predictions. The effects of different uncertainty acquisition criteria on sampling efficiency were therefore not systematically investigated; (4) it employed a support vector machine as the classifier, which was based on Euclidean distance and could lead to mathematical inaccuracies in high-dimensional Simplex compositional spaces.</p>
      <p>To overcome these limitations, this study further develops a derivatives-based framework that combines automated phase region labeling from multiple activity derivatives with pool-based active learning. We compare uncertainty sampling (US) methods<sup>[<xref ref-type="bibr" rid="B20">20</xref>-<xref ref-type="bibr" rid="B25">25</xref>]</sup> for multicomponent activity derivatives within adaptive-grid active learning<sup>[<xref ref-type="bibr" rid="B25">25</xref>-<xref ref-type="bibr" rid="B28">28</xref>]</sup>, aiming to build phase boundaries more effectively and reliably. The sampling and ML steps include: (1) collecting data from the pool of activity derivatives; (2) automated labeling of phase regions; (3) ML predicting phase boundaries and estimating uncertainty based on confidence; (4) re-sampling within the adaptive grid and updating models. These steps (3) and (4) are repeated until a specific convergence or accuracy goal is reached. Finally, the best ML algorithms, labeling approaches, and US strategies are summarized and justified. All these are detailed in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
      <fig id="fig1" position="float">
        <label>Figure 1</label>
        <caption>
          <p>Technical roadmap of active learning to determine phase boundaries, and the highlights of automated labeling and uncertainty evaluation. ML: Machine learning.</p>
        </caption>
        <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.1.jpg" />
      </fig>
    </sec>
    <sec id="sec2">
      <title>MATERIALS AND METHODS</title>
      <sec id="sec2-1">
        <title>Thermodynamic fundamentals of derivative-based ML method</title>
        <p>Based on the thermodynamics of phase diagrams, the Gibbs energy of a multi-phase system is the minimum curve that envelops the Gibbs energy of single phases and the common tangent lines/planes. It is continuous overall, but exhibits a “cusp” (i.e., a non-differentiable point) at the phase boundary composition. Therefore, by sampling the derivative of activity as a function of composition and numerically detecting the position where differentiability is lost, the phase boundary can be iteratively approximated. More details can be found in Ref.<sup>[<xref ref-type="bibr" rid="B17">17</xref>]</sup>.</p>
      </sec>
      <sec id="sec2-2">
        <title>Automated phase-region labeling</title>
        <p>In our previous work<sup>[<xref ref-type="bibr" rid="B17">17</xref>]</sup>, phase regions were identified using a single activity derivative. This approach is effective when the selected derivative exhibits clear discontinuities at phase boundaries. In a multicomponent system, however, the sensitivity of an activity derivative depends on both the selected component and the independent composition direction. Consequently, a single derivative may detect only part of the phase boundaries and may require manual assistance to complete the phase-region labels. To address this limitation, the present method combines multiple activity-derivative channels and performs boundary extraction and region labeling algorithmically, without manual boundary drawing or point-by-point label assignment. The Pd-Pt binary system employs two derivative channels of <italic>z</italic><sub>1</sub> = <inline-formula><tex-math id="M1">$$ \frac{\partial a_{Pt}}{\partial x_{Pt}} $$</tex-math></inline-formula> and <italic>z</italic><sub>2</sub> = <inline-formula><tex-math id="M1">$$ \frac{\partial a_{Pd}}{\partial x_{Pt}} $$</tex-math></inline-formula>, (<italic>a<sub>i</sub></italic> is ‘system activity’ of component <italic>i</italic> relative to a selected reference state.) while the Au-Ag-Ge ternary system employs four independent derivative channels of <italic>z</italic><sub>1</sub> = <inline-formula><tex-math id="M1">$$ \frac{\partial a_{Ag}}{\partial x_{Ag}} $$</tex-math></inline-formula>, <italic>z</italic><sub>2</sub> = <inline-formula><tex-math id="M1">$$ \frac{\partial a_{Au}}{\partial x_{Au}} $$</tex-math></inline-formula>, <italic>z</italic><sub>3</sub> = <inline-formula><tex-math id="M1">$$ \frac{\partial a_{Au}}{\partial x_{Ag}} $$</tex-math></inline-formula>, <italic>z</italic><sub>4</sub> = <inline-formula><tex-math id="M1">$$ \frac{\partial a_{Ag}}{\partial x_{Au}} $$</tex-math></inline-formula>, and an auxiliary composite channel was constructed as the weighted average <italic>z</italic><sub>5</sub> = Σ<italic><sub>k</sub></italic><italic>w<sub>k</sub>z<sub>k</sub></italic> (k = 1-4, the adjustable parameter <italic>w<sub>k</sub></italic> = 0.25 in this study). Although <italic>z</italic><sub>5</sub> introduces no independent thermodynamic information, equal-weight averaging can reinforce spatially consistent weak responses shared across multiple derivative channels while partially suppressing channel-specific variations, thereby producing a more continuous boundary signal after Sobel filtering.</p>
        <p>To quantify the spatial variation of the derivative signals, the Sobel operator<sup>[<xref ref-type="bibr" rid="B29">29</xref>]</sup> was applied to each derivative field. For a given activity-derivative channel <italic>z<sub>k</sub></italic>(<italic>x</italic>), the gradient magnitude was calculated as</p>
        <p><disp-formula> <label>(1)</label> <tex-math id="E1"> $$  G_k(x)=\left\|\nabla z_k(x)\right\|_2
=\sqrt{
\left(\frac{\partial z_k}{\partial u}\right)^2
+
\left(\frac{\partial z_k}{\partial v}\right)^2
}, $$ </tex-math></disp-formula></p>
        <p>where <italic>u</italic> and <italic>v</italic> denote the two independent coordinates of the corresponding calculation domain. The gradient magnitude <italic>G<sub>k</sub></italic>(<italic>x</italic>) characterizes the local variation of the activity-derivative field and highlights regions with strong thermodynamic contrast.</p>
        <p>To identify potential phase-boundary locations, grid points <italic>x</italic> whose exceed a percentile-based threshold <italic>P<sub>a</sub></italic>(·) (an adaptively chosen threshold that highlights regions with high derivative contrast) are classified as boundary pixels. The percentile level controls the balance between boundary completeness and noise suppression. A lower percentile retains weaker gradient responses and improves boundary continuity, but may introduce noise and fragmented regions. A higher percentile suppresses spurious boundaries, but may remove weak boundary segments and merge adjacent regions. For each available derivative channel <italic>k</italic>, the corresponding boundary candidate set is formulated as</p>
        <p><disp-formula> <label>(2)</label> <tex-math id="E1"> $$  B_k=\left\{x\mid G_k(x)\geq P_a(G_k)\right\}. $$ </tex-math></disp-formula></p>
        <p>To balance these effects, the search was initialized at 90% and performed bidirectionally from 80% to 98% in increments of 0.25 percentage points. The optimal percentile was selected from the interval where the fused region topology remained stable, with preference for fewer fragmented regions and proximity to 90%. Thermo-Calc phase labels were not used during threshold selection.</p>
        <p>Therefore, the final phase boundary can be created through a geometric union of all candidate sets <italic>B<sub>k</sub></italic>, which combines signals from various derivatives and enhances subtle boundary features</p>
        <p><disp-formula> <label>(3)</label> <tex-math id="E1"> $$  B_{\mathrm{union}}=\bigcup_{k=1}^{K}B_k, $$ </tex-math></disp-formula></p>
        <p>where <italic>K</italic> = 2 for Pd-Pt system and <italic>K</italic> = 5 for Au-Ag-Ge system. Although the labeling procedure is fully automated, a final expert inspection is recommended to identify possible artifacts associated with limited grid resolution, weak derivative signals, or extremely narrow phase regions.</p>
      </sec>
      <sec id="sec2-3">
        <title>Data pool construction</title>
        <p>In this study, we illustrate the US scheme using the same benchmark cases as in Ref.<sup>[<xref ref-type="bibr" rid="B17">17</xref>]</sup>, specifically the miscibility gap boundary of fcc#1+fcc#2 on the Pd-Pt binary isopleth (700-1,100 K)<sup>[<xref ref-type="bibr" rid="B30">30</xref>]</sup>, and the isothermal section of the Au-Ag-Ge ternary system at 800 K<sup>[<xref ref-type="bibr" rid="B31">31</xref>]</sup>. The original data pool comprises the independent chemical compositions, <italic>x<sub>j</sub></italic>, the activity of component <italic>i</italic> of the system, <italic>a<sub>i</sub></italic> (<italic>i ≤ n</italic>, <italic>j ≤ n</italic> - 1, <italic>n</italic> is the component number), and the corresponding activity derivatives <inline-formula><tex-math id="M1">$$ \frac{\partial a_{i}}{\partial x_{j}} $$</tex-math></inline-formula>, calculated from the database in Ref.<sup>[<xref ref-type="bibr" rid="B30">30</xref>,<xref ref-type="bibr" rid="B31">31</xref>]</sup> using a user-defined function. We also verified that the calculation results matched those obtained from a manual numerical finite-difference method. These activity derivatives serve as the fundamental features for phase-region labeling and subsequent ML prediction.</p>
        <p>To construct a sufficiently dense and representative sampling space, the compositional domain was discretized with a uniform resolution. For the Pd-Pt system, Pt composition and temperature were used as the independent variables. The Pt composition range of 0-1 was discretized at an interval of 0.01 in mole fraction, while the temperature range of 700-1,100 K was discretized at an interval of 25 K. The complete grid therefore contained 1,717 state points. After excluding 50 uniformly distributed initial training samples, the candidate pool contained 1,667 unlabeled points. For the Au-Ag-Ge ternary system, the compositions of Au (<italic>x<sub>Au</sub></italic>) and Ag (<italic>x<sub>Ag</sub></italic>) were used as the independent variables and discretized at intervals of 0.01. The valid compositional domain was defined under the constraint of <italic>x<sub>Au</sub></italic> + <italic>x<sub>Ag</sub></italic> + <italic>x<sub>Ge</sub></italic> = 1, and shaped into the triangle composition simplex. <italic>x<sub>Au</sub></italic> varied from 0.01 to 0.99, and for each <italic>x<sub>Au</sub></italic>, <italic>x<sub>Ag</sub></italic> varied from 0 to 1 - <italic>x<sub>Au</sub></italic>. Thus, discretization of this domain produced 5,049 valid state points. After selecting 66 uniformly distributed points for initial training, the candidate pool contained 4,983 unlabeled points. This pool serves as the ground-truth reservoir from which a small subset of labeled samples is progressively drawn during the active learning process, thereby simulating realistic experimental data acquisition under limited labelling budgets.</p>
        <p>During each active-learning iteration, five samples were selected from the candidate pool, labeled, and transferred to the training set. Therefore, after iteration <italic>r</italic>, the numbers of training and candidate samples were</p>
		<p><disp-formula> <label>(4)</label> <tex-math id="E1"> $$  N_{\mathrm{train}}(r)=N_0+5r $$ </tex-math></disp-formula></p>
        <p>and</p>
        <p><disp-formula> <label>(5)</label> <tex-math id="E1"> $$  N_{\mathrm{candidate}}(r)
=N_{\mathrm{total}}-N_0-5r, $$ </tex-math></disp-formula></p>
        <p>respectively, where <italic>N</italic><sub>0</sub> is the number of initial training samples (<italic>N</italic><sub>0</sub> = 50 for Pd-Pt system, <italic>N</italic><sub>0</sub> = 66 for Au-Ag-Ge system). Queried samples were removed from the candidate pool to prevent repeated selection.</p>
      </sec>
      <sec id="sec2-4">
        <title>ML models and US strategy</title>
        <p>In this study, ML models are employed with the explicit objective of efficiently reconstructing equilibrium phase boundaries from limited thermodynamic information. Gaussian process classifier (GPC, kernel-based)<sup>[<xref ref-type="bibr" rid="B32">32</xref>]</sup> and multi-layer perceptron (MLP, neural network-based)<sup>[<xref ref-type="bibr" rid="B33">33</xref>]</sup> approaches were adopted to train the ML models, while random forest (RF), eXtreme gradient boosting (XGB), and support vector classifier (SVC) in the previous work served as the counterparts. These models were trained on labeled datasets and iteratively updated as new informative samples were acquired. The hyperparameters of individual ML models are displayed in <inline-supplementary-material content-type="local-data" mimetype="application/zip" xlink:href="jmi6021-SupplementaryMaterials.zip">Supplementary Materials 1</inline-supplementary-material>. Brier scores and expected calibration errors (ECE) were employed to evaluate the probability calibration of the underlying models.</p>
        <p>To emulate the active learning process in practical phase-diagram construction, several dozen data points were first uniformly sampled from the pool and assembled into the initial training dataset, representing the first round of data acquisition in practical phase-diagram exploration. The remaining data served as unlabeled candidates for query selection in each active learning cycle (analogous to the supplementary data in the second or third rounds of experiments). <xref ref-type="table" rid="t1">Table 1</xref> details the size of the candidate pool and the initial training set for Pd-Pt and Au-Ag-Ge systems.</p>
        <table-wrap id="t1">
          <label>Table 1</label>
          <caption>
            <p>Size of datasets and number of decision labels for Pd-Pt and Au-Ag-Ge systems</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>System</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Initial training set</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Candidate pool</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Truncated samples</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Decision labels</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>Pd-Pt</td>
                <td>50</td>
                <td>1,667</td>
                <td>120</td>
                <td>2</td>
              </tr>
              <tr>
                <td>Au-Ag-Ge</td>
                <td>66</td>
                <td>4,983</td>
                <td>186</td>
                <td>6</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>During each iteration, query selection was guided by uncertainty evaluations. Three uncertainty acquisition strategies, i.e., least confidence (LC), margin sampling (MS), and entropy-based acquisition (EA), were employed. For each candidate point <italic>x</italic>, an uncertainty score <italic>u</italic>(<italic>x</italic>) is computed from predicted probabilities <italic>P</italic>(<italic>y</italic>|<italic>x</italic>), and points with the highest uncertainty are selected:</p>
        <p><disp-formula> <label>(6)</label> <tex-math id="E1"> $$  x^*=\underset{x\in pool}{\arg\max}\;u(x), $$ </tex-math></disp-formula></p>
        <p>where <italic>u</italic>(<italic>x</italic>) depends on the specific acquisition criterion. Then, three strategies define <italic>u</italic>(<italic>x</italic>) as follows:</p>
        <p><disp-formula> <label>(7)</label> <tex-math id="E1"> $$  u_{LC}(x)=1-\max_y P(y\mid x), $$ </tex-math></disp-formula></p>
        <p><disp-formula> <label>(8)</label> <tex-math id="E1"> $$  u_{MS}(x)
=1-\left[P(y_1\mid x)-P(y_2\mid x)\right], $$ </tex-math></disp-formula></p>
        <p>and</p>
        <p><disp-formula> <label>(9)</label> <tex-math id="E1"> $$  u_{EA}(x)
=-\sum_y P(y\mid x)\log P(y\mid x), $$ </tex-math></disp-formula></p>
        <p>where <italic>P</italic>(<italic>y</italic><sub>1</sub>|<italic>x</italic>) and <italic>P</italic>(<italic>y</italic><sub>2</sub>|<italic>x</italic>) in <italic>u<sub>MS</sub></italic>(<italic>x</italic>) denote the highest and second-highest probabilities. The LC score measures the lack of confidence in the most probable class, MS score measures the ambiguity between the two most probable classes, and EA considers uncertainty across the complete class-probability distribution.</p>
        <p>To avoid redundant sampling concentrated in a small local region of the compositional space, a diversity-aware batch selection strategy<sup>[<xref ref-type="bibr" rid="B34">34</xref>]</sup> was adopted. In each iteration, the 25 points with the highest uncertainty scores were first identified as candidate queries. The most uncertain point was selected first, after which the remaining points were selected successively by maximizing their minimum distance from the labeled set and the samples already selected in the current batch. This procedure produced a batch of five spatially diverse queries. These five points were labeled, removed from the candidate pool, added to the training set, and used to update the model.</p>
      </sec>
      <sec id="sec2-5">
        <title>Model evaluation and convergence criteria</title>
        <p>Model performance was evaluated using standard classification metrics such as Accuracy, Precision, Recall, and Macro-F1 score<sup>[<xref ref-type="bibr" rid="B35">35</xref>]</sup>. All performance metrics were calculated over the entire data pool, rather than solely on the training subset. This evaluation protocol directly reflects the accuracy of the reconstructed phase boundaries across the full compositional domain and ensures a fair and comprehensive assessment of the active learning process. Given true positives (TP), true negatives (TN), false positives (FP), and false negatives (FN), respectively, the Macro-F1 score can be calculated as follows:</p>
        <p><disp-formula> <label>(10)</label> <tex-math id="E1"> $$  Precision=\frac{TP}{TP+FP}, $$ </tex-math></disp-formula></p>
		<p><disp-formula> <label>(11)</label> <tex-math id="E1"> $$  Recall=\frac{TP}{TP+FN}, $$ </tex-math></disp-formula></p>
        <p>and</p>
        <p><disp-formula> <label>(12)</label> <tex-math id="E1"> $$  Macro\text{-}F1
=\frac{1}{C}\sum
\frac{2\,Precision\times Recall}
{Precision+Recall}, $$ </tex-math></disp-formula></p>
        <p>where <italic>C</italic> is the number of phase-region classes. Among these metrics, the Macro-F1 score was adopted as the principal classification metric because it assigns equal importance to all phase labels and is therefore less affected by differences in the sizes of individual phase regions.</p>
        <p>However, classification-based metrics do not fully characterize the geometric fidelity of reconstructed phase boundaries. Therefore, boundary-specific validation was performed against CALPHAD-calculated equilibrium boundaries obtained using Thermo-Calc. Let <italic>B</italic><sub>pred</sub> and <italic>B</italic><sub>ref</sub> denote the boundary-point sets extracted from the reconstructed and reference phase diagrams, respectively. The mean boundary displacement was calculated for both systems as the average symmetric nearest-neighbor distance:</p>
		<p><disp-formula> <label>(13)</label> <tex-math id="E1"> $$  D_{\mathrm{MBD}}
=
\frac{1}{2}
\left[
\frac{1}{|B_{\mathrm{pred}}|}
\sum_{p\in B_{\mathrm{pred}}}
\min_{q\in B_{\mathrm{ref}}}
\|p-q\|_2
+
\frac{1}{|B_{\mathrm{ref}}|}
\sum_{q\in B_{\mathrm{ref}}}
\min_{p\in B_{\mathrm{pred}}}
\|q-p\|_2
\right]. $$ </tex-math></disp-formula></p>
        <p>Here, <italic>p</italic> represents a boundary point in <italic>B</italic><sub>pred</sub>, whereas <italic>q</italic> represents a boundary point in <italic>B</italic><sub>ref</sub>. Lower <italic>D</italic><sub>MBD</sub> values indicate closer agreement between the reconstructed and reference boundaries. For the Au-Ag-Ge system, triple-junction error was additionally evaluated because multiphase junctions represent the most topologically complex regions of the phase diagram. Let <italic>J</italic><sub>pred</sub> and <italic>J</italic><sub>ref</sub> denote the sets of predicted and reference triple-junction locations, respectively. The triple-junction error was calculated as</p>
		<p><disp-formula> <label>(14)</label> <tex-math id="E1"> $$  E_{\mathrm{TJ}}
=
\frac{1}{2}
\left[
\frac{1}{|J_{\mathrm{pred}}|}
\sum_{u\in J_{\mathrm{pred}}}
\min_{v\in J_{\mathrm{ref}}}
\|u-v\|_2
+
\frac{1}{|J_{\mathrm{ref}}|}
\sum_{v\in J_{\mathrm{ref}}}
\min_{u\in J_{\mathrm{pred}}}
\|v-u\|_2
\right], $$ </tex-math></disp-formula></p>
        <p>where <italic>u</italic> and <italic>v</italic> represent individual junction locations in <italic>J</italic><sub>pred</sub> and <italic>J</italic><sub>ref</sub>, respectively. ||·||<sub>2</sub> denotes the Euclidean distance after geometric embedding of the phase-diagram coordinates. Thus, for the ternary system, each composition point was mapped from the composition simplex to two-dimensional equilateral-triangle coordinates before distance calculation:</p>
        <p><disp-formula> <label>(15)</label> <tex-math id="E1"> $$  r=(r_x,r_y)
=
\left(
x_{\mathrm{Au}}+\frac{1}{2}x_{\mathrm{Ge}},
\frac{\sqrt{3}}{2}x_{\mathrm{Ge}}
\right), $$ </tex-math></disp-formula></p>
        <p>where <italic>r</italic> denotes the two-dimensional Cartesian position vector of a composition point after mapping the ternary simplex onto an equilateral triangle. These reference results were used exclusively for external validation and were not involved in automated labeling, model training, internal testing, or query selection.</p>
        <p>For convergence criteria, Macro-F1 thresholds of 0.98 for Pd-Pt and 0.95 for Au-Ag-Ge were used as initial performance targets for comparing sampling efficiency. Achieving them required fewer iterations than the final truncation criteria. Sampling proceeded until the Macro-F1 scores reached above reference level, then continued until the Macro-F1 improvement was less than or equal to 0.001 over five consecutive iterations, at which point sampling was truncated. The total number of labeled samples at termination, including both the initial training set and subsequently queried samples, was recorded as the final truncated sample number for phase-diagram reconstruction. Following this idea, the number of final samples is reported in <xref ref-type="table" rid="t1">Table 1</xref>.</p>
        <p>The main comparisons were conducted using uniformly distributed initial training samples to ensure identical starting conditions among the models and acquisition strategies. Meanwhile, to quantify the sampling uncertainty arising from the selection of the initial labeled samples, each model-acquisition combination was additionally repeated using ten independently generated class-stratified random initial training sets of the same size.</p>
      </sec>
    </sec>
    <sec id="sec3">
      <title>RESULTS AND DISCUSSION</title>
      <sec id="sec3-1">
        <title>Automated phase-region labeling result</title>
        <p>
          <xref ref-type="fig" rid="fig2">Figure 2</xref> presents the labeled phase regions of the entire data set without using ML. Notably, the amount of data required for active learning in the real experiments is considerably smaller than that of the full data pool. The purpose of this is to demonstrate, from a bird’s-eye view, that combining multiple activity derivatives enables complete segmentation of phase regions and boundary detection without losing thermodynamic signals.</p>
        <fig id="fig2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Partial derivatives of activities and their union result. (A-C) Pd-Pt system; (D-I) Au-Ag-Ge system.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.2.jpg" />
        </fig>
        <p>For the Pd-Pt binary, a single-component activity derivative (<inline-formula><tex-math id="M1">$$ \frac{\partial a_{Pd}}{\partial x_{Pt}} $$</tex-math></inline-formula> or <inline-formula><tex-math id="M1">$$ \frac{\partial a_{Pt}}{\partial x_{Pt}} $$</tex-math></inline-formula>) is sufficient to detect the sharp change in the activity derivative at the phase boundary, thereby pinpointing the fcc#1 + fcc#2 miscibility gap. As shown in <xref ref-type="fig" rid="fig2">Figure 2A</xref>-<xref ref-type="fig" rid="fig2">C</xref>, the two derivative channels identify the same boundary structure, and their fused result does not provide a substantial additional signal.</p>
        <p>In contrast, none of the individual <italic>z</italic><sub>1</sub>-<italic>z</italic><sub>5</sub> channels can independently achieve complete phase-region segmentation for the Au-Ag-Ge ternary system, as shown in <xref ref-type="fig" rid="fig2">Figure 2D</xref>-<xref ref-type="fig" rid="fig2">I</xref>. It should be emphasized that the composite channel <italic>z</italic><sub>5</sub> does not introduce new thermodynamic information or generate boundary features that are entirely absent from any individual channel. Instead, weak boundary responses may occur at spatially consistent locations across several derivative channels but remain fragmented or below the effective detection threshold in each individual map. By averaging the normalized derivative channels, these spatially consistent responses are retained and accumulated, whereas channel-specific fluctuations are partially suppressed. After subsequent Sobel-gradient calculation and percentile-based thresholding, some previously weak boundary segments become more continuous and distinguishable. Nevertheless, <italic>z</italic><sub>5</sub> alone is insufficient to recover the complete phase-boundary topology. The final segmentation is obtained by geometrically combining the candidate boundary sets from all available derivative channels, followed by region assignment based on connectivity and distance criteria. More evidence can be found in <inline-supplementary-material content-type="local-data" mimetype="application/zip" xlink:href="jmi6021-SupplementaryMaterials.zip">Supplementary Materials 2</inline-supplementary-material>.</p>
      </sec>
      <sec id="sec3-2">
        <title>Effectiveness of active learning on phase diagram prediction</title>
        <p>
          <xref ref-type="fig" rid="fig3">Figure 3</xref> summarizes the performance of US strategies combined with ML models for reconstructing the miscibility-gap boundary in the Pd-Pt binary system. As shown in <xref ref-type="fig" rid="fig3">Figure 3A</xref>-<xref ref-type="fig" rid="fig3">C</xref>, the Macro-F1 scores generally increase as additional labeled samples are acquired during active learning. Across most acquisition strategies, GPC and MLP achieved higher and more stable Macro-F1 scores than the SVC, RF, and XGB models evaluated in our previous study<sup>[<xref ref-type="bibr" rid="B17">17</xref>]</sup>. A direct comparison in <xref ref-type="fig" rid="fig3">Figure 3D</xref>-<xref ref-type="fig" rid="fig3">F</xref> reports the number of labeled samples required by each model to reach a Macro-F1 score of 0.98. Both GPC and MLP achieve this performance using only 60 labeled samples under all three sampling strategies. Although SVC also reached the target with 60 samples under LC and EA, it required 100 samples under MS and exhibited more pronounced fluctuations during subsequent iterations.</p>
        <fig id="fig3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Comparison of the performance of different uncertainty sampling strategies and ML models on the Pd-Pt system. (A-C) Evolution of Macro-F1 score with increasing training samples; (D-F) Number of labeled samples required by each model to reach a Macro-F1 score of 0.98. US: Uncertainty sampling; ML: machine learning; MLP: multi-layer perceptron; GPC: Gaussian process classifier; SVC: support vector classifier; RF: random forest; XGB: eXtreme gradient boosting.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.3.jpg" />
        </fig>
        <p>The differences among LC, MS, and EA are relatively limited for the Pd-Pt system. This result can be attributed to its simple phase-region topology and the pronounced thermodynamic contrast across the miscibility-gap boundary. Because the activity-derivative signals already provide a clear distinction between the two-phase regions, the samples selected by different uncertainty criteria tend to concentrate near similar boundary locations. Consequently, the choice of acquisition strategy has a smaller influence on sampling efficiency for this binary system than the choice of ML model.</p>
        <p>The Au-Ag-Ge ternary system represents a substantially more challenging scenario for phase-boundary reconstruction than the Pd-Pt binary case. The performance of active learning on the Au-Ag-Ge ternary system is summarized in <xref ref-type="fig" rid="fig4">Figure 4</xref>. Note that all model-acquisition combinations in <xref ref-type="fig" rid="fig3">Figures 3</xref> and <xref ref-type="fig" rid="fig4">4A</xref>-<xref ref-type="fig" rid="fig4">C</xref> were intentionally continued to the same maximum number of labeled samples (200 for Pd-Pt, 396 for Au-Ag-Ge) to facilitate comparison of longer-term learning behavior and this range is distinct from the final truncated sample number used for phase-diagram reconstruction.</p>
        <fig id="fig4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Comparison of the performance of different uncertainty sampling strategies and ML models on the Au-Ag-Ge system. (A-C) Evolution of Macro-F1 score with increasing training samples; (D-F) Number of labeled samples required by each model to reach a Macro-F1 score of 0.95. US: Uncertainty sampling; ML: machine learning; MLP: multi-layer perceptron; GPC: Gaussian process classifier; SVC: support vector classifier; RF: random forest; XGB: eXtreme gradient boosting.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.4.jpg" />
        </fig>
        <p>Compared with the Pd-Pt system, the ternary system further confirms the advantages of probability-based models. MLP and GPC again outperform RF, XGB, and SVC in both convergence speed and final accuracy. The increased complexity of the compositional simplex requires reliable probability estimates to distinguish narrow phase regions and interconnected boundaries. Probabilistic models can provide such estimates, enabling US to focus effectively on informative compositions. Furthermore, the acquisition strategies exhibit more pronounced differences, reflecting greater heterogeneity in uncertainty across the ternary compositional domain. <xref ref-type="fig" rid="fig4">Figure 4D</xref>-<xref ref-type="fig" rid="fig4">F</xref> reports the minimum number of labeled samples required to reach a Macro-F1 score of 0.95. Among these US strategies, MS generally exhibits superior performance. Under MLP, MS requires a comparable, albeit slightly larger, number of samples than LC and EA to achieve the target Macro-F1 score. In contrast, MS demonstrates a pronounced advantage for GPC, SVC, and RF, requiring substantially fewer labeled instances.</p>
        <p>To assess the sampling uncertainty of the initial labeled samples, each model-acquisition combination was repeated ten times using class-stratified random initial sets with seeds 11, 22, 33, 44, 55, 66, 77, 88, 99, and 110. Results are shown in <inline-supplementary-material content-type="local-data" mimetype="application/zip" xlink:href="jmi6021-SupplementaryMaterials.zip">Supplementary Materials 3</inline-supplementary-material>. The performance trends from these randomized samples align with those from the uniformly distributed initial samples used in the main experiments. This consistency confirms that the main conclusions are robust across different initial sampling schemes, though remaining variability is influenced by phase-diagram complexity, the underlying model, and the acquisition strategy. In addition, the benchmarks of the random sampling strategy and the results of other baseline tests are shown in <inline-supplementary-material content-type="local-data" mimetype="application/zip" xlink:href="jmi6021-SupplementaryMaterials.zip">Supplementary Materials 4</inline-supplementary-material>.</p>
      </sec>
      <sec id="sec3-3">
        <title>Analysis of uncertainty distributions in compositional space</title>
        <p>Since the US relies heavily on predicted class probabilities, the probability calibration of MLP, GPC, and SVC was assessed using uniformly initialized training sets before active querying. <xref ref-type="fig" rid="fig5">Figure 5A</xref> and <xref ref-type="fig" rid="fig5">B</xref> show that MLP aligns more closely with the diagonal reference line, indicating more reliable probability estimates. In contrast, GPC and SVC tend to be under-confident, especially for Au-Ag-Ge. <xref ref-type="fig" rid="fig5">Figure 5C</xref> and <xref ref-type="fig" rid="fig5">D</xref> present calibration metrics, including the multiclass Brier score and ECE. For Pd-Pt, MLP achieved a Brier score of 0.0484 and an ECE of 0.0082, while GPC and SVC scored 0.0735/0.0760 and 0.0641/0.0656, respectively. For Au-Ag-Ge, the Brier scores were 0.1299, 0.3550, and 0.2579; ECE values were 0.0162, 0.1982, and 0.1840 for MLP, GPC, and SVC. The detailed calculation procedures are provided in the <inline-supplementary-material content-type="local-data" mimetype="application/zip" xlink:href="jmi6021-SupplementaryMaterials.zip">Supplementary Materials 5</inline-supplementary-material>.</p>
        <fig id="fig5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Probability calibration of MLP, GPC, and SVC under uniform initialization. (A and B) Reliability diagrams for Pd-Pt and Au-Ag-Ge systems, respectively; (C) Multiclass Brier scores; (D) ECEs. MLP: Multi-layer perceptron; GPC: Gaussian process classifier; SVC: support vector classifier; ECEs: expected calibration errors.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.5.jpg" />
        </fig>
        <p>
          <xref ref-type="fig" rid="fig6">Figure 6</xref> shows the uncertainty heat maps for the Au-Ag-Ge system after 25 iterations of active learning by examining the uncertainty landscapes generated by different combinations of ML algorithms and uncertainty acquisition strategies. Across all acquisition strategies, the heat maps reveal that regions of high predictive uncertainty consistently emerge along the phase boundaries. Regarding ML algorithm dependence, the MLP maintains relatively consistent performance across all three uncertainty strategies. It produces a smooth multi-class probability distribution using the Softmax function, with uncertainty concentrated near true phase boundaries or sparse regions, while most areas show high confidence with classification probabilities close to 0 or 1. In contrast, GPC and SVC rely on kernel-based decision functions that are sensitive to local data structure. These models tend to produce broader or fragmented uncertainty regions, particularly in areas with sparse training data.</p>
        <fig id="fig6" position="float">
          <label>Figure 6</label>
          <caption>
            <p>Uncertainty heat maps of the Au-Ag-Ge system after 25 iterations under different ML algorithms and uncertainty acquisition strategies. Rows correspond to MLP, GPC, and SVC, while columns correspond to MS, LC, and EA. The color scale represents the normalized uncertainty score, with higher values indicating greater predictive uncertainty. ML: Machine learning; MLP: multi-layer perceptron; GPC: Gaussian process classifier; SVC: support vector classifier; MS: margin sampling; LC: least confidence; EA: entropy-based acquisition.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.6.jpg" />
        </fig>
        <p>In US strategies, EA shows strong localized uncertainties at triple junctions due to the simultaneous and competitive labeling of three-phase regions, leading to redundant queries in the early steps of the active learning process. This slows down global phase boundary coverage and delays model convergence. This behavior arises because EA evaluates uncertainty based on global information of the predicted probability distribution. LC exhibits a more diffuse uncertainty distribution at phase boundaries. The uncertainty concentration is noticeable near the triple junctions, but lower than that of EA. The tendency stems from the fact that LC evaluates uncertainty solely based on the maximum predicted class probability <italic>P<sub>max</sub></italic>, without explicitly distinguishing between binary ambiguity at phase boundaries and multi-class ambiguity at the triple junctions. Therefore, LC partially alleviates the excessive local query observed in EA sampling but remains less effective in promoting uniform coverage of phase-boundary regions. In contrast, MS produces a more uniform uncertainty distribution along phase boundaries, as it evaluates uncertainty primarily through the probability gap between the two most likely classes |<italic>p</italic><sub>1</sub> - <italic>p</italic><sub>2</sub>|, making it more sensitive to the competition between labels of two-phase regions across the phase boundary and less attracted to multi-class ambiguity at triple junctions. Consequently, MS discourages oversampling at triple junctions and accelerates convergence by targeting more informative regions of the phase diagram. In this context, MS is recommended as the preferred uncertainty strategy. Overall, the MLP is recommended among the tested ML algorithms.</p>
        <p>To further illustrate this pattern, SVC is employed as a representative example. <xref ref-type="fig" rid="fig7">Figure 7A</xref> displays the uncertainty heatmaps at iterations 0, 25, and 50 for MS, LC, and EA, respectively. As iterations proceed, the uncertainty distributions gradually shrink toward the true phase boundaries, reflecting the progressive refinement of the classifier. However, the evolution patterns differ markedly among the strategies. Under EA, high-uncertainty regions remain clustered near triple-junction points even at later iterations, indicating persistent localized sampling. LC shows moderate improvement but still lacks clear boundary-focused concentration. In contrast, MS progressively forms narrow and continuous high-uncertainty bands that align closely with the true interfaces, demonstrating more effective guidance for sample selection. <xref ref-type="fig" rid="fig7">Figure 7B</xref> further confirms these trends by visualizing the final queried sample locations. Samples selected under MS are distributed uniformly along the phase boundaries, forming coherent trajectories that cover all critical regions of the compositional simplex. EA and LC, on the other hand, exhibit more scattered and uneven sampling patterns, with noticeable oversampling in junction areas and insufficient coverage of extended boundaries.</p>
        <fig id="fig7" position="float">
          <label>Figure 7</label>
          <caption>
            <p>Evolution of predictive uncertainty and final sampling distributions for SVC under different acquisition strategies. (A) Uncertainty distributions at iterations 0, 25, and 50 for MS, LC, and EA; (B) Sampling distributions after 50 iterations. Triangles denote the uniformly distributed initial labeled samples, whereas circles denote samples subsequently selected through active-learning queries. SVC: Support vector classifier; MS: margin sampling; LC: least confidence; EA: entropy-based acquisition.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.7.jpg" />
        </fig>
        <p>In addition, algorithms based on Euclidean distance, such as SVC, face limitations when dealing with compositional data in the simplex space<sup>[<xref ref-type="bibr" rid="B36">36</xref>]</sup>. Selecting different combinations of independent components in a multi-component composition simplex space can cause significant distortions in phase boundary determination<sup>[<xref ref-type="bibr" rid="B37">37</xref>]</sup>, although these effects are minor in ternary systems. Additive log-ratio (ALR)<sup>[<xref ref-type="bibr" rid="B38">38</xref>]</sup> and isometric log-ratio (ILR)<sup>[<xref ref-type="bibr" rid="B39">39</xref>]</sup> transformations remove the constraint that all compositions sum to 1; however, the introduced complexities reduce sampling efficiency [<xref ref-type="table" rid="t2">Table 2</xref>].</p>
        <table-wrap id="t2">
          <label>Table 2</label>
          <caption>
            <p>Comparison of different feature sets on the initial Macro-F1 scores and iterations required to first reach a Macro-F1 score of 0.95 under MS</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td rowspan="2" />
                <td colspan="2">
                  <bold>GPC</bold>
                </td>
                <td colspan="2">
                  <bold>MLP</bold>
                </td>
                <td colspan="2">
                  <bold>SVC</bold>
                </td>
              </tr>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Initial Macro-F1</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Iterations</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Initial Macro-F1</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Iterations</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Initial Macro-F1</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Iterations</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>ALR</td>
                <td>0.537</td>
                <td>26</td>
                <td>0.721</td>
                <td>29</td>
                <td>0.449</td>
                <td>168</td>
              </tr>
              <tr>
                <td>ILR</td>
                <td>0.507</td>
                <td>20</td>
                <td>0.693</td>
                <td>17</td>
                <td>0.499</td>
                <td>112</td>
              </tr>
              <tr>
                <td>Au, Ag</td>
                <td>0.566</td>
                <td>10</td>
                <td>0.866</td>
                <td>11</td>
                <td>0.819</td>
                <td>34</td>
              </tr>
              <tr>
                <td>Au, Ge</td>
                <td>0.548</td>
                <td>12</td>
                <td>0.868</td>
                <td>10</td>
                <td>0.832</td>
                <td>26</td>
              </tr>
              <tr>
                <td>Ag, Ge</td>
                <td>0.550</td>
                <td>14</td>
                <td>0.858</td>
                <td>8</td>
                <td>0.840</td>
                <td>28</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn>
              <p>Au-Ag, Au-Ge, and Ag-Ge denote the selected pairs of independent composition variables. “Iterations” denotes the number of active-learning iterations required to first reach a Macro-F1 score of 0.95. MS: Margin sampling; GPC: Gaussian process classifier; MLP: multi-layer perceptron; SVC: support vector classifier; ALR: additive log-ratio; ILR: isometric log-ratio.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>
          <xref ref-type="fig" rid="fig8">Figure 8</xref> shows the truncated phase boundaries using the recommended MLP + MS models. Triangles indicate the initially uniform samples, while circles mark later-acquired samples. The sampling points build up along the phase boundaries, especially near triple junctions, emphasizing the strategy’s focus on areas of high predictive uncertainty. The MLP + MS constructions of the Pd-Pt isopleth and the Au-Ag-Ge isothermal section in this work used 120 and 186 labeled samples, achieving Macro-F1 scores of 0.9994 and 0.9814, respectively, which are higher than the stable metrics of 0.980 and 0.927 in our previous work<sup>[<xref ref-type="bibr" rid="B17">17</xref>]</sup>. Meanwhile, the ML-predicted phase boundaries are validated against Thermo-Calc-calculated results by evaluating the mean boundary displacement. This shows mean boundary displacements of 0.0059 for the Pd-Pt binary system and 0.0778 for the Au-Ag-Ge ternary system. Notably, the recommended truncated sample numbers (120 for Pd-Pt and 186 for Au-Ag-Ge) are smaller than the maximum training-sample numbers intentionally displayed in <xref ref-type="fig" rid="fig3">Figure 3</xref> (200 samples) and <xref ref-type="fig" rid="fig4">Figure 4</xref> (396 samples), respectively, illustrating that the proposed framework can reconstruct an unknown phase diagram using substantially fewer samples once the convergence criterion is satisfied. The animations of the adaptive mesh grid in active learning are provided in <inline-supplementary-material content-type="local-data" mimetype="application/zip" xlink:href="jmi6021-SupplementaryMaterials.zip">Supplementary Materials 6</inline-supplementary-material>.</p>
        <fig id="fig8" position="float">
          <label>Figure 8</label>
          <caption>
            <p>Final Sampling distributions and ML-reconstructed phase diagrams obtained using MLP combined with MS for (A) Pd-Pt and (B) Au-Ag-Ge. The left panels show the labeled samples, where triangles represent the uniformly distributed initial samples and circles represent the subsequently queried samples; The right panels show the constructed phase regions, whose interfaces represent the ML-predicted phase boundaries. ML: Machine learning; MLP: multi-layer perceptron; MS: margin sampling.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jmi6021.fig.8.jpg" />
        </fig>
      </sec>
      <sec id="sec3-4">
        <title>Limitations and outlook</title>
        <p>Several limitations of the present work should be acknowledged. (1) The automated labeling procedure remains sensitive to data quality and spatial resolution, with the percentile threshold and derivative-channel weights potentially requiring case-specific tuning; (2) While the differentiability-based concept is mathematically applicable to higher-dimensional composition space, the practical scalability in terms of data-pool size and computational cost has not yet been investigated; (3) Validation against real experimental data in small size is still ongoing and has not been accomplished in this study.</p>
        <p>Our future research will focus on three main directions: (1) systematic studies of dimensionality scalability, including adaptive sampling and cost-control strategies for high-dimensional composition spaces; (2) extension of the MLP + MS framework to more complex systems, leveraging its advantages over SVC in handling high-dimensional Simplex data; and (3) comprehensive experimental validation to establish a robust tool for accelerated phase diagram determination in unknown material systems.</p>
      </sec>
    </sec>
    <sec id="sec4">
      <title>CONCLUSIONS</title>
      <p>In summary, this study presents an integrated active learning framework for efficient phase-boundary prediction by combining automated labeling from multiple activity derivatives with uncertainty-guided sampling on an adaptive compositional grid. The proposed methodology provides a systematic and data-efficient approach for reconstructing complex phase diagrams with minimal labeled thermodynamic data. The main findings of this work are summarized as follows:</p>
      <p>1. Automated labeling using multiple activity derivatives overcomes the limitation of single-derivative signals used in previous work, enabling more complete and reliable identification of phase regions in multicomponent systems.</p>
      <p>2. Margin-based US effectively focuses queries on physically meaningful boundary regions and reduces redundant sampling at triple-junction areas, leading to higher sampling efficiency and faster convergence.</p>
      <p>3. The MLP + MS configuration demonstrates superior performance compared with alternative models and strategies, achieving rapid convergence, high accuracy, and stable behavior for thermodynamics-informed phase-boundary prediction.</p>
    </sec>
  </body>
  <back>
    <sec>
      <title>DECLARATIONS</title>
      <sec>
        <title>Authors’ contributions</title>
        <p>Writing - original draft, visualization, software, investigation: Chen, D.</p>
        <p>Writing - original draft, writing - review and editing, validation, supervision, project administration, methodology, funding acquisition, conceptualization: Xu, G.</p>
        <p>Visualization, software, investigation: Deng, B.</p>
        <p>Validation, resources, project administration, funding acquisition, formal analysis, data curation: Chen, F.</p>
        <p>Software, resources, project administration, funding acquisition: Wang, Z.</p>
        <p>Supervision, resources, project administration: Xuan, C.</p>
        <p>Writing - review and editing, resources, project administration, funding acquisition: Fang, J.</p>
        <p>Writing - review and editing, validation, supervision, resources, formal analysis: Cui, Y.</p>
        <p>Supervision, resources, project administration, funding acquisition: Zhang, A.</p>
      </sec>
      <sec>
        <title>Availability of data and materials</title>
        <p>Code and data to perform ML and reproduce figures can be accessed via <uri xlink:href="https://github.com/Better-Ding/precious_metals_phase_AL">https://github.com/Better-Ding/precious_metals_phase_AL</uri>. Some results supporting the study are presented in the <inline-supplementary-material content-type="local-data" mimetype="application/zip" xlink:href="jmi6021-SupplementaryMaterials.zip">Supplementary Materials</inline-supplementary-material>.</p>
      </sec>
      <sec>
        <title>AI and AI-assisted tools statement</title>
        <p>Not applicable.</p>
      </sec>
      <sec>
        <title>Financial support and sponsorship</title>
        <p>The work was supported by the Major Scientific and Technological Project of Yunnan Precious Metals Laboratory (Grant Nos. YPML-2023050205 and YPML-202405020889) and Jiangsu Provincial Innovation Support Program-”Belt and Road” Innovation Cooperation Key Project (Grant No. BZ2023006). Xu, G. and Chen, F. acknowledge funding from the National Natural Science Foundation of China (Grants No. 52571010, 52371112) for digital design of novel alloys. Xuan, C. also acknowledges the National Natural Science Foundation of China (12211530416, 12002287) and the Research Development Fund of Xi’an Jiaotong-Liverpool University (RDF-22-01-011) for their funding support.</p>
      </sec>
      <sec>
        <title>Conflicts of interest</title>
        <p>Fang, J. and Zhang A. are affiliated with Yunnan Precious Metals Laboratory Co., Ltd., and Wang, Z. is affiliated with MatAi Co. Ltd., while the other authors have declared that they have no conflicts of interest.</p>
      </sec>
      <sec>
        <title>Ethical approval and consent to participate</title>
        <p>Not applicable.</p>
      </sec>
      <sec>
        <title>Consent for publication</title>
        <p>Not applicable.</p>
      </sec>
      <sec>
        <title>Copyright</title>
        <p>© The Author(s) 2026.</p>
      </sec>
	 <sec sec-type="supplementary-material">
      <title>Supplementary Materials</title>
          <supplementary-material content-type="local-data">
                <media xlink:href="jmi6021-SupplementaryMaterials.zip" mimetype="application/zip">
                        <caption>
                                <p>Supplementary Materials</p>
                        </caption>
                </media>
          </supplementary-material>

          </sec> 
    </sec>
    <ref-list>
      <ref id="B1">
        <label>1</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Thermodynamics and its prediction and CALPHAD modeling: review, state of the art, and perspectives</article-title>
          <source>Calphad</source>
          <year>2023</year>
          <volume>82</volume>
          <fpage>102580</fpage>
          <pub-id pub-id-type="doi">10.1016/j.calphad.2023.102580</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B2">
        <label>2</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Arróyave</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Phase stability through machine learning</article-title>
          <source>J Phase Equilib Diffus</source>
          <year>2022</year>
          <volume>43</volume>
          <fpage>606</fpage>
          <lpage>28</lpage>
          <pub-id pub-id-type="doi">10.1007/s11669-022-01009-9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B3">
        <label>3</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Chew</surname>
              <given-names>PY</given-names>
            </name>
            <name>
              <surname>Reinhardt</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Phase diagrams-Why they matter and how to predict them</article-title>
          <source>J Chem Phys</source>
          <year>2023</year>
          <volume>158</volume>
          <fpage>030902</fpage>
          <pub-id pub-id-type="doi">10.1063/5.0131028</pub-id>
          <pub-id pub-id-type="pmid">36681642</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B4">
        <label>4</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Shen</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>The synergy of machine learning and CALPHAD: revitalizing traditional approaches</article-title>
          <source>Comput Mater Sci</source>
          <year>2025</year>
          <volume>258</volume>
          <fpage>113970</fpage>
          <pub-id pub-id-type="doi">10.1016/j.commatsci.2025.113970</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B5">
        <label>5</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Kruglov</surname>
              <given-names>IA</given-names>
            </name>
            <name>
              <surname>Yanilkin</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Oganov</surname>
              <given-names>AR</given-names>
            </name>
            <name>
              <surname>Korotaev</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>Phase diagram of uranium from <italic>ab initio</italic> calculations and machine learning</article-title>
          <source>Phys Rev B</source>
          <year>2019</year>
          <volume>100</volume>
          <fpage>174104</fpage>
          <pub-id pub-id-type="doi">10.1103/physrevb.100.174104</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B6">
        <label>6</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Rosenbrock</surname>
              <given-names>CW</given-names>
            </name>
            <name>
              <surname>Gubaev</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Shapeev</surname>
              <given-names>AV</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Machine-learned interatomic potentials for alloys and alloy phase diagrams</article-title>
          <source>npj Comput Mater</source>
          <year>2021</year>
          <volume>7</volume>
          <fpage>477</fpage>
          <pub-id pub-id-type="doi">10.1038/s41524-020-00477-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B7">
        <label>7</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Zhu</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Sarıtürk</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Arróyave</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Accelerating CALPHAD-based phase diagram predictions in complex alloys using universal machine learning potentials: opportunities and challenges</article-title>
          <source>Acta Mater</source>
          <year>2025</year>
          <volume>286</volume>
          <fpage>120747</fpage>
          <pub-id pub-id-type="doi">10.1016/j.actamat.2025.120747</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B8">
        <label>8</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Xie</surname>
              <given-names>JZ</given-names>
            </name>
            <name>
              <surname>Zhou</surname>
              <given-names>XY</given-names>
            </name>
            <name>
              <surname>Jin</surname>
              <given-names>B</given-names>
            </name>
            <name>
              <surname>Jiang</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Machine learning force field-aided cluster expansion approach to phase diagram of alloyed materials</article-title>
          <source>J Chem Theory Comput</source>
          <year>2024</year>
          <volume>20</volume>
          <fpage>6207</fpage>
          <lpage>17</lpage>
          <pub-id pub-id-type="doi">10.1021/acs.jctc.4c00463</pub-id>
          <pub-id pub-id-type="pmid">38940547</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B9">
        <label>9</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Bocklund</surname>
              <given-names>B</given-names>
            </name>
            <name>
              <surname>Otis</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Egorov</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Obaied</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Roslyakova</surname>
              <given-names>I</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>ESPEI for efficient thermodynamic database development, modification, and uncertainty quantification: application to Cu–Mg</article-title>
          <source>MRS Commun</source>
          <year>2019</year>
          <volume>9</volume>
          <fpage>618</fpage>
          <lpage>27</lpage>
          <pub-id pub-id-type="doi">10.1557/mrc.2019.59</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B10">
        <label>10</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Wu</surname>
              <given-names>B</given-names>
            </name>
            <name>
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Efficient thermodynamic modelling with uncertainty quantification of the Ag–Cu–Co system from its sub-binary systems</article-title>
          <source>Mater Chem Phys</source>
          <year>2023</year>
          <volume>308</volume>
          <fpage>128276</fpage>
          <pub-id pub-id-type="doi">10.1016/j.matchemphys.2023.128276</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B11">
        <label>11</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Deffrennes</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Hallstedt</surname>
              <given-names>B</given-names>
            </name>
            <name>
              <surname>Abe</surname>
              <given-names>T</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Data-driven study of the enthalpy of mixing in the liquid phase</article-title>
          <source>Calphad</source>
          <year>2024</year>
          <volume>87</volume>
          <fpage>102745</fpage>
          <pub-id pub-id-type="doi">10.1016/j.calphad.2024.102745</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B12">
        <label>12</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wu</surname>
              <given-names>B</given-names>
            </name>
            <name>
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Estimating the temperature dependent zero-phase-fraction features in ternary phase diagram via Bayesian approach</article-title>
          <source>Scripta Mater</source>
          <year>2023</year>
          <volume>235</volume>
          <fpage>115615</fpage>
          <pub-id pub-id-type="doi">10.1016/j.scriptamat.2023.115615</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B13">
        <label>13</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Lund</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Braatz</surname>
              <given-names>RD</given-names>
            </name>
            <name>
              <surname>García</surname>
              <given-names>RE</given-names>
            </name>
          </person-group>
          <article-title>Machine learning of phase diagrams</article-title>
          <source>Mater Adv</source>
          <year>2022</year>
          <volume>3</volume>
          <fpage>8485</fpage>
          <lpage>97</lpage>
          <pub-id pub-id-type="doi">10.1039/d2ma00524g</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B14">
        <label>14</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Deffrennes</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Terayama</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Abe</surname>
              <given-names>T</given-names>
            </name>
            <name>
              <surname>Tamura</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>A machine learning–based classification approach for phase diagram prediction</article-title>
          <source>Mater Design</source>
          <year>2022</year>
          <volume>215</volume>
          <fpage>110497</fpage>
          <pub-id pub-id-type="doi">10.1016/j.matdes.2022.110497</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B15">
        <label>15</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>He</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Su</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>C</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Machine learning assisted predictions of multi-component phase diagrams and fine boundary information</article-title>
          <source>Acta Mater</source>
          <year>2022</year>
          <volume>240</volume>
          <fpage>118341</fpage>
          <pub-id pub-id-type="doi">10.1016/j.actamat.2022.118341</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B16">
        <label>16</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Terayama</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Han</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Katsube</surname>
              <given-names>R</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Acceleration of phase diagram construction by machine learning incorporating Gibbs’ phase rule</article-title>
          <source>Scripta Mater</source>
          <year>2022</year>
          <volume>208</volume>
          <fpage>114335</fpage>
          <pub-id pub-id-type="doi">10.1016/j.scriptamat.2021.114335</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B17">
        <label>17</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Li</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Xu</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Chen</surname>
              <given-names>F</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Estimating phase boundaries using sampling and machine learning of derivatives of thermochemistry properties</article-title>
          <source>Calphad</source>
          <year>2026</year>
          <volume>92</volume>
          <fpage>102920</fpage>
          <pub-id pub-id-type="doi">10.1016/j.calphad.2026.102920</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B18">
        <label>18</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Hao</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Ruban</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Xiong</surname>
              <given-names>W</given-names>
            </name>
          </person-group>
          <article-title>CALPHAD modeling based on Gibbs energy functions from zero kevin and improved magnetic model: a case study on the Cr–Ni system</article-title>
          <source>Calphad</source>
          <year>2021</year>
          <volume>73</volume>
          <fpage>102268</fpage>
          <pub-id pub-id-type="doi">10.1016/j.calphad.2021.102268</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B19">
        <label>19</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Liu</surname>
              <given-names>ZK</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <comment><italic>Computational thermodynamics of materials</italic>. Cambridge University Press: Cambridge, 2016.</comment>
          <pub-id pub-id-type="doi">10.1017/CBO9781139018265</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B20">
        <label>20</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Koizumi</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Deffrennes</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Terayama</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Tamura</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Performance of uncertainty-based active learning for efficient approximation of black-box functions in materials science</article-title>
          <source>Sci Rep</source>
          <year>2024</year>
          <volume>14</volume>
          <fpage>27019</fpage>
          <pub-id pub-id-type="doi">10.1038/s41598-024-76800-4</pub-id>
          <pub-id pub-id-type="pmid">39505990</pub-id>
          <pub-id pub-id-type="pmcid">PMC11541781</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B21">
        <label>21</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Terayama</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Tamura</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Nose</surname>
              <given-names>Y</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Efficient construction method for phase diagrams using uncertainty sampling</article-title>
          <source>Phys Rev Mater</source>
          <year>2019</year>
          <volume>3</volume>
          <fpage>033802</fpage>
          <pub-id pub-id-type="doi">10.1103/physrevmaterials.3.033802</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B22">
        <label>22</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wu</surname>
              <given-names>B</given-names>
            </name>
            <name>
              <surname>Zhang</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Zhou</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Zhang</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Uncertainty quantification of phase boundary in a composition-phase map via Bayesian strategies</article-title>
          <source>Phys Rev Mater</source>
          <year>2023</year>
          <volume>7</volume>
          <fpage>025201</fpage>
          <pub-id pub-id-type="doi">10.1103/physrevmaterials.7.025201</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B23">
        <label>23</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Liu</surname>
              <given-names>P</given-names>
            </name>
            <name>
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Qin</surname>
              <given-names>H</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Construction of material phase diagram using different uncertainty estimation strategies</article-title>
          <source>Mater Lett</source>
          <year>2025</year>
          <volume>384</volume>
          <fpage>138036</fpage>
          <pub-id pub-id-type="doi">10.1016/j.matlet.2025.138036</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B24">
        <label>24</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Varivoda</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Dong</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Omee</surname>
              <given-names>SS</given-names>
            </name>
            <name>
              <surname>Hu</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Materials property prediction with uncertainty quantification: a benchmark study</article-title>
          <source>Appl Phys Rev</source>
          <year>2023</year>
          <volume>10</volume>
          <fpage>021409</fpage>
          <pub-id pub-id-type="doi">10.1063/5.0133528</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B25">
        <label>25</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Tian</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Yuan</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Xue</surname>
              <given-names>D</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Role of uncertainty estimation in accelerating materials development via active learning</article-title>
          <source>J Appl Phys</source>
          <year>2020</year>
          <volume>128</volume>
          <fpage>014103</fpage>
          <pub-id pub-id-type="doi">10.1063/5.0012405</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B26">
        <label>26</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Tian</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Yuan</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Xue</surname>
              <given-names>D</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Determining multi-component phase diagrams with desired characteristics using active learning</article-title>
          <source>Adv Sci</source>
          <year>2020</year>
          <volume>8</volume>
          <fpage>2003165</fpage>
          <pub-id pub-id-type="doi">10.1002/advs.202003165</pub-id>
          <pub-id pub-id-type="pmid">33437586</pub-id>
          <pub-id pub-id-type="pmcid">PMC7788591</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B27">
        <label>27</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Dai</surname>
              <given-names>C</given-names>
            </name>
            <name>
              <surname>Glotzer</surname>
              <given-names>SC</given-names>
            </name>
          </person-group>
          <article-title>Efficient phase diagram sampling by active learning</article-title>
          <source>J Phys Chem B</source>
          <year>2020</year>
          <volume>124</volume>
          <fpage>1275</fpage>
          <lpage>84</lpage>
          <pub-id pub-id-type="doi">10.1021/acs.jpcb.9b09202</pub-id>
          <pub-id pub-id-type="pmid">31964140</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B28">
        <label>28</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Lookman</surname>
              <given-names>T</given-names>
            </name>
            <name>
              <surname>Balachandran</surname>
              <given-names>PV</given-names>
            </name>
            <name>
              <surname>Xue</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Yuan</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Active learning in materials science with emphasis on adaptive sampling using uncertainties for targeted design</article-title>
          <source>npj Comput Mater</source>
          <year>2019</year>
          <volume>5</volume>
          <fpage>153</fpage>
          <pub-id pub-id-type="doi">10.1038/s41524-019-0153-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B29">
        <label>29</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Kanopoulos</surname>
              <given-names>N</given-names>
            </name>
            <name>
              <surname>Vasanthavada</surname>
              <given-names>N</given-names>
            </name>
            <name>
              <surname>Baker</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Design of an image edge detection filter using the Sobel operator</article-title>
          <source>IEEE J Solid State Circuits</source>
          <volume>23</volume>
          <fpage>358</fpage>
          <lpage>67</lpage>
          <pub-id pub-id-type="doi">10.1109/4.996</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B30">
        <label>30</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Turchi</surname>
              <given-names>PEA</given-names>
            </name>
            <name>
              <surname>Drchal</surname>
              <given-names>V</given-names>
            </name>
            <name>
              <surname>Kudrnovský</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Stability and ordering properties of fcc alloys based on Rh, Ir, Pd, and Pt</article-title>
          <source>Phys Rev B</source>
          <year>2006</year>
          <volume>74</volume>
          <fpage>064202</fpage>
          <pub-id pub-id-type="doi">10.1103/physrevb.74.064202</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B31">
        <label>31</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Tang</surname>
              <given-names>C</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Zhou</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Jin</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Thermodynamic description of the Au–Ag–Ge ternary system</article-title>
          <source>Thermochim Acta</source>
          <year>2011</year>
          <volume>512</volume>
          <fpage>240</fpage>
          <lpage>6</lpage>
          <pub-id pub-id-type="doi">10.1016/j.tca.2010.11.003</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B32">
        <label>32</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Deringer</surname>
              <given-names>VL</given-names>
            </name>
            <name>
              <surname>Bartók</surname>
              <given-names>AP</given-names>
            </name>
            <name>
              <surname>Bernstein</surname>
              <given-names>N</given-names>
            </name>
            <name>
              <surname>Wilkins</surname>
              <given-names>DM</given-names>
            </name>
            <name>
              <surname>Ceriotti</surname>
              <given-names>M</given-names>
            </name>
            <name>
              <surname>Csányi</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>Gaussian process regression for materials and molecules</article-title>
          <source>Chem Rev</source>
          <year>2021</year>
          <volume>121</volume>
          <fpage>10073</fpage>
          <lpage>141</lpage>
          <pub-id pub-id-type="doi">10.1021/acs.chemrev.1c00022</pub-id>
          <pub-id pub-id-type="pmid">34398616</pub-id>
          <pub-id pub-id-type="pmcid">PMC8391963</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B33">
        <label>33</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Rumelhart</surname>
              <given-names>DE</given-names>
            </name>
            <name>
              <surname>Hinton</surname>
              <given-names>GE</given-names>
            </name>
            <name>
              <surname>Williams</surname>
              <given-names>RJ</given-names>
            </name>
          </person-group>
          <article-title>Learning representations by back-propagating errors</article-title>
          <source>Nature</source>
          <year>1986</year>
          <volume>323</volume>
          <fpage>533</fpage>
          <lpage>6</lpage>
          <pub-id pub-id-type="doi">10.1038/323533a0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B34">
        <label>34</label>
        <nlm-citation publication-type="confproc">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Yan</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <comment>A multi-candidate batch mode active learning approach. In <italic>Proceedings of the 2023 15th International Conference on Machine Learning and Computing</italic>. Association for Computing Machinery; 2023. pp 528-33.</comment>
          <pub-id pub-id-type="doi">10.1145/3587716.3587803</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B35">
        <label>35</label>
        <nlm-citation publication-type="confproc">
          <person-group person-group-type="author">
            <name>
              <surname>Yacouby</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Axman</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <comment>Probabilistic extension of Precision, Recall, and F1 score for more thorough evaluation of classification models. In <italic>Proceedings of the First Workshop on Evaluation and Comparison of NLP Systems</italic>. Association for Computational Linguistics; 2020. pp 79-91.</comment>
          <pub-id pub-id-type="doi">10.18653/v1/2020.eval4nlp-1.9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B36">
        <label>36</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Heese</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Schmid</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Walczak</surname>
              <given-names>M</given-names>
            </name>
            <name>
              <surname>Bortz</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Calibrated simplex-mapping classification</article-title>
          <source>PLoS One</source>
          <year>2023</year>
          <volume>18</volume>
          <fpage>e0279876</fpage>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0279876</pub-id>
          <pub-id pub-id-type="pmid">36649243</pub-id>
          <pub-id pub-id-type="pmcid">PMC9844900</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B37">
        <label>37</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Krajewski</surname>
              <given-names>AM</given-names>
            </name>
            <name>
              <surname>Beese</surname>
              <given-names>AM</given-names>
            </name>
            <name>
              <surname>Reinhart</surname>
              <given-names>WF</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>Efficient generation of grids and traversal graphs in compositional spaces towards exploration and path planning</article-title>
          <source>npj Unconv Comput</source>
          <year>2024</year>
          <volume>1</volume>
          <fpage>12</fpage>
          <pub-id pub-id-type="doi">10.1038/s44335-024-00012-2</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B38">
        <label>38</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Greenacre</surname>
              <given-names>M</given-names>
            </name>
            <name>
              <surname>Grunsky</surname>
              <given-names>E</given-names>
            </name>
            <name>
              <surname>Bacon-Shone</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Erb</surname>
              <given-names>I</given-names>
            </name>
            <name>
              <surname>Quinn</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>Aitchison’s compositional data analysis 40 years on: a reappraisal</article-title>
          <source>Statist Sci</source>
          <year>2023</year>
          <volume>38</volume>
          <fpage>386</fpage>
          <lpage>410</lpage>
          <pub-id pub-id-type="doi">10.1214/22-sts880</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B39">
        <label>39</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Egozcue</surname>
              <given-names>JJ</given-names>
            </name>
            <name>
              <surname>Pawlowsky-Glahn</surname>
              <given-names>V</given-names>
            </name>
            <name>
              <surname>Mateu-Figueras</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Barceló-Vidal</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Isometric logratio transformations for compositional data analysis</article-title>
          <source>Math Geol</source>
          <year>2003</year>
          <volume>35</volume>
          <fpage>279</fpage>
          <lpage>300</lpage>
          <pub-id pub-id-type="doi">10.1023/a:1023818214614</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>