<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.0 20120330//EN" "http://jats.nlm.nih.gov/publishing/1.0/JATS-journalpublishing1.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="1.0" article-type="research-article">
  <front>
    <journal-meta>
      <journal-id journal-id-type="nlm-ta">AI Agent</journal-id>
      <journal-id journal-id-type="publisher-id">aiagent</journal-id>
      <journal-title-group>
        <journal-title>AI Agent</journal-title>
      </journal-title-group>
      <issn pub-type="epub">3070-3719</issn>
      <publisher>
        <publisher-name>OAE Publishing Inc.</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.20517/aiagent.2026.31</article-id>
      <article-id pub-id-type="publisher-id">AIAgent-2026-31</article-id>
      <article-categories>
        <subj-group>
          <subject>Original Article</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>When code does not run: reproducibility challenges in materials machine learning benchmarks</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <name>
            <surname>Lyu</surname>
            <given-names>Bohui</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Bonini</surname>
            <given-names>John</given-names>
          </name>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Zhang</surname>
            <given-names>Mao</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Sadeghi</surname>
            <given-names>Amin</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Jaberi</surname>
            <given-names>Ali</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Hattrick-Simpers</surname>
            <given-names>Jason</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Choudhary</surname>
            <given-names>Kamal</given-names>
          </name>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
          <xref ref-type="aff" rid="I5">
            <sup>5</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Wines</surname>
            <given-names>Daniel</given-names>
          </name>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <name>
            <surname>Li</surname>
            <given-names>Kangming</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1">*</xref>
        </contrib>
      </contrib-group>
      <aff id="I1"><sup>1</sup>Division of Physical Science and Engineering, King Abdullah University of Science and Technology (KAUST), Thuwal 23955-6900, Saudi Arabia.</aff>
      <aff id="I2"><sup>2</sup>Material Measurement Laboratory, National Institute of Standards and Technology, Gaithersburg, MD 20899, USA.</aff>
      <aff id="I3"><sup>3</sup>Department of Materials Science and Engineering, University of Toronto, Toronto M5S 3E4, Canada.</aff>
      <aff id="I4"><sup>4</sup>Acceleration Consortium, University of Toronto, Toronto M5S 3E4, Canada.</aff>
      <aff id="I5"><sup>5</sup>Department of Electrical and Computer Engineering, Whiting School of Engineering, Johns Hopkins University, Baltimore, MD 21218, USA.</aff>
      <author-notes>
        <corresp id="cor1">Correspondence to: Prof. Kangming Li, Division of Physical Science and Engineering, King Abdullah University of Science and Technology (KAUST), Thuwal 23955-6900, Saudi Arabia. E-mail: <email>kangming.li@kaust.edu.sa</email></corresp>
        <fn fn-type="other">
          <p><bold>Received:</bold> 22 Jun 2026 | <bold>First Decision:</bold> 22 Jul 2026 | <bold>Revised:</bold> 17 Aug 2026 | <bold>Accepted:</bold> 24 Aug 2026 | <bold>Published:</bold> 28 Sep 2026</p>
        </fn>
        <fn fn-type="other">
          <p><bold>Academic Editor:</bold> Hao Li | <bold>Copy Editor:</bold> Tong Wang | <bold>Production Editor:</bold> Tong Wang</p>
        </fn>
      </author-notes>
      <pub-date pub-type="ppub">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>28</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>2</volume>
      <issue>3</issue>
      <elocation-id>23</elocation-id>
      <permissions>
        <copyright-statement>© The Author(s) 2026.</copyright-statement>
        <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
          <license-p>© The Author(s) 2026.<bold>Open Access</bold>This article is licensed under a Creative Commons Attribution 4.0 International License (<uri xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</uri>), which permits unrestricted use, sharing, adaptation, distribution and reproduction in any medium or format, for any purpose, even commercially, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons license, and indicate if changes were made.</license-p>
        </license>
      </permissions>
      <abstract>
        <p>Reproducible benchmarks are essential for advancing data-driven materials research by standardizing datasets, prediction tasks, and evaluation protocols. In machine learning benchmarks, reproducibility is often assumed to follow from releasing source code and documented software dependencies. However, dependency specifications must be sufficiently complete to reconstruct an executable software environment from a defined, isolated starting point. Because modern package ecosystems are complex and evolve over time, automated environment reconstruction and sandboxed executability checks are necessary prerequisites for assessing numerical reproducibility. Here, we systematically evaluate this aspect of reproducibility using MatBench, a benchmark platform for materials property prediction. Using the original dependency metadata without modification, only 2 of 28 submissions produced environments that could be installed and pass import checks. Rule-based and human-assisted remediation increased this number to 26 of 28. Among these, 13 submissions successfully completed a selected representative benchmark task and generated outputs, corresponding to 46.4% of all evaluated submissions. For these successfully re-executed submissions, the regenerated five-fold mean root mean square error (RMSE) values were close to the originally reported results. Similar executability issues were also observed on the JARVIS-Leaderboard. These findings demonstrate that reconstructing runnable environments from currently provided benchmark metadata remains challenging. We identify common causes of failure and propose platform-level practices for improving the long-term reproducibility and executability of machine learning benchmarks in materials informatics.</p>
      </abstract>
      <kwd-group>
        <kwd>Machine learning benchmark</kwd>
        <kwd>computational reproducibility</kwd>
        <kwd>software environment</kwd>
        <kwd>materials science</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec1">
      <title>INTRODUCTION</title>
      <p>A description of the synthesis of a particular molecule, given as a set of instructions and reactants, which only reliably yields the specified products when performed with one beaker the chemist never washes, would not be recognized as complete or scientifically reproducible. A description of a computational workflow that only reliably runs on the compute resources where it was developed should be regarded similarly. Reproducibility is fundamental to the reliability of scientific conclusions, accelerates algorithmic iteration, and enables standardized head-to-head comparisons across methods<sup>[<xref ref-type="bibr" rid="B1">1</xref>,<xref ref-type="bibr" rid="B2">2</xref>]</sup>.</p>
      <p>Accordingly, ensuring the reproducibility of computational workflows and predictive results has become imperative, as underscored by the FAIR (Findability, Accessibility, Interoperability, and Reusability) principles for digital assets<sup>[<xref ref-type="bibr" rid="B3">3</xref>]</sup>.</p>
      <p>With the rapid development of machine learning, its applications in materials science have expanded markedly. Tasks such as rapid screening of <italic>in silico</italic>-generated materials<sup>[<xref ref-type="bibr" rid="B4">4</xref>-<xref ref-type="bibr" rid="B6">6</xref>]</sup>, acceleration of molecular simulations<sup>[<xref ref-type="bibr" rid="B7">7</xref>,<xref ref-type="bibr" rid="B8">8</xref>]</sup>, inverse design of novel materials<sup>[<xref ref-type="bibr" rid="B9">9</xref>-<xref ref-type="bibr" rid="B11">11</xref>]</sup>, and autonomous experimentation<sup>[<xref ref-type="bibr" rid="B12">12</xref>,<xref ref-type="bibr" rid="B13">13</xref>]</sup> have substantially advanced the field, enabling the discovery and development of new materials. The reproducibility of machine learning workflows in materials science remains an ongoing challenge and has received increasing attention<sup>[<xref ref-type="bibr" rid="B14">14</xref>,<xref ref-type="bibr" rid="B15">15</xref>]</sup>. To support transparent comparison and algorithm reuse, benchmark platforms have been developed that standardize datasets and evaluation pipelines. Analogous to benchmark platforms in computer science<sup>[<xref ref-type="bibr" rid="B14">14</xref>-<xref ref-type="bibr" rid="B16">16</xref>]</sup>, AI for materials science has produced a growing suite of benchmarking resources<sup>[<xref ref-type="bibr" rid="B17">17</xref>-<xref ref-type="bibr" rid="B19">19</xref>]</sup>. These platforms typically require source code together with software requirements in accompanying metadata. However, the executability of these resources is often overlooked, whereas reproducing results typically requires setting up the correct software environment to run the source code. If that environment cannot be established, the available data and code alone are insufficient to reproduce the reported results.</p>
      <p>Among benchmark platforms for materials science, MatBench is a machine-learning benchmark that standardizes datasets, prediction tasks, evaluation splits, and leaderboards for comparing materials-property prediction models. MatBench comprises 13 supervised tasks covering composition- and structure-based prediction of key inorganic materials properties [<xref ref-type="fig" rid="fig1">Figure 1</xref>]. Its workflow uses predefined five-fold cross-validation, with model predictions and evaluation metrics recorded through the MatBench framework. The platform supports comparison across diverse models, from descriptor-based pipelines to advanced crystal graph neural networks.</p>
      <fig id="fig1" position="float">
        <label>Figure 1</label>
        <caption>
          <p>MatBench infrastructure and submission protocol.</p>
        </caption>
        <graphic xlink:href="aiagent2031.fig.1.jpg"/>
      </fig>
      <p>As a case study, we attempted to reproduce the reported results on MatBench by re-executing the original evaluation pipeline of 28 algorithms using the provided source code and documented environment specifications. Unexpectedly, the process was far from straightforward: we encountered numerous obstacles, most of which were related to the inability to instantiate Python environments from the specifications provided by the platforms. While we managed to set up most environments by relaxing software version constraints, the resulting environments often differed from those originally used to generate the uploaded results, which in turn prevented the successful execution of the source code or reproduction of the uploaded results. We therefore categorized the algorithms according to the difficulty of environment reconstruction and summarized the problems encountered at different stages of the reproduction process. Accordingly, we identify common obstacles to re-execution and propose platform-level recommendations for improving benchmark reproducibility.</p>
    </sec>
    <sec id="sec2">
      <title>METHOD</title>
      <sec id="sec2-1">
        <title>Assessment framework</title>
        <p>Each MatBench submission contains three principal components: a metadata file (info.json) describing the algorithm and its software requirements, source code for training and evaluating the model, and a compressed result file (results.json.gz) containing the submitted predictions and benchmark metrics [<xref ref-type="fig" rid="fig1">Figure 1</xref>]. We evaluated 28 submitted algorithms using their publicly available metadata and source code, while treating the submitted result files as the reference for numerical comparison.</p>
        <p>The assessment comprised three sequential levels. First, environment reconstruction tested whether the dependency information supplied with a submission was sufficient to establish an import-valid Python environment. An environment was considered successfully reconstructed when, under at least one tested Python version and installation route, dependency installation completed and all imports identified in the submitted execution scripts could be loaded without error. This criterion evaluates environment availability and import compatibility but does not establish that the complete benchmark workflow is executable.</p>
        <p>Second, code executability tested whether the submitted source code could complete a selected representative MatBench task and generate benchmark outputs in the reconstructed environment. Third, numerical result reproducibility quantified the difference between the regenerated benchmark metric and the value recorded in the original submission. These three levels were evaluated separately because successful environment reconstruction does not necessarily guarantee successful code execution, and successful execution does not necessarily guarantee numerical agreement with the reported result. The overall workflow is illustrated in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p>
        <fig id="fig2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Operational workflow used in this study to assess practical reproducibility on MatBench. LLM: Large language model.</p>
          </caption>
          <graphic xlink:href="aiagent2031.fig.2.jpg"/>
        </fig>
      </sec>
      <sec id="sec2-2">
        <title>Environment reconstruction</title>
        <p>Environment reconstruction was conducted in clean, isolated Linux runners using GitHub Actions. For each submission, the dependency specifications in info.json were tested across Python versions 3.6-3.12 using pip and conda-based installation routes. After installation, the pipeline automatically scanned the source code to collect all import statements and performed an import-based sanity check. The reconstruction procedure comprised three stages of progressively increasing intervention.</p>
        <p>In Stage 1, dependencies were installed using the author-provided package names and version constraints without modification. Submissions that completed installation and passed the import check were assigned to Category 1 (Cat. 1).</p>
        <p>Submissions that failed Stage 1 proceeded to Stage 2, in which predefined error-guided rules were used to relax unavailable or conflicting version constraints. The rules operated only on dependency specifications and did not modify the submitted model or evaluation code. Submissions that passed the import check after this automated remediation were assigned to Cat. 2. The error-classification rules, retry procedure, and workflow pseudocode are provided in <inline-supplementary-material content-type="local-data" mimetype="application/pdf" xlink:href="aiagent2031-SupplementaryMaterials.zip">Supplementary Method 1</inline-supplementary-material>.</p>
        <p>Submissions that remained unresolved proceeded to Stage 3. At this stage, the metadata and execution scripts were reviewed manually to identify omitted dependencies and potential incompatibilities. Package versions, installation order, and the use of pip, conda, or a combination of the two were adjusted where necessary. A large language model (LLM) was used as a human-supervised troubleshooting aid for selected difficult cases, but all suggested changes were reviewed, implemented, and validated by the authors. No submitted model or evaluation code was modified during environment reconstruction. The complete intervention protocol, stopping criteria, and LLM-assisted procedure are described in <inline-supplementary-material content-type="local-data" mimetype="application/pdf" xlink:href="aiagent2031-SupplementaryMaterials.zip">Supplementary Method 2</inline-supplementary-material>.</p>
        <p>The four environment categories were assigned according to the minimum level of intervention required to satisfy the environment-reconstruction criterion:</p>
        <p>These categories describe the reconstruction pathway and the minimum intervention required; they do not indicate whether the source code subsequently executed successfully or whether the regenerated numerical results agreed with the original submission. Cat. 4 likewise denotes failure under the predefined testing protocol and should not be interpreted as proof that reconstruction is impossible under all configurations.</p>
      </sec>
      <sec id="sec2-3">
        <title>Code re-execution and numerical comparison</title>
        <p>Submissions in Cat. 1-3 were advanced to code re-execution. Their reconstructed environments were installed on an High-Performance Computing (HPC) system running Rocky Linux 9.4 (Blue Onyx), equipped with an NVIDIA A100-SXM4-80GB Graphics Processing Unit (GPU) and CUDA 12.8. Environments that passed the GitHub Actions validation could also be installed on this system.</p>
        <p>To limit the computational cost while covering both MatBench input classes, we selected the smallest representative task from each class: steels for composition-based models and jdft2d for structure-based models. Each submission was evaluated on the applicable task according to its input type. Any submission-specific exceptions and the task selected for each algorithm are provided in <inline-supplementary-material content-type="local-data" mimetype="application/pdf" xlink:href="aiagent2031-SupplementaryMaterials.zip">Supplementary Table 1</inline-supplementary-material>, available as a separate Excel file. Code re-execution was considered successful when the submitted workflow completed the selected task and generated predictions and benchmark metrics in the expected MatBench output format. Runtime failures and their causes were recorded separately from failures of environment reconstruction.</p>
        <p>For submissions that completed re-execution, numerical agreement was quantified using the relative difference between the regenerated and originally reported five-fold mean root mean square error (RMSE):</p>
          <p><disp-formula> <label>(1)</label> <tex-math id="E1"> $$ \Delta_{\text {RMSE }}(\%)=\frac{\text { RMSE }_{\text {re-executed }}-\text { RMSE }_{\text {reported }}}{\text { RMSE }_{\text {reported }}} \times 100 . $$ </tex-math></disp-formula></p>
          <p>Here, RMSE<sub>reported</sub> is the mean RMSE across the five predefined cross-validation folds recorded in the original submission, and RMSE<sub>re-executed</sub> is the corresponding value obtained through re-execution. Positive values indicate higher error in the re-executed result, whereas negative values indicate lower error. We report the signed and absolute differences rather than imposing an arbitrary binary threshold for numerical reproducibility.</p>
        </sec>
      </sec>
    <sec id="sec3">
      <title>RESULTS AND DISCUSSION</title>
      <sec id="sec3-1">
        <title>MatBench</title>
        <sec id="sec3-1-1">
          <title>Environment reconstruction</title>
        </sec>
        <sec id="sec3-1-1-1">
          <title>Stage 1</title>
          <p>We first attempted to build the environments by strictly following the dependency specifications provided by the authors. This approach represents the most straightforward workflow from an end user’s perspective. Among 28 distinct algorithms, only two of them could be installed and all functions required by the execution scripts could be successfully imported. The Auto-sklearn submission provides an environment backup file (environment.yml), enabling the environment to be installed directly, whereas the Ax_10_90_CrabNet_v1.2.7 implementation installs cleanly on Python 3.7. For the remaining cases, neither conda nor pip could satisfy all version constraints. The failures mainly fall into three categories [<xref ref-type="table" rid="t1">Table 1</xref>]: (i) For CrabNet the metadata was not machine-readable in practice, preventing our scripts from extracting the necessary information; for this case, we attempted manual environment reconstruction in Stage 3; (ii) Unavailability of specific pinned package versions. The most frequent error arose from a hard requirement on matbench v0.1.0; this version is no longer available in standard package repositories, leading to installation failures in 11 environments; (iii) Version conflicts among dependencies. These errors commonly occur when the version constraints for packages such as matbench, scikit-learn, pymatgen, and matminer cannot be satisfied simultaneously, resulting in environment reconstruction failures. The latter two types of errors may be attributable to the age of the submissions, all of which were uploaded two to three years ago. Over this period, the underlying package ecosystem has evolved substantially: some older versions have been deprecated or removed from distribution channels, and newer releases have introduced incompatibilities and dependency conflicts. We therefore infer that these changes collectively contributed to the installation and runtime errors observed in Stage 1. These unresolved issues motivated a second stage focused on error-guided automatic remediation.</p>
          <table-wrap id="t1">
            <label>Table 1</label>
            <caption>
              <p>Representative package installation errors</p>
            </caption>
            <table frame="hsides" rules="groups">
  <tbody>
    <tr>
      <td>
        <bold>Error type</bold>
      </td>
      <td>
        <bold>Example output</bold>
      </td>
    </tr>
    <tr>
      <td>Metadata is not machine-readable</td>
      <td>CrabNet<break/>Extracting packages from info.json...<break/>AttributeError: ‘str’ object has no attribute ‘get’</td>
    </tr>
    <tr>
      <td>Version not found</td>
      <td>automatminer_expressv2020, darwin, cgcnnv2019, GN-OA, <italic>etc.</italic><break/>ERROR: Could not find a version that satisfies the requirement matbench==0.1.0 (from versions: 0.2, 0.3, 0.4, 0.5, 0.6)<break/>modnet_v0.1.10<break/>ERROR: Could not find a version that satisfies the requirement modnet==0.1.10 (from versions: 0.1.1, 0.1.2, 0.1.3, 0.1.4, 0.1.5, 0.1.6, 0.1.7, 0.1.8, 0.1.9, 0.1.10.dev0, 0.1.11.dev0, 0.1.11, 0.1.12.dev0, 0.1.12, 0.1.13, 0.2.0, 0.2.1, 0.3.0, 0.4.0, 0.4.1, 0.4.2, 0.4.3, 0.4.4, 0.4.5)<break/>ERROR: No matching distribution found for modnet==0.1.10</td>
    </tr>
    <tr>
      <td>Dependency conflict</td>
      <td>modnet_v0.1.12<break/>The conflict is caused by:<break/>modnet 0.1.12 depends on pymatgen &lt;2020.9 and >= 2020<break/>matbench 0.2 depends on pymatgen==2021.2.16<break/>TPOT<break/>The conflict is caused by:<break/>The user requested scikit-learn==1.2.2<break/>tpot 0.11.7 depends on scikit-learn>=0.22.0<break/>matbench 0.6 depends on scikit-learn==1.0.1<break/>matbench 0.5 depends on scikit-learn==1.0<break/>matbench 0.4 depends on scikit-learn==1.0<break/>matbench 0.3 depends on scikit-learn==0.24.2<break/>matbench 0.2 depends on scikit-learn==0.24.1<break/>Ax_CrabNet_v1.2.1<break/>ERROR: Cannot install ax-platform==0.2.3, crabnet==1.2.1, matbench==0.5 and scikit_learn==1.0.2 because these package versions have conflicting dependencies.<break/>The conflict is caused by:<break/>The user requested scikit_learn==1.0.2<break/>ax-platform 0.2.3 depends on scikit-learn<break/>crabnet 1.2.1 depends on scikit-learn<break/>matbench 0.5 depends on scikit-learn==1.0</td>
    </tr>
  </tbody>
</table>
          </table-wrap>
        </sec>
        <sec id="sec3-1-1-2">
          <title>Stage 2</title>
          <p>Based on the error messages from the algorithms that failed in Stage 1, we hypothesized that appropriately relaxing the version constraints of the implicated packages could alleviate many installation failures. Therefore, we extracted the error logs and attempted to fix the environment reconstruction process automatically, which we refer to as Stage 2. Following these error-guided fixes, 15 of the remaining 26 algorithms installed successfully and completed the import checks without errors. The detailed procedures and rationale are provided in the Methods section. For the remaining 11 algorithms, we observed two principal issues [<xref ref-type="table" rid="t2">Table 2</xref>]. (i) Several packages invoked by the code were not fully enumerated in the metadata files, resulting in import errors; (ii) Even after removing all version pins, some environments still exhibited dependency conflicts that prevented installation. These issues are difficult to resolve reliably through a fully automated workflow. Therefore, for the remaining algorithms, we adopted an iterative troubleshooting strategy that combined manual intervention with LLM-assisted guidance to provision the environments.</p>
          <table-wrap id="t2">
            <label>Table 2</label>
            <caption>
              <p>Representative package installation errors</p>
            </caption>
            <table frame="hsides" rules="groups">
  <tbody>
    <tr>
      <td>
        <bold>Error type</bold>
      </td>
      <td>
        <bold>Example output</bold>
      </td>
    </tr>
    <tr>
      <td>Missing packages in the metadata files</td>
      <td>Matformer<break/>from torch_scatter import gather_csr, scatter, segment_csr ModuleNotFoundError: No module named ‘torch_scatter<break/>lattice_xgboost<break/>import google<break/>ModuleNotFoundError: No module named ‘google’</td>
    </tr>
    <tr>
      <td>Without all version pins, some environments still exhibited dependency conflicts</td>
      <td>automatminer_expressv2020<break/>The conflict is caused by:<break/>automatminer 1.0.0.20191110 depends on matminer==0.6.2<break/>matbench 0.6 depends on matminer==0.7.4<break/>automatminer 1.0.0.20191110 depends on matminer==0.6.2<break/>matbench 0.5 depends on matminer==0.7.4<break/>automatminer 1.0.0.20191110 depends on matminer==0.6.2<break/>matbench 0.4 depends on matminer==0.7.4<break/>automatminer 1.0.0.20191110 depends on matminer==0.6.2<break/>matbench 0.3 depends on matminer==0.7.3<break/>automatminer 1.0.0.20191110 depends on matminer==0.6.2<break/>matbench 0.2 depends on matminer==0.6.5<break/>coGN<break/>ERROR: pip’s dependency resolver does not currently take into account all the packages that are installed. This behavior is the source of the following dependency conflicts<break/>kgcnn 3.0.0 requires numpy>=1.23.0, but you have numpy 1.22.4 which is incompatible. kgcnn 3.0.0 requires scikit-learn>=1.1.3, but you have scikit-learn 1.0.1 which is incompatible<break/>kgcnn 3.0.0 requires scipy>=1.9.3, but you have scipy 1.7.3 which is incompatible<break/>pyxtal 1.1.3 requires pymatgen>=2024.3.1, but you have pymatgen 2023.9.25 which is incompatible<break/>tensorflow 2.20.0 requires numpy>=1.26.0, but you have numpy 1.22.4 which is incompatible<break/>from numpy._typing import ArrayLike<break/>ModuleNotFoundError: No module named ‘numpy._typing’</td>
    </tr>
  </tbody>
</table>
          </table-wrap>
        </sec>
        <sec id="sec3-1-1-3">
          <title>Stage 3</title>
          <p>As shown in <xref ref-type="table" rid="t2">Table 2</xref>, incomplete dependency information caused import failures for several algorithms. We therefore identified and manually installed the missing packages before repeating the import checks. For submissions whose metadata could not be reliably parsed by our automated workflow [<xref ref-type="table" rid="t1">Table 1</xref>], the environments were reconstructed manually following the authors instructions. Where necessary, this process was supported by LLM-assisted troubleshooting, particularly to identify compatible package versions and adjust the installation order. Stage 3 successfully recovered environments for 9 additional algorithms, while 2 remained unresolved after all permitted troubleshooting attempts. The final classification is summarized in <xref ref-type="table" rid="t3">Table 3</xref>.</p>
          <table-wrap id="t3">
            <label>Table 3</label>
            <caption>
              <p>Categories of MatBench algorithms based on reproducibility of the Python environment</p>
            </caption>
            <table frame="hsides" rules="groups">
  <tbody>
    <tr>
      <td>
        <bold>Categories</bold>
      </td>
      <td>
        <bold>Algorithms</bold>
      </td>
    </tr>
    <tr>
      <td>Cat. 1: Can set up the environment directly (2/28)</td>
      <td>Auto-sklearn, Ax_10_90_CrabNet_v1.2.7</td>
    </tr>
    <tr>
      <td>Cat. 2: Can set up the environment by auto-fixing (15/28)</td>
      <td>alignn, Ax_CrabNet_v1.2.1, Ax_SAASBO_CrabNet_v1.2.7, DimeNetPP_kgcnn_v2.1.0, dummy, Finder_v1.2_composition, Finder_v1.2_structure, GN-OA, gptchem, MegNet_kgcnn_v2.1.0, modnet_v0.1.10, modnet_v0.1.12, RFLR, SchNet_kgcnn_v2.1.0, TPOT</td>
    </tr>
    <tr>
      <td>Cat. 3: Can set up the environment with human effort (9/28)</td>
      <td>automatminer_expressv2020, CrabNet, cgcnnv2019, CrabNet_v1.2.1, darwin, lattice_xgboost, matformer, coGN, coNGN</td>
    </tr>
    <tr>
      <td>Cat. 4: Cannot set up environment despite substantial effort (2/28)</td>
      <td>DeeperGATGNN, rf</td>
    </tr>
  </tbody>
</table>
          </table-wrap>
        </sec>
        <sec id="sec3-1-2">
          <title>Reproducing the benchmark results</title>
          <p>To set up the Python environments successfully, we adjusted version specifications for selected packages, either automatically through our scripts or manually, which resolved many installation errors. However, such modifications may introduce potential risks: the execution scripts may become incompatible, or replacing author-specified algorithm versions may affect benchmark outcomes.</p>
          <p>Execution of the provided code showed that only half of the algorithms with successfully reconstructed environments produced benchmark outputs. For the remaining algorithms, runtime failures occurred because the required version of a key dependency could not be installed. As shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>, these failures resulted from renamed Application Programming Interfaces (APIs), changed entry points after package updates, or relocated modules after code refactoring. For example, the training script in alignn was previously invoked as train_folder, whereas in newer releases it has been renamed to train_alignn. Similarly, modnet imports Structure directly from pymatgen, whereas, in the reconstructed environment, the installed pymatgen version exposes Structure only through pymatgen.core.structure. In addition, gptchem attempts to access an OpenAI token resource; however, the referenced API endpoint is no longer available, resulting in an HTTP 404 (“Not Found”) error. In principle, one could modify the execution scripts or patch the corresponding package source code to restore compatibility; however, such interventions are beyond the scope of this manuscript. We also identified a special case that led to irreproducibility for reasons unrelated to dependency resolution: the run.py file in the cgcnnv2019 submission is empty. Thus, even though the environment can be installed, the benchmark tasks cannot be executed. This issue could be addressed by correcting the submission files. The names of the 26 algorithms and their corresponding code executability outcomes are summarized in <xref ref-type="table" rid="t4">Table 4</xref>.</p>
          <fig id="fig3" position="float">
            <label>Figure 3</label>
            <caption>
              <p>Code execution results and runtime error messages. API: Application Programming Interface.</p>
            </caption>
            <graphic xlink:href="aiagent2031.fig.3.jpg"/>
          </fig>
          <table-wrap id="t4">
            <label>Table 4</label>
            <caption>
              <p>Executability of the submitted code in the 26 reconstructed environments</p>
            </caption>
            <table frame="hsides" rules="groups">
  <tbody>
    <tr>
      <td>
        <bold>Executability</bold>
      </td>
      <td>
        <bold>Algorithms</bold>
      </td>
    </tr>
    <tr>
      <td>Can reproduce the result (13/26)</td>
      <td>Auto-sklearn, Ax_10_90_CrabNet_v1.2.7, Ax_CrabNet_v1.2.1, CrabNet, CrabNet_v1.2.1, coGN, coNGN, DimeNetPP_kgcnn_v2.1.0, dummy, Finder_v1.2_composition, Finder_v1.2_structure, RFLR, lattice_xgboost</td>
    </tr>
    <tr>
      <td>Cannot obtain the benchmark result (13/26)</td>
      <td>alignn, automatminer_expressv2020, Ax_SAASBO_CrabNet_v1.2.7, cgcnnv2019, darwin, GN-OA, gptchem, matformer, MegNet_kgcnn_v2.1.0, modnet_v0.1.10, modnet_v0.1.12, SchNet_kgcnn_v2.1.0, TPOT</td>
    </tr>
  </tbody>
</table>
          </table-wrap>
          <p>The comparison between the results obtained in our reproduction runs and those reported in the submissions is shown in <xref ref-type="fig" rid="fig4">Figure 4</xref>. For the 13 submissions that completed re-execution on the selected representative tasks, the regenerated five-fold mean RMSE values were close to the originally reported values, with the maximum absolute difference being 7.2%. This numerical agreement suggests that, for these successfully re-executed submissions and the specific tasks examined, the benchmark outcomes were relatively stable once a runnable environment had been established.</p>
          <fig id="fig4" position="float">
            <label>Figure 4</label>
            <caption>
              <p>Percentage difference between the five-fold mean RMSE obtained from re-execution and that recorded in the original MatBench submission, calculated using Eq. (1). Positive values indicate that the re-executed RMSE was higher than the reported value. In contrast, negative values indicate that it was lower. Values of 0 indicate that the reported and re-executed five-fold mean RMSE values were identical within the numerical precision of the stored results. Blue and red bars represent composition-based and structure-based tasks, respectively. RMSE: Root mean square error; RFLR: Random Forest by Lorenz Romaner; coGN: Connectivity optimized Graph Network; coNGN: Connectivity optimized Nested Graph Network.</p>
            </caption>
            <graphic xlink:href="aiagent2031.fig.4.jpg"/>
          </fig>
        </sec>
      </sec>
      <sec id="sec3-2">
        <title>JARVIS-Leaderboard</title>
        <p>Environment instantiation was also identified as a nontrivial issue in a parallel reproducibility effort we conducted on the JARVIS-Leaderboard benchmark set<sup>[<xref ref-type="bibr" rid="B19">19</xref>]</sup>. Submissions to this platform contain model predictions on the platform benchmarks, an executable run.sh script (which in most cases executes a python script), and a metadata.json file which plays a similar role as the info.json file used in MatBench submissions where the most relevant information for the environment is in the software_used field.</p>
        <p>The reproducibility effort for JARVIS-Leaderboard models aimed not only to set up a valid environment in which to run the models but to do so using a purely functional Nix packaging ecosystem<sup>[<xref ref-type="bibr" rid="B20">20</xref>]</sup>. The functional packaging paradigm models the software environment as the output of a pure function whose inputs are the pure functions that similarly output the environment’s direct dependencies<sup>[<xref ref-type="bibr" rid="B21">21</xref>]</sup>. This functional relation continues recursively such that an environment closure precisely specifies dependencies as a directed acyclic graph with nodes ranging from the model benchmark script, through the Python interpreter and modules, to low-level transitive dependencies such as the C standard library. The functional approach then makes full or partial updates to the environment straightforward, while only rebuilding portions of the dependency graph downstream of any changes. This is in contrast with the more common imperative model of instantiating software environments, where a series of steps are performed, each mutating the filesystem of some OS image to add, remove, or update packages to arrive at a given environment. Stronger reproducibility guarantees can then be made in the functional model without the dependence on the details of an entire OS image<sup>[<xref ref-type="bibr" rid="B22">22</xref>-<xref ref-type="bibr" rid="B24">24</xref>]</sup>. See <xref ref-type="fig" rid="fig5">Figure 5</xref> for an example graph of the package dependencies which can readily be constructed from the Nix function describing an environment.</p>
        <fig id="fig5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Runtime dependencies of the “Elemnet” model environment constructed from the Nix packaging done for JARVIS-Leaderboard. The environment is the node at the very top while deeper levels of transitive dependencies are shown lower in the graph. While package names and versions are present for those who wish to zoom in, the purpose of the image is to demonstrate the complexity of the environment needed for even a simple model, and the ability of the Nix approach to capture these requirements.</p>
          </caption>
          <graphic xlink:href="aiagent2031.fig.5.jpg"/>
        </fig>
        <p>To narrow the initial scope of this work, we focus on packaging just the models which provide predictions for the exfoliation energy benchmark using Nix. This was chosen to be generally cheaper to test as the prediction output is a single scalar, and the training dataset is smaller than for other benchmarks. This allows efforts to be directed toward reproducing usable environments for the model to run rather than other factors. The general approach was to start by writing a Nix function for each model’s environment containing only the software in JARVIS-Leaderboard entry. This function is modified by adding packages, constraining package versions, and/or adding patches to the source script until a trial training and inference run succeeds. Source scripts were also patched to enable Command-Line Interface (CLI) flags for running only the desired benchmark (some scripts ran more than just exfoliation energy) and running in a “test mode” to reduce the number of training iterations.</p>
        <p>Nix packages for non-Python dependencies and the Python interpreter itself were obtained from the nixpkgs repository<sup>[<xref ref-type="bibr" rid="B25">25</xref>]</sup>. While nixpkgs also contains a number of Python packages, it does not contain all of the Python Package Index (PyPI) or some other packages that are more easily handled by Python-specific tooling. For Python-specific environments we prepared a pyproject.toml file<sup>[<xref ref-type="bibr" rid="B26">26</xref>]</sup> for each module compatible with the PDM<sup>[<xref ref-type="bibr" rid="B27">27</xref>]</sup> Python dependency management tool. This was then incorporated into the Nix function via dream2nix<sup>[<xref ref-type="bibr" rid="B28">28</xref>]</sup>. No attempt was made to constrain dependency versions to those used by the original authors. The JARVIS-Leaderboard project only required the software_used field of metadata.json to contain a string of comma-separated software projects, so such constraints were rarely provided. For models where version constraints were provided, they were utilized only on select packages when unconstrained dependency versions resulted in errors.</p>
        <p>In addition to just identifying dependencies and versions, it also became necessary to patch the source files of some benchmark scripts. Some patches were needed due to API changes in dependencies (similar to those mentioned in Section “Reproducing the benchmark results”). Other patches were needed for Nix-specific reasons, mostly to avoid hardcoding filesystem paths for output data. In particular, mutating the source directory at runtime breaks the purely functional model. Patches were also added to create a uniform CLI interface to select the benchmark set to run and a-test flag which runs a computationally cheaper version of the model to test the environment (e.g., with fewer training iterations). Making edits across multiple contribution scripts from different authors to achieve similar behavior modification was facilitated by LLM models, though the patches were sufficiently contained in scope to be reviewed and judged to be very unlikely to have an impact on results.</p>
        <p>Some models (namely, those from the kgcnn group) contained a conda environment.yaml file with specific versions; while not used directly (as conda was not used in our packaging approach), this was the most complete of the environments documented (indeed it seems it may be overcomplete, likely containing some packages which are not used). Most models had incomplete descriptions of the software used. In this case, Python reflection capabilities were used to obtain a list of required modules which could then be used to identify packages needed in the environment. One unfortunate aspect of the Python ecosystem is that module names and package names need not be the same; indeed, the appropriate “jarvis” module itself is provided by the package “jarvis-tools”. Even with reflection tools, this task could only be partially automated. In addition to identifying the required packages, the package versions were also constrained where needed. Additional patching (not related to the Nix filepath constraints or added CLI flags) was then also applied where needed. Some remaining errors were a result of changes in the PyPI, such as the package “sklearn” being renamed to “scikit-learn”. The matminer-lgbm model also had an error related to the fact that the JARVIS-Leaderboard repo had changed a directory name where benchmark data was stored and the script was not kept in sync with this change.</p>
        <p>After these modifications, still 3 out of the 14 models require a more substantial effort to properly package their environments. The alignn model depends on the deep graph library (DGL) which is not present in the PyPI and requires additional packaging work to be amenable for use with the Nix approach. The provided scripts for the cgcnn model download data and code which are then executed all at runtime breaking the separation between environment instantiation and model training/inference execution, the untangling of which was not perused. Finally, the matformer model run script appears to activate a conda environment presumably present on the contributor’s machine, but which is not present in the leaderboard contribution.</p>
        <p>The central issue of insufficiently specified environments is just as present in JARVIS-Leaderboard as in MatBench and the Nix approach to defining and testing environments taken comes with a distinct set of tradeoffs. Writing Nix functions for model environments is likely less familiar to most materials data science researchers. However, this barrier can likely be lowered significantly by LLMs, especially since Nix environments are fully represented in text. The Nix approach enables sandboxed environments to be built and run locally while iterating to debug issues without needing to rebuild portions of the environment which remain unchanged between iterations. GPU acceleration is crucial to the field, but was not fully addressed at this stage of the work. This “hardware dependent software” has some necessary implications on reproducibility and requires some special treatment within the functional packing model. Multiple, evolving approaches currently exist within the ecosystem which will be utilized in subsequent work<sup>[<xref ref-type="bibr" rid="B29">29</xref>,<xref ref-type="bibr" rid="B30">30</xref>]</sup>.</p>
      </sec>
      <sec id="sec3-3">
        <title>Discussion</title>
        <p>Based on the discussion above, we conclude that the main reproducibility challenges on the MatBench and JARVIS-Leaderboard platform stem from the difficulty of accurately reconstructing the Python environments originally used by the authors. Once a runnable environment can be successfully established, the reported results are generally highly reproducible [<xref ref-type="fig" rid="fig6">Figure 6</xref>]. The limitations of the currently provided metadata can be summarized as follows: (i) the list of required packages is sometimes incomplete, leading to import errors before execution can begin; (ii) specific package versions may exhibit dependency conflicts, whereas substituting alternative versions can render the code incompatible; and (iii) as the package ecosystem evolves, some previously specified dependencies may become unavailable, causing environment reconstruction to fail. Therefore, additional measures are needed to improve the success rate of environment reconstruction and the executability of the submitted code, thereby strengthening the reproducibility of benchmark results.</p>
        <fig id="fig6" position="float">
          <label>Figure 6</label>
          <caption>
            <p>The real bottleneck in reproducing the results is the difficulty in reconstructing the Python environment used by the author.</p>
          </caption>
          <graphic xlink:href="aiagent2031.fig.6.jpg"/>
        </fig>
        <p>Based on these observations, we propose several recommendations for benchmark platform maintainers to improve the reproducibility of submitted algorithms as shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>. These recommendations can be broadly divided into two categories: submission-time validation and preservation of validated environments. First, when a new algorithm is submitted to a benchmark platform, a continuous integration workflow could be used to automatically validate the corresponding Python environment. Submissions could be accepted only after they successfully pass environment reconstruction and import checks. To further strengthen this validation process, the workflow could include a lightweight smoke test that executes a minimal benchmark run within the same gated submission pipeline, ensuring that the submitted code is not only installable but also executable. Once a valid environment has been established, it should be preserved to support future re-execution. For example, authors or platform maintainers could export an environment backup file, such as environment.yml, which provides a user-friendly starting point for environment reconstruction. Third, as package ecosystems evolve, an environment backup file may become insufficiently robust for resolving the intended dependency set. In such cases, tools such as conda-lock can post-process environment.yml file by generating a fully resolved lockfile that explicitly specifies the concrete dependency versions to be installed. As a result, even if upstream package repositories change in the future, environment resolution remains unaffected. Fourth, in some cases, the target package is no longer available (e.g., matbench v0.1.0), making it impossible for even a lockfile to retrieve the required artifacts. In such situations, another practical option is to use a containerization tool (e.g., conda-pack) to package the entire environment into a relocatable archive that users can activate and run directly.</p>
        <fig id="fig7" position="float">
          <label>Figure 7</label>
          <caption>
            <p>Platform-level recommendations for better environment reproducibility.</p>
          </caption>
          <graphic xlink:href="aiagent2031.fig.7.jpg"/>
        </fig>
        <p>Notably, these improvements can be achieved without changing the current submission format. An environment backup file can be automatically exported, and the installed environment can be post-processed using package locker or containerization tools by configuring a continuous integration workflow. As a result, maintainers would not need to invest substantial additional manual effort for future submissions, because continuous integration workflows can perform validation checks while improving environment reproducibility. A working example of the code and end-to-end execution workflow has been uploaded to our repository.</p>
        <p>Our recommendations primarily emphasize the reconstruction of executable software environments. However, the reproducibility of benchmark results may also be affected by differences in operating systems, hardware driver versions, hardware configurations, and nondeterministic behavior in numerical computation. Because it is impractical to require all training runs to be performed on identical machines, several additional strategies could be considered to improve the executability, accessibility, and robustness of algorithms hosted on benchmark platforms. For instance, environment reconstruction and execution workflows could be provided through a hosted cloud service, allowing users to test both environment configuration and code execution on a widely accessible managed cloud environment. Alternatively, callable model interfaces, associated metadata, and execution specifications could be published through Machine Learning Operations (MLOps) platforms such as Garden-AI, which may improve model discoverability and simplify remote invocation, particularly for standardized machine-learning workflows. Another option is to distribute each algorithm through a container (e.g., as a Docker image or Dockerfile), which can preserve much of the user-space software stack and reduce dependency drift caused by changes in the Python package ecosystem. Nevertheless, containerized execution does not eliminate all sources of variation, as host drivers, GPU hardware, system architecture, and registry maintenance may still affect execution. These approaches can improve different aspects of reproducibility and usability, but they also impose additional burdens on method developers, algorithm contributors, and platform maintainers in terms of packaging, validation, infrastructure, storage, and long-term maintenance. These approaches address different aspects of environment preservation and workflow accessibility, and no single strategy eliminates all sources of computational variation. The trade-off between these operational costs and the resulting gains in reproducibility should be carefully considered. Their respective capabilities, limitations, and maintenance requirements are summarized in <xref ref-type="table" rid="t5">Table 5</xref>.</p>
        <table-wrap id="t5">
          <label>Table 5</label>
          <caption>
            <p>Comparison of common strategies for preserving and providing executable machine-learning environments</p>
          </caption>
          <table frame="hsides" rules="groups">
  <tbody>
    <tr>
      <td>
        <bold>Approach</bold>
      </td>
      <td>
        <bold>Main capability</bold>
      </td>
      <td>
        <bold>Remaining limitations</bold>
      </td>
      <td>
        <bold>Maintenance burden</bold>
      </td>
    </tr>
    <tr>
      <td>requirements.txt or environment.yml</td>
      <td>Records direct dependencies and provides an editable starting point for environment reconstruction</td>
      <td>Does not necessarily preserve exact transitive dependencies or package builds; reconstruction may fail if packages or repositories change</td>
      <td>Low</td>
    </tr>
    <tr>
      <td>conda-lock</td>
      <td>Generates a fully resolved lockfile containing exact package versions and builds for specified platforms</td>
      <td>Still depends on the continued availability of the referenced packages and channels; does not preserve the operating system, drivers, or hardware</td>
      <td>Low to moderate</td>
    </tr>
    <tr>
      <td>conda-pack</td>
      <td>Archives an already installed Conda environment for direct relocation and reuse</td>
      <td>Portability may be limited across operating systems, architectures, system libraries, and hardware configurations</td>
      <td>Moderate</td>
    </tr>
    <tr>
      <td>Docker or Apptainer image</td>
      <td>Preserves most of the user space software stack and reduces dependency drift</td>
      <td>Does not preserve the host kernel, GPU drivers, hardware, or external services; image registries and security updates require maintenance</td>
      <td>Moderate to high</td>
    </tr>
    <tr>
      <td>Nix</td>
      <td>Describes the environment through a declarative dependency graph and supports traceable, reproducible builds</td>
      <td>Requires specialized packaging knowledge; GPU support, proprietary software, and external resources may require additional configuration</td>
      <td>Moderate to high</td>
    </tr>
    <tr>
      <td>Hosted cloud or notebook service</td>
      <td>Provides an accessible and partially standardized execution platform</td>
      <td>Base images, available hardware, service policies, and package versions may change over time; continued availability depends on the service provider</td>
      <td>Continuous</td>
    </tr>
    <tr>
      <td>Hosted MLOps or model-serving platform</td>
      <td>Provides standardized model interfaces, remote execution, metadata management, and improved discoverability</td>
      <td>May restrict modification of the underlying environment and depends on long-term platform hosting, infrastructure, and service compatibility</td>
      <td>High</td>
    </tr>
  </tbody>
</table>
          <table-wrap-foot>
            <fn id="t5FN1">
              <p>GPU: Graphics Processing Unit; MLOps: Machine Learning Operations.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
    </sec>
    <sec id="sec4">
      <title>CONCLUSION</title>
      <p>In this work, we assessed the practical reproducibility of 28 MatBench submissions using the publicly available source code and dependency metadata. Only 2 of the 28 submissions could be installed from the original specifications and pass the import-based environment validation without modification. Using a three-stage workflow consisting of strict environment reconstruction, automated dependency remediation, and targeted human-supervised intervention, we established runnable environments for 26 submissions. These 26 submissions subsequently entered the code re-execution stage, of which 13 completed the selected representative benchmark task and generated benchmark outputs. For these 13 cases, the regenerated five-fold mean RMSE values obtained for the selected representative tasks were generally close to the values reported in the original submissions. Similar issues have been found on JARVIS-Leaderboard, a recent materials benchmark platform. Based on this analysis, we recommend that benchmark platforms adopt continuous integration-based validation of submissions and, within the workflow, automatically generate an environment backup file for each submission while employing stronger environment-preservation mechanisms and tools. Together, these measures can reduce the maintenance burden on users and maintainers while improving long-term reusability and trust.</p>
    </sec>
  </body>
  <back>
    <sec>
      <title>DECLARATIONS</title>
      <sec>
        <title>Acknowledgments</title>
        <p>For computer time, this research used Shaheen III managed by the Supercomputing Core Laboratory at King Abdullah University of Science and Technology (KAUST) in Thuwal, Saudi Arabia.</p>
      </sec>
      <sec>
        <title>Authors’ contributions</title>
        <p>Conceptualization: Lyu, B.; Bonini, J.; Wines, D.; Li, K.</p>
        <p>Methodology: Lyu, B.; Bonini, J.; Wines, D.; Li, K.</p>
        <p>Investigation and data curation: Lyu, B.; Bonini, J.; Zhang, M.</p>
        <p>Formal analysis and visualization: Lyu, B.; Bonini, J.</p>
        <p>Writing - original draft: Lyu, B.; Bonini, J.</p>
        <p>Writing - review and editing: Lyu, B.; Bonini, J.; Zhang, M.; Sadeghi, A.; Jaberi, A.; Hattrick-Simpers, J.; Choudhary, K.; Wines, D.; Li, K.</p>
        <p>Supervision: Li, K.</p>
      </sec>
      <sec>
        <title>Availability of data and materials</title>
        <p>The GitHub Action workflows for MatBench are available at <uri xlink:href="https://github.com/BohuiLyu/matbench_reproduce.git">https://github.com/BohuiLyu/matbench_reproduce.git</uri>. The Nix packaging for JARVIS-Leaderboard is available at <uri xlink:href="https://github.com/usnistgov/reproducible-models">https://github.com/usnistgov/reproducible-models</uri>.</p>
      </sec>
      <sec>
        <title>AI and AI-assisted tools statement</title>
        <p>During the preparation of this manuscript, the AI tool GPT-5.4 Thinking (version 5.4, released 2026-03-05) was accessed through the ChatGPT web interface and used as a human-supervised troubleshooting assistant for software-environment reconstruction. Package requirements, installation commands, and associated error messages were provided to the model to obtain possible remediation suggestions. The model did not execute commands or autonomously modify any code or environment. The tool did not influence the study design, data collection, analysis, interpretation, or the scientific content of the work. All authors take full responsibility for the accuracy, integrity, and final content of the manuscript.</p>
      </sec>
      <sec>
        <title>Financial support and sponsorship</title>
        <p>Li, K. acknowledges funding support from King Abdullah University of Science and Technology (KAUST). NIST work was funded solely by the United States Government.<break/><break/>Certain commercial equipment, instruments, software, or materials are identified in this paper in order to specify the experimental procedure adequately. Such identifications are not intended to imply recommendation or endorsement by NIST, nor it is intended to imply that the materials or equipment identified are necessarily the best available for the purpose.</p>
      </sec>
      <sec>
        <title>Conflicts of interest</title>
        <p>All authors declared that there are no conflicts of interest.</p>
      </sec>
      <sec>
        <title>Ethical approval and consent to participate</title>
        <p>Not applicable.</p>
      </sec>
      <sec>
        <title>Consent for publication</title>
        <p>Not applicable.</p>
      </sec>
      <sec>
        <title>Copyright</title>
        <p>© The Author(s) 2026.</p>
      </sec>
	  <sec sec-type="supplementary-material">
      <title>Supplementary Materials</title>
          <supplementary-material content-type="local-data">
                <media xlink:href="aiagent2031-SupplementaryMaterials.zip" mimetype="application/pdf">
                        <caption>
                                <p>Supplementary Materials</p>
                        </caption>
                </media>
          </supplementary-material>
          </sec>
    </sec>
    <ref-list>
      <ref id="B1">
        <label>1</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Lejaeghere</surname>
              <given-names>K.</given-names>
            </name>
            <name>
              <surname>Bihlmayer</surname>
              <given-names>G.</given-names>
            </name>
            <name>
              <surname>Björkman</surname>
              <given-names>T.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>Reproducibility in density functional theory calculations of solids</article-title>
          <source>Science</source>
          <year>2016</year>
          <volume>351</volume>
          <fpage>aad3000</fpage>
          <pub-id pub-id-type="doi">10.1126/science.aad3000</pub-id>
          <pub-id pub-id-type="pmid">27013736</pub-id>
        </element-citation>
      </ref>
      <ref id="B2">
        <label>2</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Bosoni</surname>
              <given-names>E.</given-names>
            </name>
            <name>
              <surname>Beal</surname>
              <given-names>L.</given-names>
            </name>
            <name>
              <surname>Bercx</surname>
              <given-names>M.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>How to verify the precision of density-functional-theory implementations via reproducible and universal workflows</article-title>
          <source>Nat. Rev. Phys.</source>
          <year>2023</year>
          <volume>6</volume>
          <fpage>45</fpage>
          <lpage>58</lpage>
          <pub-id pub-id-type="doi">10.1038/s42254-023-00655-3</pub-id>
        </element-citation>
      </ref>
      <ref id="B3">
        <label>3</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wilkinson</surname>
              <given-names>M. D.</given-names>
            </name>
            <name>
              <surname>Dumontier</surname>
              <given-names>M.</given-names>
            </name>
            <name>
              <surname>Aalbersberg</surname>
              <given-names>I. J.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>The FAIR Guiding Principles for scientific data management and stewardship</article-title>
          <source>Sci. Data</source>
          <year>2016</year>
          <volume>3</volume>
          <fpage>160018</fpage>
          <pub-id pub-id-type="doi">10.1038/sdata.2016.18</pub-id>
          <pub-id pub-id-type="pmid">26978244</pub-id>
          <pub-id pub-id-type="pmcid">PMC4792175</pub-id>
        </element-citation>
      </ref>
      <ref id="B4">
        <label>4</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Boyd</surname>
              <given-names>P. G.</given-names>
            </name>
            <name>
              <surname>Chidambaram</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>García-Díez</surname>
              <given-names>E.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>Data-driven design of metal-organic frameworks for wet flue gas CO<sub>2</sub> capture</article-title>
          <source>Nature</source>
          <year>2019</year>
          <volume>576</volume>
          <fpage>253</fpage>
          <lpage>6</lpage>
          <pub-id pub-id-type="doi">10.1038/s41586-019-1798-7</pub-id>
          <pub-id pub-id-type="pmid">31827290</pub-id>
        </element-citation>
      </ref>
      <ref id="B5">
        <label>5</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>M.</given-names>
            </name>
            <name>
              <surname>Jiang</surname>
              <given-names>J.</given-names>
            </name>
          </person-group>
          <article-title>Accelerating discovery of polyimides with intrinsic microporosity for membrane‐based gas separation: synergizing physics‐informed performance metrics and active learning</article-title>
          <source>Adv. Funct. Mater.</source>
          <year>2024</year>
          <volume>34</volume>
          <fpage>2314683</fpage>
          <pub-id pub-id-type="doi">10.1002/adfm.202314683</pub-id>
        </element-citation>
      </ref>
      <ref id="B6">
        <label>6</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Tang</surname>
              <given-names>H.</given-names>
            </name>
            <name>
              <surname>Jiang</surname>
              <given-names>J.</given-names>
            </name>
          </person-group>
          <article-title>In silico screening and design strategies of ethane-selective metal-organic frameworks for ethane/ethylene separation</article-title>
          <source>AIChE J.</source>
          <year>2020</year>
          <volume>67</volume>
          <fpage>e17025</fpage>
          <pub-id pub-id-type="doi">10.1002/aic.17025</pub-id>
        </element-citation>
      </ref>
      <ref id="B7">
        <label>7</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>T.</given-names>
            </name>
            <name>
              <surname>He</surname>
              <given-names>X.</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>M.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>Ab initio characterization of protein molecular dynamics with AI<sup>2</sup>BMD</article-title>
          <source>Nature</source>
          <year>2024</year>
          <volume>635</volume>
          <fpage>1019</fpage>
          <lpage>27</lpage>
          <pub-id pub-id-type="doi">10.1038/s41586-024-08127-z</pub-id>
          <pub-id pub-id-type="pmid">39506110</pub-id>
          <pub-id pub-id-type="pmcid">PMC11602711</pub-id>
        </element-citation>
      </ref>
      <ref id="B8">
        <label>8</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Unke</surname>
              <given-names>O. T.</given-names>
            </name>
            <name>
              <surname>Chmiela</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Sauceda</surname>
              <given-names>H. E.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>Machine learning force fields</article-title>
          <source>Chem. Rev.</source>
          <year>2021</year>
          <volume>121</volume>
          <fpage>10142</fpage>
          <lpage>86</lpage>
          <pub-id pub-id-type="doi">10.1021/acs.chemrev.0c01111</pub-id>
          <pub-id pub-id-type="pmid">33705118</pub-id>
          <pub-id pub-id-type="pmcid">PMC8391964</pub-id>
        </element-citation>
      </ref>
      <ref id="B9">
        <label>9</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Chung</surname>
              <given-names>Y. G.</given-names>
            </name>
            <name>
              <surname>Gómez-Gualdrón</surname>
              <given-names>D. A.</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>P.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>In silico discovery of metal-organic frameworks for precombustion CO<sub>2</sub> capture using a genetic algorithm</article-title>
          <source>Sci. Adv.</source>
          <year>2016</year>
          <volume>2</volume>
          <fpage>e1600909</fpage>
          <pub-id pub-id-type="doi">10.1126/sciadv.1600909</pub-id>
          <pub-id pub-id-type="pmid">27757420</pub-id>
          <pub-id pub-id-type="pmcid">PMC5065252</pub-id>
        </element-citation>
      </ref>
      <ref id="B10">
        <label>10</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Nandy</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Yue</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Oh</surname>
              <given-names>C.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>A database of ultrastable MOFs reassembled from stable fragments with machine learning models</article-title>
          <source>Matter</source>
          <year>2023</year>
          <volume>6</volume>
          <fpage>1585</fpage>
          <lpage>603</lpage>
          <pub-id pub-id-type="doi">10.1016/j.matt.2023.03.009</pub-id>
        </element-citation>
      </ref>
      <ref id="B11">
        <label>11</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Terrones</surname>
              <given-names>G. G.</given-names>
            </name>
            <name>
              <surname>Huang</surname>
              <given-names>S. P.</given-names>
            </name>
            <name>
              <surname>Rivera</surname>
              <given-names>M. P.</given-names>
            </name>
            <name>
              <surname>Yue</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Hernandez</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Kulik</surname>
              <given-names>H. J.</given-names>
            </name>
          </person-group>
          <article-title>Metal-organic framework stability in water and harsh environments from data-driven models trained on the diverse WS24 data set</article-title>
          <source>J. Am. Chem. Soc.</source>
          <year>2024</year>
          <volume>146</volume>
          <fpage>20333</fpage>
          <lpage>48</lpage>
          <pub-id pub-id-type="doi">10.1021/jacs.4c05879</pub-id>
          <pub-id pub-id-type="pmid">38984798</pub-id>
        </element-citation>
      </ref>
      <ref id="B12">
        <label>12</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Stach</surname>
              <given-names>E.</given-names>
            </name>
            <name>
              <surname>Decost</surname>
              <given-names>B.</given-names>
            </name>
            <name>
              <surname>Kusne</surname>
              <given-names>A. G.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>Autonomous experimentation systems for materials development: a community perspective</article-title>
          <source>Matter</source>
          <year>2021</year>
          <volume>4</volume>
          <fpage>2702</fpage>
          <lpage>26</lpage>
          <pub-id pub-id-type="doi">10.1016/j.matt.2021.06.036</pub-id>
        </element-citation>
      </ref>
      <ref id="B13">
        <label>13</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Dai</surname>
              <given-names>T.</given-names>
            </name>
            <name>
              <surname>Vijayakrishnan</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Szczypiński</surname>
              <given-names>F. T.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>Autonomous mobile robots for exploratory synthetic chemistry</article-title>
          <source>Nature</source>
          <year>2024</year>
          <volume>635</volume>
          <fpage>890</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="doi">10.1038/s41586-024-08173-7</pub-id>
          <pub-id pub-id-type="pmid">39506122</pub-id>
          <pub-id pub-id-type="pmcid">PMC11602721</pub-id>
        </element-citation>
      </ref>
      <ref id="B14">
        <label>14</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Krizhevsky</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Sutskever</surname>
              <given-names>I.</given-names>
            </name>
            <name>
              <surname>Hinton</surname>
              <given-names>G. E.</given-names>
            </name>
          </person-group>
          <article-title>ImageNet classification with deep convolutional neural networks</article-title>
          <source>Commun. ACM</source>
          <year>2017</year>
          <volume>60</volume>
          <fpage>84</fpage>
          <lpage>90</lpage>
          <pub-id pub-id-type="doi">10.1145/3065386</pub-id>
        </element-citation>
      </ref>
      <ref id="B15">
        <label>15</label>
        <element-citation publication-type="conference">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Singh</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Michael</surname>
              <given-names>J.</given-names>
            </name>
            <name>
              <surname>Hill</surname>
              <given-names>F.</given-names>
            </name>
            <name>
              <surname>Levy</surname>
              <given-names>O.</given-names>
            </name>
            <name>
              <surname>Bowman</surname>
              <given-names>S. R.</given-names>
            </name>
          </person-group>
          <comment>GLUE: a multi-task benchmark and analysis platform for natural language understanding. In <italic>Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP</italic>; Linzen, T.; Chrupała, G.; Alishahi, A., Eds.; Association for Computational Linguistics: Brussels, Belgium, 2018; pp 353-5</comment>
          <pub-id pub-id-type="doi">10.18653/v1/W18-5446</pub-id>
        </element-citation>
      </ref>
      <ref id="B16">
        <label>16</label>
        <element-citation publication-type="conference">
          <person-group person-group-type="author">
            <name>
              <surname>Reddi</surname>
              <given-names>V. J.</given-names>
            </name>
            <name>
              <surname>Cheng</surname>
              <given-names>C.</given-names>
            </name>
          </person-group>
          <comment>Kanter, D., et al. MLPerf inference benchmark. In <italic>Proceedings of the ACM/IEEE 47th Annual International Symposium on Computer Architecture (ISCA)</italic>; 2020; pp 446-59</comment>
          <pub-id pub-id-type="doi">10.1109/ISCA45697.2020.00045</pub-id>
        </element-citation>
      </ref>
      <ref id="B17">
        <label>17</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Dunn</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>Q.</given-names>
            </name>
            <name>
              <surname>Ganose</surname>
              <given-names>A.</given-names>
            </name>
            <name>
              <surname>Dopp</surname>
              <given-names>D.</given-names>
            </name>
            <name>
              <surname>Jain</surname>
              <given-names>A.</given-names>
            </name>
          </person-group>
          <article-title>Benchmarking materials property prediction methods: the Matbench test set and Automatminer reference algorithm. <italic>npj Comput. Mater.</italic> <bold>2020</bold>, <italic>6</italic>, 138</article-title>
          <pub-id pub-id-type="doi">10.1038/s41524-020-00406-3</pub-id>
        </element-citation>
      </ref>
      <ref id="B18">
        <label>18</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Riebesell</surname>
              <given-names>J.</given-names>
            </name>
            <name>
              <surname>Goodall</surname>
              <given-names>R. E. A.</given-names>
            </name>
            <name>
              <surname>Benner</surname>
              <given-names>P.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>A framework to evaluate machine learning crystal stability predictions</article-title>
          <source>Nat. Mach. Intell.</source>
          <year>2025</year>
          <volume>7</volume>
          <fpage>836</fpage>
          <lpage>47</lpage>
          <pub-id pub-id-type="doi">10.1038/s42256-025-01055-1</pub-id>
        </element-citation>
      </ref>
      <ref id="B19">
        <label>19</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Choudhary</surname>
              <given-names>K.</given-names>
            </name>
            <name>
              <surname>Wines</surname>
              <given-names>D.</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>K.</given-names>
            </name>
            <etal/>
          </person-group>
          <article-title>JARVIS-Leaderboard: a large scale benchmark of materials design methods</article-title>
          <source>npj Comput. Mater.</source>
          <year>2024</year>
          <volume>10</volume>
          <fpage>93</fpage>
          <pub-id pub-id-type="doi">10.1038/s41524-024-01259-w</pub-id>
        </element-citation>
      </ref>
      <ref id="B20">
        <label>20</label>
        <element-citation publication-type="web">
          <comment>NixOS. Declarative builds and deployments. <uri xlink:href="https://nixos.org/">https://nixos.org/</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B21">
        <label>21</label>
        <element-citation publication-type="web">
          <person-group person-group-type="author">
            <name>
              <surname>Dolstra</surname>
              <given-names>E.</given-names>
            </name>
          </person-group>
          <comment>The purely functional software deployment model. Ph.D. Thesis, Utrecht, The Netherlands: Utrecht University, 2006. <uri xlink:href="https://edolstra.github.io/pubs/phd-thesis.pdf">https://edolstra.github.io/pubs/phd-thesis.pdf</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B22">
        <label>22</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Kowalewski</surname>
              <given-names>M.</given-names>
            </name>
            <name>
              <surname>Seeber</surname>
              <given-names>P.</given-names>
            </name>
          </person-group>
          <article-title>Sustainable packaging of quantum chemistry software with the Nix package manager</article-title>
          <source>Int. J. Quantum Chem.</source>
          <year>2022</year>
          <volume>122</volume>
          <fpage>e26872</fpage>
          <pub-id pub-id-type="doi">10.1002/qua.26872</pub-id>
        </element-citation>
      </ref>
      <ref id="B23">
        <label>23</label>
        <element-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Hausch</surname>
              <given-names>M.</given-names>
            </name>
            <name>
              <surname>Hauser</surname>
              <given-names>S.</given-names>
            </name>
            <name>
              <surname>Uekermann</surname>
              <given-names>B.</given-names>
            </name>
          </person-group>
          <article-title>Improving reproducibility of scientific software using Nix/NixOS: a case study on the preCICE ecosystem</article-title>
          <source>Electron. Commun. EASST</source>
          <year>2025</year>
          <volume>83</volume>
          <pub-id pub-id-type="doi">10.14279/eceasst.v83.2613</pub-id>
        </element-citation>
      </ref>
      <ref id="B24">
        <label>24</label>
        <element-citation publication-type="web">
		<person-group person-group-type="author">
            <name>
              <surname>Bandt</surname>
              <given-names>C.</given-names>
            </name>
          </person-group>
          <comment>Assessment of reproducibility in Nix. <uri xlink:href="https://doi.org/10.5281/zenodo.17372811">https://doi.org/10.5281/zenodo.17372811</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B25">
        <label>25</label>
        <element-citation publication-type="web">
          <comment><italic>Nixpkgs</italic>. <uri xlink:href="https://github.com/NixOS/nixpkgs">https://github.com/NixOS/nixpkgs</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B26">
        <label>26</label>
        <element-citation publication-type="web">
          <person-group person-group-type="author">
            <name>
              <surname>Cannon</surname>
              <given-names>B.</given-names>
            </name>
            <name>
              <surname>Ingram</surname>
              <given-names>D.</given-names>
            </name>
          </person-group>
          <comment>Ganssle, P., et al. PEP 621 - Storing project metadata in pyproject.toml. <uri xlink:href="https://peps.python.org/pep-0621/">https://peps.python.org/pep-0621/</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B27">
        <label>27</label>
        <element-citation publication-type="web">
          <comment><italic>PDM - A Modern Python Package and Dependency Manager</italic>. <uri xlink:href="https://pdm-project.org/">https://pdm-project.org/</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B28">
        <label>28</label>
        <element-citation publication-type="web">
          <comment><italic>DREAM2NIX - Automate Reproducible Packaging for Various Language Ecosystems</italic>. <uri xlink:href="https://github.com/nix-community/dream2nix">https://github.com/nix-community/dream2nix</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B29">
        <label>29</label>
        <element-citation publication-type="web">
          <comment><italic>NixGL - A Wrapper Tool for Nix OpenGL Application</italic>. <uri xlink:href="https://github.com/nix-community/nixGL">https://github.com/nix-community/nixGL</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
      <ref id="B30">
        <label>30</label>
        <element-citation publication-type="web">
          <comment><italic>Nix System Graphics - Run Graphics Accelerated Nix Applications on Any Linux Distribution</italic>. <uri xlink:href="https://github.com/soupglasses/nix-system-graphics">https://github.com/soupglasses/nix-system-graphics</uri>. (accessed 2026-09-11)</comment>
        </element-citation>
      </ref>
    </ref-list>
  </back>
</article>
