diff --git a/.gitignore b/.gitignore index 63ca072d..70b1644b 100644 --- a/.gitignore +++ b/.gitignore @@ -49,6 +49,8 @@ confs # Generated MatMaster skill archive apex-flow.zip -# Large DeePMD checkpoints (fetch with scripts/fetch_models.py; keep frozen *.pb in git) +# Exclude arbitrary DeePMD training/source checkpoints, but keep the one +# authoritative single-task DPA4 model bundled by apex-flow. apex/skills/apex-flow/models/**/*.pt +!apex/skills/apex-flow/models/DPA4-alloytongqi/model.pt apex/skills/apex-flow/models/**/*.partial diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml new file mode 100644 index 00000000..47f604f4 --- /dev/null +++ b/.pre-commit-config.yaml @@ -0,0 +1,36 @@ +# Local pre-commit checks (no remote hook repos; works behind restricted proxies). +# Install once: +# pip install pre-commit && pre-commit install +repos: + - repo: local + hooks: + - id: check-merge-conflict + name: check merge conflict markers + entry: python3 -c + args: + - | + import pathlib, re, sys + pat = re.compile(r"^(<<<<<<< |>>>>>>> )") + bad = False + for raw in sys.argv[1:]: + path = pathlib.Path(raw) + if not path.is_file(): + continue + try: + text = path.read_text(encoding="utf-8", errors="ignore") + except Exception: + continue + for i, line in enumerate(text.splitlines(), 1): + if pat.match(line): + print(f"{path}:{i}: merge conflict marker") + bad = True + raise SystemExit(1 if bad else 0) + language: system + types: [text] + + - id: apex-unit-smoke + name: APEX unit smoke (gamma + ticket + failed-result) + entry: bash -c 'cd tests && python3 -m unittest -v test_gamma.TestGamma.test_compute_lower test_skill_scripts.TestGenerateConfigHelpers.test_get_bohrium_ticket_success test_ops.TestTaskStatusHelpers.test_is_failed_task_result_accepts_dict_and_mapping_like' + language: system + pass_filenames: false + types: [python] diff --git a/BCC-HCP_interface_concentration.pdf b/BCC-HCP_interface_concentration.pdf new file mode 100644 index 00000000..02611e5d Binary files /dev/null and b/BCC-HCP_interface_concentration.pdf differ diff --git a/CHANGELOG-1.3.0.md b/CHANGELOG-1.3.0.md index 2e32c76b..fa6b0e09 100644 --- a/CHANGELOG-1.3.0.md +++ b/CHANGELOG-1.3.0.md @@ -15,6 +15,8 @@ Major additions include: - Finite-temperature lattice workflow improvements - Finite-temperature elasticity workflow support and example documentation - Annealing workflow support for LAMMPS +- Melting-point workflow support for solid-liquid coexistence calculations +- Image-resident DPA4/PT2 runtime support with a fail-closed T4 qualification profile - Shape-controlled automatic supercell generation - `req_calc`-based relaxation/property selection for joint workflows - GUI-side batched submission for large configuration sets @@ -74,6 +76,21 @@ Major additions include: - Added LAMMPS annealing workflow support. - Improved generated annealing input scripts for heating, cooling, RDF output, and holding stages. +### Melting Point Workflow + +- Added a LAMMPS solid-liquid coexistence workflow for estimating melting points across multiple temperatures and replicas. +- Added restart-file staging, interface-axis controls, per-temperature restart validation, and generated skill templates. +- Added input validation and regression coverage for melting-specific temperature, replica, restart, and backend constraints. + +### DPA4/PT2 Runtime and Qualification + +- Added image-resident DeePMD PT2 model handling so immutable runtime artifacts are not copied into task upload packages. +- Added DPA4-aware LAMMPS input generation, including the required atom maps, runtime plugin auto-loading, and rejection of incompatible legacy plugin commands. +- Added an audited `DPA4-alloytongqi` source checkpoint, container wrapper templates, runtime manifest template, and a reproducible CPU/GPU and phonoLAMMPS benchmark harness. +- Added a single-source DPA4 runtime profile used by generation, recommendation, standalone validation, and submission. The bundled profile remains fail-closed until an immutable image `ref@digest` passes the recorded post-snapshot T4 qualification. +- Restricted the qualified production contract to one MPI rank on one `c4_m15_1 * NVIDIA T4`, with exact image, wrapper, model hash, and dispatcher validation. +- Added complete automatic `type_map` expansion across every resolved structure and every effective LAMMPS `overwrite_interaction`. + ### Workflow Selection and Submission - Added `req_calc`-based workflow selection for relaxation and property calculations. @@ -94,6 +111,7 @@ Major additions include: - `.debug.log` - `.debug.stdout` - `.debug.stderr` +- Added bounded retries for transient LAMMPS remote-startup failures, including header-only logs, retry evidence preservation, and explicit retry classification in task status files. ### Reporting @@ -126,4 +144,5 @@ Major additions include: - Added runnable examples for RSS, GammaSurface, and finite-temperature elasticity. - Added GUI developer documentation. - Added `monty` to the package dependencies and constrained supported Python versions to `<3.13`. -- Added support for Phonopy v4 in terms of phonon calculation. \ No newline at end of file +- Added Phonopy v4-compatible setup, force-constant, and band-generation fallbacks. +- Expanded the bundled `apex-flow` skill with backend-aware generation and validation for Bohrium, local debug, local cluster, VASP, LAMMPS, DPA4/PT2, melting-point, gamma, and finite-temperature workflows. diff --git a/README.md b/README.md index f9e39acf..8f15be89 100644 --- a/README.md +++ b/README.md @@ -57,6 +57,7 @@ APEX currently offers calculation methods for the following alloy properties: - [3.6 After Submission](#36-after-submission) - [3.7 Graphical Interface (GUI)](#37-graphical-interface-gui) - [3.8 Bohrium Account Defaults](#38-bohrium-account-defaults) + - [3.9 Agent Skill](#39-agent-skill) - [4. Detailed Parameter Reference](#4-detailed-parameter-reference) - [4.1 Global Configuration](#41-global-configuration-globaljson) - [4.2 Calculation Parameters](#42-calculation-parameters-paramjson) @@ -70,9 +71,10 @@ APEX currently offers calculation methods for the following alloy properties: - [4.10 Gamma Line/Surface](#410-gamma-line-generalized-stacking-fault) - [4.11 Phonon Spectra](#411-phonon-spectra) - [4.12 Grüneisen Parameters and Thermal Expansion](#412-grüneisen-parameters-and-thermal-expansion) - - [4.13 Finite-Temperature Lattice Parameters](#413-finite-temperature-lattice-parameters) - - [4.14 Finite-Temperature Elastic constant](#414-finite-temperature-elastic-constant) - - [4.15 Annealing](#415-annealing) + - [4.13 Two-Phase Coexistence Melting Point](#413-two-phase-coexistence-melting-point) + - [4.14 Finite-Temperature Lattice Parameters](#414-finite-temperature-lattice-parameters) + - [4.15 Finite-Temperature Elastic constant](#415-finite-temperature-elastic-constant) + - [4.16 Annealing](#416-annealing) - [More Resources](#more-resources) ## 1.Installation @@ -237,9 +239,9 @@ Create `global_bohrium.json` to submit workflows to the Bohrium cloud platform: ```json { - "lammps_image_name": "registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3", + "lammps_image_name": "registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2", "lammps_run_command":"lmp -in in.lammps", - "scass_type":"c8_m31_1 * NVIDIA T4" + "scass_type":"c16_m120_1 * NVIDIA L20" } ``` @@ -255,6 +257,14 @@ Or set it directly: ```shell apex account --email YOUR_EMAIL --password YOUR_PASSWD --program-id 1234 + +# AccessKey authentication (email/password not required) +apex account --access-key YOUR_ACCESS_KEY --program-id 1234 + +# Clear both login methods, or only one method +apex account --clear +apex account --clear access-key +apex account --clear email ``` When running `apex submit -c global_bohrium.json`, values in your json file still have highest priority and override the saved defaults. @@ -440,6 +450,14 @@ apex account # non-interactive mode apex account --email YOUR_EMAIL --password YOUR_PASSWD --program-id 1234 + +# AccessKey mode (takes precedence over email/password) +apex account --access-key YOUR_ACCESS_KEY --program-id 1234 + +# Clear both login methods, only AccessKey, or only email/password +apex account --clear +apex account --clear access-key +apex account --clear email ``` Useful commands: @@ -454,11 +472,32 @@ When you run `apex submit -c global_bohrium.json`, APEX auto-fills these default - `dflow_host`: `https://workflows.deepmodeling.com` - `k8s_api_server`: `https://workflows.deepmodeling.com` - `batch_type`: `Bohrium` -- `context_type`: `Bohrium` +- `context_type`: `Bohrium` (AccessKey mode exchanges a short-lived ticket for DPDispatcher) - `apex_image_name`: `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` Priority rule: values in your `-c` json file override account defaults. +### 3.9 Agent Skill + +APEX ships two Agent-skill editions with different execution assumptions: + +- `apex skill` prints a local Agent installation prompt. During installation, + the Agent asks whether calculations use Bohrium cloud, the local workstation, + or a local Slurm/PBS cluster, then installs the matching execution profile. + The Bohrium profile uses `apex account` and runs `apex submit` directly from + the local machine; it does not require an access-key ticket or outer job. +- `apex skill --zip` writes `apex-flow.zip` for Bohrium Cloud/MatMaster. This + separate edition uses a ticket and a lightweight outer submission job because + the cloud container cannot read the user's local APEX account file. + +```shell +# Print the local Agent installation prompt +apex skill + +# Build the Bohrium Cloud/MatMaster upload archive +apex skill --zip +``` + ## 4. Detailed Parameter Reference @@ -511,6 +550,7 @@ Priority rule: values in your `-c` json file override account defaults. | `email` | String | `None` | Bohrium account email. | | `phone` | String | `None` | Bohrium account phone. | | `password` | String | `None` | Bohrium password. | +| `access_key` | String | `None` | Bohrium AccessKey. When set, it takes precedence over email/password authentication. | | `program_id` | Integer | `None` | Bohrium program ID. | | `scass_type` | String | `None` | Bohrium node type. | @@ -710,17 +750,46 @@ Notes: APEX generates displaced slab structures from the specified Miller plane and slip directions. The slip vector always follows the primary direction. Use predefined slip systems for FCC, BCC, and HCP crystals when possible to avoid invalid structures. +#### Physically recommended slip planes + +The following plane families are the most important physical defaults for +dislocation slip and generalized stacking-fault calculations. The listed +directions are representative members; symmetry-equivalent planes, opposite +directions, and other deliberately selected paths remain valid user inputs. + +| Crystal | Physically important plane families | Principal Burgers vectors / paths | +|---------|-------------------------------------|------------------------------------| +| **FCC** | close-packed $\{111\}$ | perfect $a/2\langle110\rangle$; Shockley partial $a/6\langle112\rangle$ | +| **BCC** | primary $\{110\}$; also $\{112\}$ and $\{123\}$ | $a/2\langle111\rangle$ | +| **HCP** | basal $\{0001\}$, prismatic $\{10\bar{1}0\}$ / $\{11\bar{2}0\}$, and pyramidal $\{10\bar{1}1\}$ / $\{11\bar{2}2\}$ | $\langle a\rangle$, $\langle c\rangle$, and $\langle c+a\rangle$ systems | + +Only the representative systems on these recommended plane families use the +predefined crystallographic orientation and automatic slip-length metadata +below. If an input plane/direction pair is not in this recommended registry, +both `gamma` and `gamma_surface` emit a warning and fall back to the same +geometric construction: after any HCP Miller--Bravais conversion, APEX checks +only that `plane_miller · slip_direction == 0`. A non-orthogonal pair is still +an error. This warning-only fallback permits intentional non-primary studies, +but the generated slab and displacement vectors must be inspected manually. +For a Gamma line or Gamma surface with `parent_lattice` set, APEX automatically resolves the +integer mapping from the supplied relaxed RSS/SQS cell to that parent lattice. +The user-facing Miller plane and slip direction therefore remain parent-lattice +indices even when the input is a large, sheared, chemically disordered +supercell. APEX does not symmetrize the atoms or lattice. Set +`require_orthogonal_cell = true` when a zero-tilt Cartesian-z cell is part of +the scientific protocol. This is a strict gate: APEX never Gram--Schmidts a +periodic cell because doing so changes its periodic boundaries. The legacy +alias `orthogonalize_cell = true` has the same fail-closed validation meaning. + +Recommended representative systems: + | Crystal | Plane Miller | Slip direction | Secondary direction | Default slip length | |---------|--------------|----------------|---------------------|---------------------| -| **FCC** | $(001)$ | $[100]$ | $[010]$ | $a$ | -| | $(110)$ | $[\bar{1}10]$ | $[001]$ | $\sqrt{2}a$ | -| | $(111)$ | $[11\bar{2}]$ | $[\bar{1}10]$ | $\sqrt{6}a$ | +| **FCC** | $(111)$ | $[11\bar{2}]$ | $[\bar{1}10]$ | $\sqrt{6}a$ | | | $(111)$ | $[\bar{1}\bar{1}2]$ | $[1\bar{1}0]$ | $\sqrt{6}a$ | | | $(111)$ | $[\bar{1}10]$ | $[\bar{1}\bar{1}2]$ | $\sqrt{2}a$ | | | $(111)$ | $[1\bar{1}0]$ | $[11\bar{2}]$ | $\sqrt{2}a$ | -| **BCC** | $(001)$ | $[100]$ | $[010]$ | $a$ | -| | $(111)$ | $[\bar{1}10]$ | $[\bar{1}\bar{1}2]$ | $\frac{\sqrt{2}}{2}a$ | -| | $(110)$ | $[\bar{1}11]$ | $[00\bar{1}]$ | $\frac{\sqrt{3}}{2}a$ | +| **BCC** | $(110)$ | $[\bar{1}11]$ | $[00\bar{1}]$ | $\frac{\sqrt{3}}{2}a$ | | | $(110)$ | $[1\bar{1}\bar{1}]$ | $[001]$ | $\frac{\sqrt{3}}{2}a$ | | | $(112)$ | $[11\bar{1}]$ | $[\bar{1}10]$ | $\frac{\sqrt{3}}{2}a$ | | | $(112)$ | $[\bar{1}\bar{1}1]$ | $[1\bar{1}0]$ | $\frac{\sqrt{3}}{2}a$ | @@ -740,19 +809,30 @@ APEX generates displaced slab structures from the specified Miller plane and sli | | $(\bar{1}2\bar{1}2)$ | $[10\bar{1}0]$ | $[1\bar{2}13]$ | $\sqrt{3}a$ | | | $(\bar{1}2\bar{1}2)$ | $[1\bar{2}13]$ | $[\bar{1}010]$ | $\sqrt{a^2+c^2}$ | +`Default slip length` is APEX's legacy automatic scan span and is not always +one elementary Burgers-vector magnitude. Set `slip_length` explicitly for a +one-Burgers-vector Gamma line. For a complete 2D Gamma surface, prefer +`closed_loop = true`, which derives periodic in-plane translations instead. + Key parameters: | Key | Type | Default | Description | |-----|------|---------|-------------| | `plane_miller` | Sequence[Int] | `None` | Miller indices of the target plane. | | `slip_direction` | Sequence[Int] | `None` | Primary slip direction. | +| `parent_lattice` | String or `None` | `None` | Parent lattice (`bcc`, `fcc`, or `hcp`) for RSS/disordered supercells. Gamma line and Gamma surface automatically infer the integer parent-to-supercell mapping and interpret `plane_miller` / `slip_direction` in the parent basis without symmetrizing the supplied geometry. | | `slip_length` | Float or Sequence | `1` | Slip magnitude (vector format `[x, y, z]` means $\sqrt{(xa)^2 + (yb)^2 + (zc)^2}$). | | `plane_shift` | Float | `0` | Shift of the displacement plane in units of lattice parameter `c`. | -| `n_steps` | Integer | `10` | Number of sampling points along the slip. | -| `vacuum_size` | Float | `0` | Added vacuum layer thickness (Å). | -| `supercell_size` | Sequence[Int] | `[1, 1, 5]` | Slab supercell size. | +| `n_steps` | Integer | `10` | Number of equal slip increments; the uniform path contains `n_steps + 1` points including zero. | +| `displacement_points` | Sequence[Float] or `None` | `None` | Gamma-line-only explicit normalized fractions in `[0,1]`; unique and must include zero. | +| `vacuum_size` | Float | `20` | Added vacuum layer thickness (Å). | +| `require_orthogonal_cell` | Bool | `false` | Fail unless the generated slab is orthogonal, zero-tilt, and has a Cartesian-z normal. No cell orthogonalization is applied. | +| `supercell_size` | Sequence[Int] | `[1, 1, 5]` | In-plane replication and target number of Miller-plane spacings. | +| `min_slab_height` | Float or `None` | `None` | Minimum material-slab thickness (Å); APEX adds only the oriented-cell repeats required to reach it. | +| `max_atoms` | Integer or `None` | `None` | Stop structure generation when the final slab exceeds this atom count. | +| `min_distance` | Float | `0.2` | Stop when any periodic atom-pair distance is below this threshold (Å). | | `add_fix` | Sequence[String] | `["true","true","false"]` | Position constraints along x/y/z. | -| `closed_loop` | Bool | `false` | when `true`, derive two periodic in-plane translation vectors from the generated slab. and slip_length or slip_length_y will be **ignored** | +| `closed_loop` | Bool | `false` | When `true`, derive two periodic in-plane translation vectors from the generated slab; combining it with `slip_length` or `slip_length_y` is an error. | Example: @@ -769,12 +849,27 @@ Example: "plane_shift": 0.25 }, "supercell_size": [1, 1, 6], + "min_slab_height": 18, + "max_atoms": 216, "vacuum_size": 10, "add_fix": ["true", "true", "false"], "n_steps": 10 } ``` -To preview structure behave as expected before brusting computational resource, you can use `preview` to generate a gif file to visulize it. + +The third `supercell_size` value is handled as a Miller-plane count, avoiding +floating-point promotion such as `ceil(2.0000000000000004) = 3`. Generation +statistics are written to `slab_generation.json`, including the requested and +effective plane counts, material-slab height, atom count, and minimum pair +distance. `min_slab_height`, `max_atoms`, and `min_distance` apply equally to +`gamma` and `gamma_surface`. + +Before committing computational resources, use `preview` to generate GIFs of +the displaced structures. Gamma line and gamma surface previews write both the +slip-plane and parent-`bc` projections by default; use `--gif-view default` to +request the legacy single Cartesian view. The rendered viewport includes the +projected unit-cell boundary, so a configured vacuum layer remains visible in +views with a slab-normal component. ```shell apex preview gammaline.json @@ -782,6 +877,23 @@ apex preview gammaline.json Nested dictionaries (`fcc`, `bcc`, `hcp`, etc.) override the top-level parameters for the corresponding lattice type. +For chemically disordered RSS structures that symmetry analysis classifies as +`other`, set `parent_lattice` explicitly. Gamma-line and Gamma-surface generation then infer the +integer parent-supercell mapping, convert the parent Miller indices internally, +uses the reciprocal-lattice plane normal, chooses the elementary parent Burgers +translation, freezes the atom split before adding vacuum, and records the full +mapping/geometry/provenance in `gamma_geometry.json`. Generation fails closed +if the mapping, layer-gap split, minimum distance, or parent-translation +topology is not valid. For an RSS, the relaxed-coordinate and chemical +`u = 1` mismatches are recorded as diagnostics because one elementary parent +translation is not a symmetry of the disordered relaxed configuration. The +supplied relaxed cell is not symmetrized. + +Both properties write `gamma_geometry.json`, freeze the same material-internal +fault split before adding vacuum, and normalize by the recorded number of +interfaces: `Delta E/A` for a vacuum slab and `Delta E/(2A)` for a fully +periodic zero-vacuum cell. Post-processing fails if task areas differ from the +zero-displacement reference. Similarly, to investigate Gamma Surface, change the type to `gamma_surface`, and adjust steps accordingly. `gamma_surface` keeps the same crystallographic interface: `plane_miller` and @@ -794,22 +906,43 @@ adds vacuum along the selected fault normal for slab/free-surface calculations. { "type": "gamma_surface", "req_calc": true, - "plane_miller": [1, 1, 0], - "slip_direction": [1, -1, -1], + "plane_miller": [1, 1, 1], + "slip_direction": [-1, 1, 0], + "bcc": { + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1] + }, + "hcp": { + "plane_miller": [0, 0, 0, 1], + "slip_direction": [2, -1, -1, 0] + }, "supercell_size": [1, 1, 20], "vacuum_size": 15, - "closed_loop": false, + "closed_loop": true, "add_fix": ["true", "true", "false"], "n_steps_x": 20, "n_steps_y": 20 } ] ``` + +The top-level pair is the FCC recommendation and is also the explicit fallback +for structures classified as `other`; edit it for a disordered BCC/HCP parent. +Recognized BCC and HCP structures use their nested recommendations. +`closed_loop = true` is recommended for a complete 2D surface because it +derives periodic in-plane translations and verifies closure at all grid +corners. Set it to `false` only when intentionally defining custom +`slip_length` / `slip_length_y` paths. + ### 4.11 Phonon spectra APEX integrates parts of [dflow-phonon](https://github.com/Chengqian-Zhang/dflow-phonon) and wraps [Phonopy](https://github.com/phonopy/phonopy) / [phonoLAMMPS](https://github.com/abelcarreras/phonolammps). [SeeK-path](https://seekpath.readthedocs.io/en/latest/index.html) automatically generates high-symmetry k-paths. -> **Important:** LAMMPS phonon and Grüneisen workflows always use `registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3`. During `apex submit`, APEX overrides any other configured LAMMPS image for these properties. +> **Important:** LAMMPS phonon and Grüneisen workflows using GPU potentials +> (`deepmd`, `mace`, `nep`) use +> `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2`. +> CPU potentials keep the CPU-safe `apex-flow:1.3.0.post` image; 0.0.2 must +> not be paired with a `*_cpu` machine. | Key | Type | Default | Description | |-----|------|---------|-------------| @@ -850,22 +983,82 @@ APEX supports Grüneisen workflows based on phonon calculations at multiple stra For `full` mode, use fixed-volume internal relaxation in `cal_setting` (`relax_pos = true`, `relax_shape = false`, `relax_vol = false`) so the phonon and energy points share the intended volume grid. -### 4.13 Finite-temperature lattice parameters +### 4.13 Two-phase coexistence melting point + +`melting_point` is a LAMMPS-only direct two-phase coexistence workflow. For +each target temperature and velocity-seed replica, APEX premelts the upper +part of the cell while pinning the lower crystal, conditions the liquid at the +target temperature, then releases the complete cell under zero-pressure NPT +dynamics. The melting bracket is determined from the sign of the local- +$q_6$-derived interface velocity. A positive velocity denotes solid growth; a +negative velocity denotes liquid growth. The reported uncertainty is half the +temperature interval between the nearest consensus solid- and liquid-side +endpoints, not a thermodynamic confidence interval. -APEX supports lattice parameter calculations at finite temperatures using molecular dynamics in LAMMPS. -This workflow performs NVT equilibration at target temperatures and averages lattice parameters over the equilibrated trajectory. +```json +{ + "type": "melting_point", + "method": "two_phase", + "supercell_size": [1, 1, 2], + "cal_setting": { + "temperature": [1600, 1650, 1700], + "premelt_temperature": 4500, + "premelt_steps": 5000, + "conditioning_steps": 5000, + "production_steps": 100000, + "timestep": 0.001, + "tdamp": 0.1, + "pdamp": 1.0, + "pressure": 0.0, + "barostat": "iso", + "interface_axis": "z", + "liquid_fraction": 0.5, + "dump_step": 100, + "thermo_step": 100, + "restart_interval": 10000, + "q6_cutoff": 3.5, + "q6_neighbors": 12, + "replicas": 3, + "velocity_seeds": { + "premelt": 324159, + "condition": 271828, + "release": 161803 + } + } +} +``` + +If `replicas > 1` and one seed triplet is supplied, APEX deterministically +derives independent triplets for later replicas. Alternatively, provide a +list of explicit seed triplets. Independent random-alloy chemical +realizations should be supplied as separate entries in `structures`; APEX +does not silently randomize chemistry inside this property. + +The result directory contains `result.json`, `result.out`, +`melting_point_tidy.csv`, `solid_fraction_vs_time.png`, +`interface_velocity_vs_temperature.png`, and +`solid_liquid_interface_snapshots.png`. Every temperature task retains +`dump.melting`, `log.lammps`, `MeltingPoint.json`, task status evidence, +alternating periodic restarts `restart.melting.1/2`, and the normal-completion +checkpoint `restart.melting.final`. `restart_interval` is measured in LAMMPS +timesteps and defaults to 10000 (10 ps for the default 0.001 ps timestep). + +### 4.14 Finite-temperature lattice parameters + +APEX supports lattice parameter calculations at finite temperatures using NpT molecular dynamics in LAMMPS and VASP. ABACUS is not supported for this property. +The workflow runs separate equilibration and production stages and averages lattice parameters from the production trajectory. VASP uses Langevin–Parrinello–Rahman NpT (`MDALGO=3`, `ISIF=3`) and therefore requires a VASP executable compiled with `-Dtbdyn`. | Key | Type | Default | Description | |-----|------|---------|-------------| | `supercell_size` | Sequence[Int] | `[2, 2, 2]` | Supercell dimensions for the simulation. | -LAMMPS-specific calculation settings in `cal_setting`: +Common and LAMMPS calculation settings in `cal_setting`: | Key | Type | Default | Description | |-----|------|---------|-------------| -| `temperature` | Sequence[Float] | Required | Target temperatures (K) for lattice parameter calculation, e.g., `[200, 400, 600, 800]`. | -| `equi_step` | Integer | `80000` | Number of equilibration steps before averaging. | -| `ave_step` | Integer | `40000` | Number of steps for averaging lattice parameters. | +| `temperature` | Sequence[Float] | `[200, 400, 600, 800]` (LAMMPS), `[300, 500, 700, 900, 1100, 1300, 1500]` (VASP) | Target temperatures (K). | +| `equi_step` | Integer | `80000` (LAMMPS), `5000` (VASP) | Number of equilibration steps before averaging. | +| `ave_step` | Integer | `40000` (LAMMPS), `10000` (VASP) | Number of production steps used for statistics. | | `timestep` | Float | `0.001` | MD timestep (ps). | | `tdamp` | Float | `0.1` | Thermostat damping parameter. | | `pdamp` | Float | `1.0` | Barostat damping parameter. | @@ -873,6 +1066,8 @@ LAMMPS-specific calculation settings in `cal_setting`: | `N_repeat` | Integer | `10` | Number of average samples. | | `N_freq` | Integer | `2000` | Sample output frequency. | +For VASP, use `timestep_fs` (fs) and `pressure_kbar` (kbar); `langevin_gamma`, `langevin_gamma_l`, and `pmass` configure the Langevin thermostat and lattice barostat. Results preserve the legacy `temperature -> [a, b, c, T]` entries and add `_metadata` with the mean, standard deviation, block standard error, and sample count for cell tensors, lengths, angles, and volume. + Example: ```json @@ -893,7 +1088,7 @@ Example: } ``` -### 4.14 Finite-temperature elastic constant +### 4.15 Finite-temperature elastic constant APEX supports elastic constant calculations at finite temperatures using molecular dynamics in LAMMPS. This implementation uses the noise-cancellation method (see [DOI: 10.1103/sd49-wqd6](https://doi.org/10.1103/sd49-wqd6)) to compute temperature-dependent elastic constants from stress fluctuations. @@ -942,17 +1137,18 @@ Example: } ``` -### 4.15 Annealing +### 4.16 Annealing -APEX supports annealing simulations using molecular dynamics in LAMMPS. -This workflow equilibrates the structure at a starting temperature, ramps to a target temperature, cools to an ending temperature, and performs final equilibration. Post-processing extracts radial distribution functions (RDF), mean squared displacement (MSD), and volume-temperature data from the heating and cooling stages for report visualization. +APEX supports annealing simulations using molecular dynamics in LAMMPS and VASP. ABACUS is not supported for this property. +The default `protocol="ramp_cool"` preserves the legacy schedule: equilibrate at a starting temperature, ramp to a target, cool to an ending temperature, and perform final equilibration. VASP also supports `protocol="coexistence"`, a fixed-target-temperature equilibration followed by a fixed-temperature production stage. Both VASP modes use `MDALGO=3` and require `-Dtbdyn`. | Key | Type | Default | Description | |-----|------|---------|-------------| +| `protocol` | String | `"ramp_cool"` | `"ramp_cool"` for the legacy heat/cool schedule, or VASP-only `"coexistence"` for fixed-T equilibration and production. | | `supercell_size` | Sequence[Int] | `[2, 2, 2]` | Supercell dimensions for the simulation. | | `supercell_length` | Float | `None` | Optional target supercell length. When provided, APEX derives a near-cubic replication from the relaxed structure. | -LAMMPS-specific calculation settings in `cal_setting`: +Calculation settings in `cal_setting`: | Key | Type | Default | Description | |-----|------|---------|-------------| @@ -961,10 +1157,11 @@ LAMMPS-specific calculation settings in `cal_setting`: | `end_temp` | Float | `4` | Final cooling temperature (K). | | `temp_ramp_rate` | Float | `1000` | Heating rate used to derive `ramp_step` when explicit step counts are not provided. Alias: `ramp_rate`. | | `cool_rate` | Float | `temp_ramp_rate` | Cooling rate used to derive `cool_step` when explicit step counts are not provided. | -| `equi_step` | Integer | `20000` | Initial equilibration steps at `start_temp`. | -| `ramp_step` | Integer | rate-derived | Heating steps from `start_temp` to `target_temp`. Alias: `temp_ramp_step`. | -| `hold_step` | Integer | `20000` | Final equilibration steps. | -| `cool_step` | Integer | rate-derived | Cooling steps from `target_temp` to `end_temp`. Alias: `temp_decline_step`. | +| `equi_step` | Integer | `20000` (`100` for DFT) | Initial equilibration steps at `start_temp`. | +| `ramp_step` | Integer | rate-derived (`200` for DFT) | Heating steps from `start_temp` to `target_temp`. Alias: `temp_ramp_step`. | +| `hold_step` | Integer | `20000` (`100` for DFT) | Final equilibration steps. | +| `cool_step` | Integer | rate-derived (`200` for DFT) | Cooling steps from `target_temp` to `end_temp`. Alias: `temp_decline_step`. | +| `production_step` | Integer | `10000` | Production steps for VASP `protocol="coexistence"`. | | `thermostat` | String | `"nose_hoover"` | Thermostat method: `"nose_hoover"` or `"langevin"`. | | `ensemble` | String | `"npt"` | Ensemble. For Nose-Hoover use `"npt"` or `"nvt"`; for Langevin use `"nph"` or `"nve"`. | | `timestep` | Float | `0.001` | MD timestep (ps). | @@ -985,6 +1182,8 @@ LAMMPS-specific calculation settings in `cal_setting`: | `msd_nrepeat` | Integer | `1` | Number of MSD samples per average. | | `msd_nfreq` | Integer | `200` | MSD output frequency. | +VASP uses the `timestep_fs` and `pressure_kbar` fields described for `finite_t_latt`. Ramp/cool stages or coexistence equilibration/production stages run sequentially inside each task; `OUTCAR` and `XDATCAR` provide RDF, MSD, volume, pressure, temperature, and energy data for post-processing. + Example: ```json diff --git a/apex/account.py b/apex/account.py index 48ec9095..bc5023fc 100644 --- a/apex/account.py +++ b/apex/account.py @@ -8,6 +8,11 @@ BOHRIUM_WORKFLOWS_HOST = "https://workflows.deepmodeling.com" +SANDBOX_DFLOW_HOST = "https://lbg-workflow-dflow.dp.tech" +SANDBOX_DISPATCHER_IMAGE = ( + "registry.dp.tech/dptech/polycalibur:dpdispatcher-storehost-plan-a-20260811" +) + DEFAULT_BOHRIUM_CONFIG = { "dflow_host": BOHRIUM_WORKFLOWS_HOST, "k8s_api_server": BOHRIUM_WORKFLOWS_HOST, @@ -15,7 +20,20 @@ "context_type": "Bohrium", "apex_image_name": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", } -SENSITIVE_KEYS = {"password"} + +DEFAULT_OPENAPI_CONFIG = { + "dflow_host": SANDBOX_DFLOW_HOST, + "k8s_api_server": SANDBOX_DFLOW_HOST, + "batch_type": "OpenAPI", + "context_type": "OpenAPI", + "platform": "ali", + "machine_type": "c16_m120_1 * NVIDIA L20", + "output_log": False, + "dispatcher_image": SANDBOX_DISPATCHER_IMAGE, + "apex_image_name": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", +} + +SENSITIVE_KEYS = {"password", "access_key", "app_key"} ACCOUNT_FILE_ENV = "APEX_ACCOUNT_FILE" @@ -84,7 +102,33 @@ def mask_sensitive_config(config: dict) -> dict: return masked +def _is_openapi_context(config_dict: dict) -> bool: + """Check if config explicitly requests OpenAPI Sandbox mode.""" + for key in ("context_type", "batch_type"): + value = config_dict.get(key) + if isinstance(value, str) and value.lower() == "openapi": + return True + machine = config_dict.get("machine", {}) + if isinstance(machine, dict): + for key in ("context_type", "batch_type"): + value = machine.get(key) + if isinstance(value, str) and value.lower() == "openapi": + return True + dflow_config = config_dict.get("dflow_config", {}) + if isinstance(dflow_config, dict): + if dflow_config.get("host") == SANDBOX_DFLOW_HOST: + return True + return ( + config_dict.get("dflow_host") == SANDBOX_DFLOW_HOST + or config_dict.get("k8s_api_server") == SANDBOX_DFLOW_HOST + or "machine_type" in config_dict + ) + + def _is_bohrium_context(config_dict: dict) -> bool: + # OpenAPI is a separate context, not legacy Bohrium + if _is_openapi_context(config_dict): + return False for key in ("context_type", "batch_type"): value = config_dict.get(key) if isinstance(value, str) and "bohrium" in value.lower(): @@ -97,18 +141,23 @@ def _is_bohrium_context(config_dict: dict) -> bool: return True remote_profile = machine.get("remote_profile", {}) if isinstance(remote_profile, dict) and any( - key in remote_profile for key in ("email", "password", "program_id")): + key in remote_profile + for key in ("email", "password", "program_id", "project_id", "access_key")): return True dflow_config = config_dict.get("dflow_config", {}) if isinstance(dflow_config, dict): if dflow_config.get("host") == BOHRIUM_WORKFLOWS_HOST: return True return any( - key in config_dict for key in ("email", "password", "program_id", "phone", "bohrium_config") + key in config_dict + for key in ( + "email", "password", "program_id", "phone", "access_key", "bohrium_config" + ) ) or config_dict.get("dflow_host") == BOHRIUM_WORKFLOWS_HOST def _is_explicit_non_bohrium_context(config_dict: dict) -> bool: + """Check if context is explicitly non-Bohrium AND non-OpenAPI (e.g. SSH, Local).""" context_values = [] for key in ("context_type", "batch_type"): value = config_dict.get(key) @@ -122,7 +171,10 @@ def _is_explicit_non_bohrium_context(config_dict: dict) -> bool: context_values.append(value.lower()) if not context_values: return False - return all("bohrium" not in value for value in context_values) + return all( + "bohrium" not in value and "openapi" not in value + for value in context_values + ) def should_apply_bohrium_defaults( @@ -146,14 +198,34 @@ def merge_bohrium_defaults( ) -> dict: user_config = copy.deepcopy(config_dict or {}) account_config = load_account_config() + + # Check if this is an OpenAPI Sandbox config + if _is_openapi_context(user_config): + merged = copy.deepcopy(DEFAULT_OPENAPI_CONFIG) + _deep_update(merged, account_config) + _deep_update(merged, user_config) + missing_required = [ + key for key in ("access_key", "project_id") + if merged.get(key) in (None, "") + ] + if missing_required: + logging.warning( + "Missing OpenAPI Sandbox fields: %s. " + "Set BOHRIUM_ACCESS_KEY / BOHRIUM_PROJECT_ID or provide them in %s.", + ", ".join(missing_required), + config_file or "the config file" + ) + return merged + if not should_apply_bohrium_defaults(user_config, config_file, account_config): return user_config merged = copy.deepcopy(DEFAULT_BOHRIUM_CONFIG) _deep_update(merged, account_config) _deep_update(merged, user_config) + auth_fields = () if merged.get("access_key") else ("email", "password") missing_required = [ - key for key in ("email", "password", "program_id") + key for key in (*auth_fields, "program_id") if merged.get(key) in (None, "") ] if missing_required: @@ -185,15 +257,53 @@ def _prompt_program_id(default: Optional[int]) -> Optional[int]: def prompt_for_account_fields(current_config: dict) -> dict: + current_method = "access-key" if current_config.get("access_key") else "email" + print("Select Bohrium authentication method:") + print(" 1) Email/password") + print(" 2) AccessKey") + while True: + choice = input( + f"Authentication method [1/2, empty keeps {current_method}]: " + ).strip().lower() + if not choice: + method = current_method + break + if choice in {"1", "email", "email/password", "email-password"}: + method = "email" + break + if choice in {"2", "access", "access-key", "accesskey"}: + method = "access-key" + break + print("Choose 1 for email/password or 2 for AccessKey.") + updated = {} - email = _prompt_value("Bohrium email", current_config.get("email")) - if email is not None: - updated["email"] = email - password = getpass("Bohrium password [leave empty to keep current]: ").strip() - if password: - updated["password"] = password - elif current_config.get("password"): - updated["password"] = current_config["password"] + if method == "email": + # Selecting password authentication must disable AccessKey precedence. + updated["access_key"] = None + email = _prompt_value("Bohrium email", current_config.get("email")) + if email is not None: + updated["email"] = email + password = getpass( + "Bohrium password [leave empty to keep current]: " + ).strip() + if password: + updated["password"] = password + elif current_config.get("password"): + updated["password"] = current_config["password"] + else: + access_key = getpass( + "Bohrium AccessKey [leave empty to keep current]: " + ).strip() + if access_key: + updated["access_key"] = access_key + elif current_config.get("access_key"): + updated["access_key"] = current_config["access_key"] + + if method == "access-key": + program_id = _prompt_program_id(current_config.get("program_id")) + if program_id is not None: + updated["program_id"] = program_id + return updated program_id = _prompt_program_id(current_config.get("program_id")) if program_id is not None: updated["program_id"] = program_id @@ -212,6 +322,16 @@ def account_from_args(args) -> None: current_raw = load_account_config(account_path) current = copy.deepcopy(DEFAULT_BOHRIUM_CONFIG) _deep_update(current, current_raw) + account_changed = False + clear_target = getattr(args, "clear", None) + clear_keys = { + "all": ("email", "password", "access_key"), + "email": ("email", "password"), + "access-key": ("access_key",), + } + for key in clear_keys.get(clear_target, ()): + if current.pop(key, None) is not None: + account_changed = True cli_updates = {} for key in ( @@ -221,6 +341,7 @@ def account_from_args(args) -> None: "context_type", "email", "password", + "access_key", "program_id", "apex_image_name", ): @@ -228,14 +349,19 @@ def account_from_args(args) -> None: if value is not None: cli_updates[key] = value - if not cli_updates and not args.show and not args.non_interactive: + if ( + not cli_updates + and clear_target is None + and not args.show + and not args.non_interactive + ): cli_updates = prompt_for_account_fields(current) - if cli_updates: + if cli_updates or account_changed: _deep_update(current, cli_updates) saved_path = save_account_config(current, account_path) print(f"Saved account config to {saved_path}") - if args.show or not cli_updates: + if args.show or not (cli_updates or account_changed): print(json.dumps(mask_sensitive_config(current), indent=2)) print(f"Config path: {account_path}") diff --git a/apex/config.py b/apex/config.py index b26d3cb2..0afa6c9a 100644 --- a/apex/config.py +++ b/apex/config.py @@ -8,6 +8,11 @@ from apex.utils import update_dict +# Default dispatcher sidecar image with storeHost fix for OpenAPI Sandbox +SANDBOX_DISPATCHER_IMAGE = ( + "registry.dp.tech/dptech/polycalibur:dpdispatcher-storehost-plan-a-20260811" +) + @dataclass class Config: @@ -28,10 +33,20 @@ class Config: phone: str = None email: str = None password: str = None + access_key: str = None + app_key: str = None program_id: int = None job_type: str = "container" platform: str = "ali" + # OpenAPI Sandbox config + project_id: int = None + machine_type: str = None + image_address: str = None + output_log: bool = False + ignore_exit_code: bool = False + sandbox_job_name: str = None + # DispachterExecutor config dispatcher_config: dict = None dispatcher_image: str = None @@ -62,7 +77,7 @@ class Config: exclude_upload_files: list = field(default_factory=list) lammps_image_name: str = ( "registry.dp.tech/dptech/dp/native/prod-397637/" - "deepmd-kit-phonolammps:3.1.3" + "apex-flow:1.3.0.post" ) lammps_run_command: str = None phonolammps_run_command: str = None @@ -97,11 +112,15 @@ class Config: dynamodb_table_name: str = "apex_results" def __post_init__(self): - # judge if running dflow on the Bohrium + # judge if running dflow on the Bohrium (or OpenAPI Sandbox dflow) + _bohrium_dflow_hosts = { + "https://workflows.deepmodeling.com", + "https://lbg-workflow-dflow.dp.tech", + } try: - assert self.dflow_config["host"] == "https://workflows.deepmodeling.com" + assert self.dflow_config["host"] in _bohrium_dflow_hosts except (AssertionError, TypeError): - if self.dflow_host == "https://workflows.deepmodeling.com": + if self.dflow_host in _bohrium_dflow_hosts: self.is_bohrium_dflow = True else: self.is_bohrium_dflow = True @@ -127,6 +146,31 @@ def __post_init__(self): } if self.machine: update_dict(self.machine_dict, self.machine) + elif self.context_type in ["OpenAPI", "openapi"]: + # OpenAPI Sandbox mode — uses access_key auth and machine_type + _ak = self.access_key + _pid = self.project_id or self.program_id + self.machine_dict = { + "batch_type": "OpenAPI", + "context_type": "OpenAPI", + "local_root": self.local_root or "./work", + "remote_profile": { + "access_key": _ak, + "project_id": int(_pid) if _pid else None, + "app_key": self.app_key or "agent", + "image_address": self.image_address or self.run_image_name, + "platform": self.platform or "ali", + "machine_type": self.machine_type, + "job_name": self.sandbox_job_name or "apex-sandbox-job", + "output_log": self.output_log, + "ignore_exit_code": self.ignore_exit_code, + }, + } + # For OpenAPI mode, ensure dispatcher image is set + if not self.dispatcher_image: + self.dispatcher_image = SANDBOX_DISPATCHER_IMAGE + if self.machine: + update_dict(self.machine_dict, self.machine) elif self.context_type in ["SSHContext", "sshcontext", "SSH", "ssh"]: self.machine_dict = { @@ -195,7 +239,18 @@ def dflow_config_dict(self): @property def dflow_s3_config_dict(self): dflow_s3_config = {} - if self.is_bohrium_dflow: + if self.context_type in ["OpenAPI", "openapi"]: + # OpenAPI Sandbox: use access_key-based TiefblueClient + _pid = self.project_id or self.program_id + dflow_s3_config = { + "repo_key": "oss-bohrium", + "storage_client": TiefblueClient( + access_key=self.access_key, + project_id=int(_pid) if _pid else None, + app_key=self.app_key or "agent", + ) + } + elif self.is_bohrium_dflow: dflow_s3_config = { "repo_key": "oss-bohrium", "storage_client": TiefblueClient() @@ -206,12 +261,27 @@ def dflow_s3_config_dict(self): @property def bohrium_config_dict(self): - bohrium_config = { - "username": self.email, - "phone": self.phone, - "password": self.password, - "project_id": self.program_id - } + if self.context_type in ["OpenAPI", "openapi"]: + # OpenAPI Sandbox mode uses access_key authentication + _pid = self.project_id or self.program_id + bohrium_config = { + "access_key": self.access_key, + "project_id": int(_pid) if _pid else None, + "app_key": self.app_key or "agent", + } + else: + # Legacy Bohrium supports either email/password or AccessKey. + bohrium_config = { + "username": self.email, + "phone": self.phone, + "password": self.password, + "project_id": self.program_id, + "access_key": self.access_key, + # dflow sends this header when exchanging an AccessKey for a ticket. + "app_key": self.app_key if self.app_key is not None else ( + "" if self.access_key else None + ), + } if self.bohrium_config: update_dict(bohrium_config, self.bohrium_config) return bohrium_config @@ -227,6 +297,9 @@ def dispatcher_config_dict(self): "command": self.dispatcher_command, "remote_command": self.dispatcher_remote_command } + # OpenAPI Sandbox requires BOHRIUM_USE_SANDBOX=1 in dispatcher sidecar + if self.context_type in ["OpenAPI", "openapi"]: + dispatcher_config["envs"] = {"BOHRIUM_USE_SANDBOX": "1"} if self.dispatcher_config: update_dict(dispatcher_config, self.dispatcher_config) return dispatcher_config diff --git a/apex/core/calculator/Lammps.py b/apex/core/calculator/Lammps.py index 1084fd36..f7e74c45 100644 --- a/apex/core/calculator/Lammps.py +++ b/apex/core/calculator/Lammps.py @@ -59,6 +59,14 @@ def _render_phonon_input(conf, type_map, interaction, model_param, task_param=No ) +def _render_melting_point_input(conf, type_map, interaction, model_param, task_param=None): + from apex.core.property.MeltingPoint import render_melting_point_lammps_input + + return render_melting_point_lammps_input( + conf, type_map, interaction, model_param, task_param + ) + + def _finitetlatt_file_manifest(model_files, default_manifest): from apex.core.property.FiniteTlatt.lammps import get_lammps_file_manifest @@ -89,6 +97,7 @@ def _eos_runtime_policy(default_policy): "gamma": _render_gamma_input, "gamma_surface": _render_gamma_input, "phonon": _render_phonon_input, + "melting_point": _render_melting_point_input, } PROPERTY_LAMMPS_FILE_MANIFESTS = { @@ -140,6 +149,24 @@ def __init__(self, inter_parameter, path_to_poscar): self.inter_type = inter_parameter["type"] self.type_map = inter_parameter["type_map"] self.in_lammps = inter_parameter.get("in_lammps", "auto") + self.model_in_image = inter_parameter.get("model_in_image", False) + if not isinstance(self.model_in_image, bool): + raise ValueError("interaction.model_in_image must be a boolean") + if self.model_in_image: + model = inter_parameter.get("model") + if not lammps_utils.is_deepmd_pt2(inter_parameter): + raise ValueError( + "interaction.model_in_image is only supported with " + "type=deepmd and deepmd_runtime=dpa4_pt2" + ) + if ( + not isinstance(model, str) + or not os.path.isabs(model) + or not model.lower().endswith(".pt2") + ): + raise ValueError( + "an image-resident DPA4 model must be an absolute .pt2 path" + ) if self.inter_type in MULTI_MODELS_INTER_TYPE: self.model = list(map(os.path.abspath, inter_parameter["model"])) else: @@ -173,13 +200,17 @@ def set_inter_type_func(self): def set_model_param(self): deepmd_version = self.inter.get("deepmd_version", "2.1.1") if self.inter_type == "deepmd": - model_name = os.path.basename(self.model) + model_name = self.model if self.model_in_image else os.path.basename(self.model) self.model_param = { "type": self.inter_type, "model_name": [model_name], "param_type": self.type_map, "deepmd_version": deepmd_version, } + if "deepmd_runtime" in self.inter: + self.model_param["deepmd_runtime"] = self.inter["deepmd_runtime"] + if self.model_in_image: + self.model_param["model_in_image"] = True elif self.inter_type in ["meam", "snap"]: model_name = list(map(os.path.basename, self.model)) self.model_param = { @@ -222,6 +253,10 @@ def symlink_force(self, target, link_name): os.symlink(target, link_name) def make_potential_files(self, output_dir): + if self.model_in_image: + dumpfn(self.inter, os.path.join(output_dir, "inter.json"), indent=4) + return + parent_dir = os.path.join(output_dir, "../../") if self.inter_type in MULTI_MODELS_INTER_TYPE: model_file = map(os.path.basename, self.model) @@ -301,7 +336,32 @@ def make_input_file(self, output_dir, task_type, task_param): ) maxeval = cal_setting["maxeval"] - if cal_type == "relaxation": + if task_type == "melting_point": + fc = _render_melting_point_input( + "conf.lmp", + self.type_map, + self.inter_func, + self.model_param, + task_param, + ) + elif task_type == "finite_t_latt": + fc = lammps_utils.make_lammps_FiniteTlatt( + "conf.lmp", + self.type_map, + self.inter_func, + self.model_param, + cal_setting, + ) + elif task_type in ["annealing", "Annealing"]: + # MD annealing schedule: equilibrate -> ramp -> hold -> cool + fc = lammps_utils.make_lammps_annealing( + "conf.lmp", + self.type_map, + self.inter_func, + self.model_param, + cal_setting, + ) + elif cal_type == "relaxation": relax_pos = cal_setting["relax_pos"] relax_shape = cal_setting["relax_shape"] relax_vol = cal_setting["relax_vol"] @@ -402,25 +462,6 @@ def make_input_file(self, output_dir, task_type, task_param): self.model_param, output_dir, ) - elif task_type in ["annealing", "Annealing"]: - # MD annealing schedule: equilibrate -> ramp -> hold -> cool - fc = lammps_utils.make_lammps_annealing( - "conf.lmp", - self.type_map, - self.inter_func, - self.model_param, - cal_setting, - ) - - elif task_type == "finite_t_latt": - fc = lammps_utils.make_lammps_FiniteTlatt( - "conf.lmp", - self.type_map, - self.inter_func, - self.model_param, - cal_setting, - ) - else: raise RuntimeError("not supported calculation type for LAMMPS") @@ -431,6 +472,9 @@ def make_input_file(self, output_dir, task_type, task_param): ): fc = _apply_gamma_fix_to_lammps_input(fc, task_param["add_fix"]) + # This also covers user-supplied and property-specific LAMMPS inputs. + fc = lammps_utils.ensure_atom_map_before_read_data(fc, self.model_param) + dumpfn(task_param, os.path.join(output_dir, "task.json"), indent=4) in_lammps_not_link_list = ["eos", "finite_t_elastic"] @@ -456,9 +500,26 @@ def compute(self, output_dir): except Exception: task_param = {} task_type = task_param.get("type", task_param.get("cal_type")) - if task_type in ["annealing", "Annealing"]: + if task_type in ["annealing", "Annealing", "melting_point"]: return None + status_path = os.path.join(output_dir, "apex_task_status.json") + if os.path.isfile(status_path): + try: + task_status = loadfn(status_path) + except Exception as exc: + raise RuntimeError( + f"cannot parse LAMMPS task status {status_path}: {exc}" + ) from exc + if task_status.get("state") != "succeeded": + raise RuntimeError( + "LAMMPS task failed before post-processing: " + f"state={task_status.get('state')!r}, " + f"reason={task_status.get('reason')!r}, " + f"exit_code={task_status.get('exit_code')!r}, " + f"message={task_status.get('message')!r}" + ) + log_lammps = os.path.join(output_dir, "log.lammps") dump_lammps = os.path.join(output_dir, "dump.relax") if not os.path.isfile(log_lammps) or not os.path.isfile(dump_lammps): @@ -488,6 +549,7 @@ def _parse_dump_file(self, dump_lammps, box, coord, vol, force): with open(dump_lammps, "r") as fin: dump = fin.read().split("\n") dumptime = [] + type_list = [] for idx, ii in enumerate(dump): if ii == "ITEM: TIMESTEP": box.append([]) @@ -539,6 +601,10 @@ def _parse_dump_file(self, dump_lammps, box, coord, vol, force): fz = float(dump[idx + 9 + jj].split()[7]) force[-1].append([fx, fy, fz]) + if not dumptime: + raise RuntimeError( + f"LAMMPS dump contains no TIMESTEP frames: {dump_lammps}" + ) return dumptime, type_list def _check_lammps_finished(self, log_lammps): @@ -679,10 +745,26 @@ def _prepare_result_dict(self, atom_numbs, type_map_list, type_list, box, coord, } return result_dict - def forward_files(self, property_type="relaxation"): - model_files = list(map(os.path.basename, self.model)) if self.inter_type in MULTI_MODELS_INTER_TYPE else [os.path.basename(self.model)] + def forward_files(self, property_type="relaxation", task_param=None): + model_files = [] if self.model_in_image else ( + list(map(os.path.basename, self.model)) + if self.inter_type in MULTI_MODELS_INTER_TYPE + else [os.path.basename(self.model)] + ) if property_type == "finite_t_latt": return ["in.lammps", "variable_FiniteTlatt.in"] + model_files + elif property_type == "melting_point": + files = [ + "in.lammps", + "variable_MeltingPoint.in", + "MeltingPoint.json", + ] + restart_files = (task_param or {}).get("cal_setting", {}).get( + "restart_files" + ) + if restart_files is not None: + files.append("restart.coexistence.start") + return files + model_files elif property_type in ["annealing", "Annealing"]: return ["in.lammps", "variable_Annealing.in"] + model_files elif property_type == "finite_t_elastic": @@ -700,10 +782,16 @@ def forward_files(self, property_type="relaxation"): return ["conf.lmp", "in.lammps"] + model_files def forward_common_files(self, property_type="relaxation"): - model_files = list(map(os.path.basename, self.model)) if self.inter_type in MULTI_MODELS_INTER_TYPE else [os.path.basename(self.model)] + model_files = [] if self.model_in_image else ( + list(map(os.path.basename, self.model)) + if self.inter_type in MULTI_MODELS_INTER_TYPE + else [os.path.basename(self.model)] + ) if property_type not in ["eos"]: if property_type == "finite_t_latt": return ["in.lammps", "variable_FiniteTlatt.in"] + model_files + elif property_type == "melting_point": + return ["in.lammps"] + model_files elif property_type in ["annealing", "Annealing"]: return ["in.lammps", "variable_Annealing.in"] + model_files elif property_type == "finite_t_elastic": @@ -732,6 +820,14 @@ def backward_files(self, property_type="relaxation"): ] elif property_type == "finite_t_latt": return ["log.lammps", "outlog"] + debug_files + ["dump.relax", "average_box.txt"] + elif property_type == "melting_point": + return [ + "log.lammps", + "outlog", + *debug_files, + "dump.melting", + "restart.melting.*", + ] elif property_type in ["annealing", "Annealing"]: return [ "log.lammps", diff --git a/apex/core/calculator/VASP.py b/apex/core/calculator/VASP.py index 8242e863..634f0bbe 100644 --- a/apex/core/calculator/VASP.py +++ b/apex/core/calculator/VASP.py @@ -1,8 +1,9 @@ import os import logging +import re from dpdata import LabeledSystem -from monty.serialization import dumpfn +from monty.serialization import dumpfn, loadfn from pymatgen.core.structure import Structure from pymatgen.io.vasp import Incar, Kpoints @@ -15,6 +16,8 @@ class VASP(Task): + _STAGE_PLAN = "apex_vasp_stage_plan.json" + def __init__(self, inter_parameter, path_to_poscar): self.inter = inter_parameter self.inter_type = inter_parameter["type"] @@ -23,6 +26,89 @@ def __init__(self, inter_parameter, path_to_poscar): self.potcars = inter_parameter["potcars"] self.path_to_poscar = path_to_poscar + @staticmethod + def _validate_md_encut(incar, potcar_path): + """Validate ENCUT against POTCAR ENMAX values without retaining POTCAR data.""" + if not os.path.isfile(potcar_path) or "ENCUT" not in incar: + return + enmax_values = [] + with open(potcar_path, encoding="utf-8", errors="ignore") as fp: + for line in fp: + match = re.search(r"\bENMAX\s*=\s*([-+0-9.eE]+)", line) + if match: + enmax_values.append(float(match.group(1))) + if not enmax_values: + return + required = 1.3 * max(enmax_values) + actual = float(incar["ENCUT"]) + if actual + 1.0e-12 < required: + raise ValueError( + f"ENCUT={actual:g} eV is below the MD minimum of 1.3 * " + f"max(POTCAR ENMAX)={required:g} eV" + ) + + @classmethod + def _md_base_incar( + cls, template, cal_setting, timestep_fs, pressure, gamma + ): + defaults = { + "PREC": "Accurate", + "EDIFF": 1.0e-6, + "ISMEAR": 1, + "SIGMA": 0.2, + "LASPH": True, + "LREAL": "Auto", + "ALGO": "Normal", + "IBRION": 0, + "MDALGO": 3, + "ISIF": 3, + "ISYM": 0, + "LWAVE": False, + "LCHARG": False, + "NBLOCK": 1, + } + base = Incar(dict(template)) + for tag, default in defaults.items(): + base[tag] = cal_setting.get(tag.lower(), cal_setting.get(tag, default)) + base.update( + { + "POTIM": timestep_fs, + "PSTRESS": pressure, + "LANGEVIN_GAMMA": gamma, + "LANGEVIN_GAMMA_L": float( + cal_setting.get("langevin_gamma_l", 10.0) + ), + "PMASS": float(cal_setting.get("pmass", 1000.0)), + } + ) + base.pop("SMASS", None) + for key in ("ediffg", "encut", "kspacing", "kgamma"): + if key in cal_setting: + base[key.upper()] = cal_setting[key] + return base + + @staticmethod + def _archive_stage_commands(suffix): + return [ + f"mv OUTCAR OUTCAR.{suffix}\n", + f"[ ! -f OSZICAR ] || mv OSZICAR OSZICAR.{suffix}\n", + f"cp CONTCAR CONTCAR.{suffix}\n", + f"[ ! -f XDATCAR ] || mv XDATCAR XDATCAR.{suffix}\n", + "cp CONTCAR POSCAR\n", + ] + + @classmethod + def _write_stage_plan(cls, output_dir, task_type, stages): + dumpfn( + { + "schema": "apex.vasp.stage-plan/v1", + "task_type": task_type, + "stages": stages, + }, + os.path.join(output_dir, cls._STAGE_PLAN), + indent=4, + ) + def make_potential_files(self, output_dir): potcar_not_link_list = {"vacancy", "interstitial"} task_type = output_dir.split("/")[-2].split("_")[0] @@ -69,6 +155,325 @@ def make_input_file(self, output_dir, task_type, task_param): prop_type = task_param.get("type", "relaxation") cal_type = task_param["cal_type"] cal_setting = task_param["cal_setting"] + md_incar = incar_relax + if "input_prop" in cal_setting and os.path.isfile(cal_setting["input_prop"]): + md_incar = incar_upper( + Incar.from_file(os.path.abspath(cal_setting["input_prop"])) + ) + + if task_type == "finite_t_latt": + metadata = loadfn(os.path.join(output_dir, "FiniteTlatt.json")) + temperature = float(metadata["temperature"]) + timestep_fs = float( + cal_setting.get( + "timestep_fs", 1000.0 * float(cal_setting.get("timestep", 0.001)) + ) + ) + pressure = float(cal_setting.get("pressure_kbar", 0.0)) + species = [] + for site in Structure.from_file(self.path_to_poscar): + name = site.specie.symbol + if name not in species: + species.append(name) + gamma = cal_setting.get("langevin_gamma", 10.0) + if isinstance(gamma, (int, float)): + gamma = [float(gamma)] * len(species) + else: + gamma = [float(value) for value in gamma] + if len(gamma) != len(species): + raise ValueError( + "langevin_gamma must contain one value per POSCAR species" + ) + + base = self._md_base_incar( + md_incar, cal_setting, timestep_fs, pressure, gamma + ) + base.update({"TEBEG": temperature, "TEEND": temperature}) + self._validate_md_encut(base, os.path.join(output_dir, "POTCAR")) + equi = Incar(dict(base)) + equi["NSW"] = int(cal_setting["equi_step"]) + production = Incar(dict(base)) + production["NSW"] = int(cal_setting["ave_step"]) + equi.write_file(os.path.join(output_dir, "INCAR.equi")) + production.write_file(os.path.join(output_dir, "INCAR.production")) + # Keep a conventional INCAR for VASP tooling and for the APEX Run + # OP's mandatory-input validation. The staged command replaces it + # explicitly before each VASP invocation. + equi.write_file(os.path.join(output_dir, "INCAR")) + + kspacing = base.get("KSPACING") + if kspacing is None: + raise RuntimeError("KSPACING must be given in INCAR") + kgamma = base.get("KGAMMA", False) + ret = vasp_utils.make_kspacing_kpoints( + self.path_to_poscar, kspacing, kgamma + ) + Kpoints.from_str(ret).write_file(os.path.join(output_dir, "KPOINTS")) + stage_plan = [] + commands = ["set -e\n"] + + schedule = cal_setting.get("nvt_temperature_schedule") + nvt_step = int(cal_setting.get("nvt_step", 0)) + if schedule is not None: + try: + schedule = [float(value) for value in schedule] + except (TypeError, ValueError) as exc: + raise ValueError( + "nvt_temperature_schedule must be a numeric list" + ) from exc + if len(schedule) < 2: + raise ValueError( + "nvt_temperature_schedule must contain at least " + "a start and end temperature" + ) + if nvt_step <= 0: + raise ValueError( + "nvt_step must be positive when " + "nvt_temperature_schedule is given" + ) + if abs(schedule[-1] - temperature) > 1.0e-8: + raise ValueError( + "the final nvt_temperature_schedule temperature must " + "match the finite_t_latt target temperature" + ) + for index, (first_temp, last_temp) in enumerate( + zip(schedule[:-1], schedule[1:]) + ): + suffix = f"nvt_{index:03d}" + incar_name = f"INCAR.{suffix}" + outcar_name = f"OUTCAR.{suffix}" + oszicar_name = f"OSZICAR.{suffix}" + nvt = Incar(dict(base)) + nvt.update({ + "ISIF": 2, + "NSW": nvt_step, + "TEBEG": first_temp, + "TEEND": last_temp, + }) + for key in ("LANGEVIN_GAMMA_L", "PMASS", "PSTRESS"): + nvt.pop(key, None) + nvt.write_file(os.path.join(output_dir, incar_name)) + if index == 0: + nvt.write_file(os.path.join(output_dir, "INCAR.nvt")) + stage_plan.append({ + "name": ( + f"nvt_{index:03d}_{first_temp:g}K_to_" + f"{last_temp:g}K" + ), + "incar": incar_name, + "outcar": outcar_name, + "oszicar": oszicar_name, + "contcar": f"CONTCAR.{suffix}", + "xdatcar": f"XDATCAR.{suffix}", + "expected_ionic_steps": nvt_step, + "temperature_start_K": first_temp, + "temperature_end_K": last_temp, + }) + commands.extend([ + f"cp {incar_name} INCAR\n", + 'eval "$APEX_RUN_COMMAND"\n', + *self._archive_stage_commands(suffix), + ]) + else: + nvt = Incar(dict(base)) + nvt["ISIF"] = 2 + nvt["NSW"] = nvt_step + for key in ("LANGEVIN_GAMMA_L", "PMASS", "PSTRESS"): + nvt.pop(key, None) + nvt.write_file(os.path.join(output_dir, "INCAR.nvt")) + if nvt_step > 0: + stage_plan.append({ + "name": "nvt", + "incar": "INCAR.nvt", + "outcar": "OUTCAR.nvt", + "oszicar": "OSZICAR.nvt", + "contcar": "CONTCAR.nvt", + "xdatcar": "XDATCAR.nvt", + "expected_ionic_steps": nvt_step, + "temperature_start_K": temperature, + "temperature_end_K": temperature, + }) + commands.extend([ + "cp INCAR.nvt INCAR\n", + 'eval "$APEX_RUN_COMMAND"\n', + *self._archive_stage_commands("nvt"), + ]) + + stage_plan.append({ + "name": "equi", + "incar": "INCAR.equi", + "outcar": "OUTCAR.equi", + "oszicar": "OSZICAR.equi", + "contcar": "CONTCAR.equi", + "xdatcar": "XDATCAR.equi", + "expected_ionic_steps": int(equi["NSW"]), + "temperature_start_K": temperature, + "temperature_end_K": temperature, + }) + commands.extend([ + "cp INCAR.equi INCAR\n", + 'eval "$APEX_RUN_COMMAND"\n', + *self._archive_stage_commands("equi"), + ]) + stage_plan.append({ + "name": "production", + "incar": "INCAR.production", + "outcar": "OUTCAR", + "oszicar": "OSZICAR", + "contcar": "CONTCAR", + "xdatcar": "XDATCAR", + "expected_ionic_steps": int(production["NSW"]), + "temperature_start_K": temperature, + "temperature_end_K": temperature, + }) + commands.extend([ + "cp INCAR.production INCAR\n", + 'eval "$APEX_RUN_COMMAND"\n', + ]) + with open(os.path.join(output_dir, "run_command"), "w") as fp: + fp.writelines(commands) + self._write_stage_plan( + output_dir, "finite_t_latt", stage_plan + ) + return + + if task_type == "annealing": + metadata = loadfn(os.path.join(output_dir, "Annealing.json")) + timestep_fs = float(metadata["timestep_fs"]) + pressure = float(cal_setting.get("pressure_kbar", 0.0)) + species = [] + for site in Structure.from_file(self.path_to_poscar): + name = site.specie.symbol + if name not in species: + species.append(name) + gamma = cal_setting.get("langevin_gamma", 10.0) + if isinstance(gamma, (int, float)): + gamma = [float(gamma)] * len(species) + else: + gamma = [float(value) for value in gamma] + if len(gamma) != len(species): + raise ValueError( + "langevin_gamma must contain one value per POSCAR species" + ) + base = self._md_base_incar( + md_incar, cal_setting, timestep_fs, pressure, gamma + ) + self._validate_md_encut(base, os.path.join(output_dir, "POTCAR")) + if metadata.get("protocol", "ramp_cool") == "coexistence": + stages = [ + ( + "equi", + metadata["target_temp"], + metadata["target_temp"], + metadata["equi_step"], + ), + ( + "production", + metadata["target_temp"], + metadata["target_temp"], + metadata["production_step"], + ), + ] + else: + stages = [ + ( + "eq", + metadata["start_temp"], + metadata["start_temp"], + metadata["equi_step"], + ), + ( + "ramp", + metadata["start_temp"], + metadata["target_temp"], + metadata["ramp_step"], + ), + ( + "decline", + metadata["target_temp"], + metadata["end_temp"], + metadata["cool_step"], + ), + ( + "final_eq", + metadata["end_temp"], + metadata["end_temp"], + metadata["final_equi_step"], + ), + ] + for name, first_temp, last_temp, nsteps in stages: + incar = Incar(dict(base)) + incar.update( + { + "TEBEG": float(first_temp), + "TEEND": float(last_temp), + "NSW": int(nsteps), + } + ) + incar.write_file(os.path.join(output_dir, f"INCAR.{name}")) + # Keep a stable transfer manifest across both protocols. + if metadata.get("protocol", "ramp_cool") == "coexistence": + aliases = { + "eq": "equi", + "ramp": "equi", + "decline": "production", + "final_eq": "production", + } + else: + aliases = {"equi": "eq", "production": "final_eq"} + for alias, source in aliases.items(): + Incar.from_file( + os.path.join(output_dir, f"INCAR.{source}") + ).write_file(os.path.join(output_dir, f"INCAR.{alias}")) + # The APEX Run OP requires INCAR during task staging. The staged + # command still selects the appropriate INCAR. at runtime. + Incar.from_file( + os.path.join(output_dir, f"INCAR.{stages[0][0]}") + ).write_file(os.path.join(output_dir, "INCAR")) + + kspacing = base.get("KSPACING") + if kspacing is None: + raise RuntimeError("KSPACING must be given in INCAR") + ret = vasp_utils.make_kspacing_kpoints( + self.path_to_poscar, kspacing, base.get("KGAMMA", False) + ) + Kpoints.from_str(ret).write_file(os.path.join(output_dir, "KPOINTS")) + stage_plan = [] + with open(os.path.join(output_dir, "run_command"), "w") as fp: + fp.write("set -e\nrm -f OUTCAR.apex XDATCAR.apex\n") + for name, _first, _last, nsteps in stages: + if int(nsteps) <= 0: + continue + fp.write( + f"cp INCAR.{name} INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + f"printf '\\nAPEX_STAGE {name}\\n' >> OUTCAR.apex\n" + "cat OUTCAR >> OUTCAR.apex\n" + f"printf '\\nAPEX_STAGE {name}\\n' >> XDATCAR.apex\n" + "[ ! -f XDATCAR ] || cat XDATCAR >> XDATCAR.apex\n" + f"cp OUTCAR OUTCAR.{name}\n" + f"[ ! -f OSZICAR ] || cp OSZICAR OSZICAR.{name}\n" + f"cp CONTCAR CONTCAR.{name}\n" + f"[ ! -f XDATCAR ] || cp XDATCAR XDATCAR.{name}\n" + "cp CONTCAR POSCAR\n" + ) + stage_plan.append({ + "name": name, + "incar": f"INCAR.{name}", + "outcar": f"OUTCAR.{name}", + "oszicar": f"OSZICAR.{name}", + "contcar": f"CONTCAR.{name}", + "xdatcar": f"XDATCAR.{name}", + "expected_ionic_steps": int(nsteps), + "temperature_start_K": float(_first), + "temperature_end_K": float(_last), + }) + fp.write( + "mv OUTCAR.apex OUTCAR\n" + "[ ! -f XDATCAR.apex ] || mv XDATCAR.apex XDATCAR\n" + ) + self._write_stage_plan(output_dir, "annealing", stage_plan) + return # user input INCAR for APEX calculation if "input_prop" in cal_setting and os.path.isfile(cal_setting["input_prop"]): @@ -200,6 +605,10 @@ def _link_file(self, target, link_name): os.symlink(target, link_name) def compute(self, output_dir): + task_json = os.path.join(output_dir, "task.json") + task_param = loadfn(task_json) if os.path.isfile(task_json) else {} + if task_param.get("type") in {"finite_t_latt", "annealing"}: + return None outcar = os.path.join(output_dir, "OUTCAR") if not os.path.isfile(outcar): logging.warning("cannot find OUTCAR in " + output_dir + " skip") @@ -233,9 +642,37 @@ def compute(self, output_dir): return outcar_dict def forward_files(self, property_type="relaxation"): + if property_type == "finite_t_latt": + return [ + "INCAR.nvt", + "INCAR.nvt.*", + "INCAR.equi", + "INCAR.production", + self._STAGE_PLAN, + "run_command", + "POSCAR", + "KPOINTS", + "POTCAR", + ] + if property_type == "annealing": + return [ + "INCAR.eq", + "INCAR.ramp", + "INCAR.decline", + "INCAR.final_eq", + "INCAR.equi", + "INCAR.production", + self._STAGE_PLAN, + "run_command", + "POSCAR", + "KPOINTS", + "POTCAR", + ] return ["INCAR", "POSCAR", "KPOINTS", "POTCAR"] def forward_common_files(self, property_type="relaxation"): + if property_type in {"finite_t_latt", "annealing"}: + return ["POTCAR"] potcar_not_link_list = ["vacancy", "interstitial"] if property_type == "elastic": return ["INCAR", "KPOINTS", "POTCAR"] @@ -245,6 +682,10 @@ def forward_common_files(self, property_type="relaxation"): return ["INCAR", "POTCAR"] def backward_files(self, property_type="relaxation"): + if property_type == "finite_t_latt": + return ["OUTCAR", "outlog", "OSZICAR", "CONTCAR", "XDATCAR"] + if property_type == "annealing": + return ["OUTCAR", "outlog", "OSZICAR", "XDATCAR", "CONTCAR"] if property_type in {"phonon", "gruneisen"}: return ["OUTCAR", "outlog", "CONTCAR", "OSZICAR", "XDATCAR", "vasprun.xml"] else: diff --git a/apex/core/calculator/__init__.py b/apex/core/calculator/__init__.py index 3fc41b72..6631ae29 100644 --- a/apex/core/calculator/__init__.py +++ b/apex/core/calculator/__init__.py @@ -10,3 +10,19 @@ 'mace', 'nep' ] + + +def lammps_model_files_for_cleanup(inter_param): + """Return staged model paths that APEX may remove after retrieval. + + Image-resident models are immutable runtime assets rather than staged task + files. Never turn their absolute paths into cleanup commands. + """ + if inter_param.get("model_in_image") is True: + return [] + model = inter_param.get("model") + if isinstance(model, str): + return [model] + if isinstance(model, list): + return list(model) + return [] diff --git a/apex/core/calculator/lib/lammps_utils.py b/apex/core/calculator/lib/lammps_utils.py index 4cadf8de..c1a9d197 100644 --- a/apex/core/calculator/lib/lammps_utils.py +++ b/apex/core/calculator/lib/lammps_utils.py @@ -14,6 +14,59 @@ upload_packages.append(__file__) +DEEPMD_RUNTIME_DPA4_PT2 = "dpa4_pt2" + + +def is_deepmd_pt2(param): + """Return whether an interaction uses the DeepMD AOTI ``.pt2`` runtime.""" + return ( + param.get("type") == "deepmd" + and str(param.get("deepmd_runtime", "")).strip().lower() + == DEEPMD_RUNTIME_DPA4_PT2 + ) + + +def ensure_atom_map_before_read_data(input_text, param): + """Enable a LAMMPS atom map before each PT2 box-loading command. + + DPA4 AOTI inference consumes LAMMPS atom IDs through the DeepMD plugin. + Keep legacy DeepMD inputs byte-for-byte unchanged unless the interaction + explicitly opts into ``deepmd_runtime: dpa4_pt2``. + """ + if not is_deepmd_pt2(param): + return input_text + + lines = input_text.splitlines(keepends=True) + rewritten = [] + previous_command = [] + for line in lines: + command = line.split("#", 1)[0].strip().split() + if ( + command[:2] == ["plugin", "load"] + and len(command) >= 3 + and os.path.basename(command[2].strip("'\"")) == "libdeepmd_lmp.so" + ): + raise ValueError( + "DPA4/PT2 inputs must use LAMMPS_PLUGIN_PATH auto-loading; " + "explicitly loading legacy libdeepmd_lmp.so is not supported" + ) + if ( + command + and command[0] in {"read_data", "read_restart"} + and not ( + previous_command[:2] == ["atom_modify", "map"] + and len(previous_command) >= 3 + and previous_command[2].lower() in {"yes", "array", "hash"} + ) + ): + newline = "\r\n" if line.endswith("\r\n") else "\n" + rewritten.append(f"atom_modify map yes{newline}") + rewritten.append(line) + if command: + previous_command = command + return "".join(rewritten) + + def cvt_lammps_conf(fin, fout, type_map, ofmt="lammps/data"): """ Format convert from fin to fout, specify the output format by ofmt @@ -262,7 +315,7 @@ def make_lammps_eval(conf, type_map, interaction, param): ret += "dimension 3\n" ret += "boundary p p p\n" ret += "atom_style atomic\n" - if param["type"] == "mace": + if param["type"] in {"deepmd", "mace"}: ret += "atom_modify map yes\n" ret += "newton on\n" ret += "box tilt large\n" @@ -298,7 +351,7 @@ def make_lammps_eval(conf, type_map, interaction, param): ret += 'print "Final volume per atoms = ${Vpa}"\n' ret += 'print "Final Base area = ${AA}"\n' ret += 'print "Final Stress (xx yy zz xy xz yz) = ${Pxx} ${Pyy} ${Pzz} ${Pxy} ${Pxz} ${Pyz}"\n' - return ret + return ensure_atom_map_before_read_data(ret, param) def make_lammps_equi( @@ -320,10 +373,9 @@ def make_lammps_equi( make lammps input for equilibritation """ deepmd_version = param.get("deepmd_version", None) - is_new_dpmd = False - if deepmd_version: - split_v = deepmd_version.split('.') - is_new_dpmd = bool(int(split_v[0]) >= 2 and int(split_v[1]) >= 1 and int(split_v[2]) >= 5) + is_new_dpmd = bool( + deepmd_version and Version(deepmd_version) >= Version("2.1.5") + ) prop_type = kwargs.get("prop_type", "others") dump_step = 100 # detour sychronizing problem of dumping in new version of deepmd-kit >=2.1.5 @@ -338,7 +390,7 @@ def make_lammps_equi( ret += "dimension 3\n" ret += "boundary p p p\n" ret += "atom_style atomic\n" - if param["type"] == "mace": + if param["type"] in {"deepmd", "mace"}: ret += "atom_modify map yes\n" ret += "newton on\n" ret += "box tilt large\n" @@ -385,7 +437,7 @@ def make_lammps_equi( ret += 'print "Final volume per atoms = ${Vpa}"\n' ret += 'print "Final Base area = ${AA}"\n' ret += 'print "Final Stress (xx yy zz xy xz yz) = ${Pxx} ${Pyy} ${Pzz} ${Pxy} ${Pxz} ${Pyz}"\n' - return ret + return ensure_atom_map_before_read_data(ret, param) def make_lammps_elastic( @@ -402,7 +454,7 @@ def make_lammps_elastic( ret += "dimension 3\n" ret += "boundary p p p\n" ret += "atom_style atomic\n" - if param["type"] == "mace": + if param["type"] in {"deepmd", "mace"}: ret += "atom_modify map yes\n" ret += "newton on\n" ret += "box tilt large\n" @@ -435,7 +487,7 @@ def make_lammps_elastic( ret += 'print "Final energy per atoms = ${Epa}"\n' ret += 'print "Final volume per atoms = ${Vpa}"\n' ret += 'print "Final Stress (xx yy zz xy xz yz) = ${Pxx} ${Pyy} ${Pzz} ${Pxy} ${Pxz} ${Pyz}"\n' - return ret + return ensure_atom_map_before_read_data(ret, param) def make_lammps_FiniteTlatt(conf, type_map, interaction, param, cal_setting=None): type_map_list = element_list(type_map) @@ -461,6 +513,9 @@ def make_lammps_FiniteTlatt(conf, type_map, interaction, param, cal_setting=None ret += "dimension 3\n" ret += "boundary p p p\n" ret += "atom_style atomic\n" + if param["type"] in {"deepmd", "mace"}: + ret += "atom_modify map yes\n" + ret += "newton on\n" ret += "box tilt large\n" ret += "read_data %s\n" % conf ret += "replicate ${nx} ${ny} ${nz}\n" @@ -518,7 +573,7 @@ def make_lammps_FiniteTlatt(conf, type_map, interaction, param, cal_setting=None ret += 'print "Final Base area = ${AA}"\n' ret += 'print "Final Stress (xx yy zz xy xz yz) = ${Pxx} ${Pyy} ${Pzz} ${Pxy} ${Pxz} ${Pyz}"\n' ret += 'print "Final Length (box_x box_y box_z) = ${lx} ${ly} ${lz}"\n' - return ret + return ensure_atom_map_before_read_data(ret, param) def make_lammps_FiniteTelastic(conf, type_map, interaction, param, task_dir="."): type_map_list = element_list(type_map) @@ -538,7 +593,7 @@ def setup_from_data(): text += "dimension 3\n" text += "boundary p p p\n" text += "atom_style atomic\n" - if param["type"] == "mace": + if param["type"] in {"deepmd", "mace"}: text += "atom_modify map yes\n" text += "newton on\n" text += "box tilt large\n" @@ -554,7 +609,7 @@ def setup_from_restart(): text += "dimension 3\n" text += "boundary p p p\n" text += "atom_style atomic\n" - if param["type"] == "mace": + if param["type"] in {"deepmd", "mace"}: text += "atom_modify map yes\n" text += "newton on\n" text += "box tilt large\n" @@ -614,7 +669,7 @@ def force_field_setup(): ret += "print \"Final energy per atoms = ${Epa}\"\n" ret += "print \"Final volume per atoms = ${Vpa}\"\n" ret += "print \"Final Stress (xx yy zz xy xz yz) = ${Pxx} ${Pyy} ${Pzz} ${Pxy} ${Pxz} ${Pyz}\"\n" - return ret + return ensure_atom_map_before_read_data(ret, param) def make_lammps_press_relax( conf, @@ -650,7 +705,7 @@ def make_lammps_press_relax( ret += "dimension 3\n" ret += "boundary p p p\n" ret += "atom_style atomic\n" - if param["type"] == "mace": + if param["type"] in {"deepmd", "mace"}: ret += "atom_modify map yes\n" ret += "newton on\n" ret += "box tilt large\n" @@ -687,7 +742,7 @@ def make_lammps_press_relax( ret += 'print "Final energy per atoms = ${Epa} eV"\n' ret += 'print "Final volume per atoms = ${Vpa} A^3"\n' ret += 'print "Final Stress (xx yy zz xy xz yz) = ${Pxx} ${Pyy} ${Pzz} ${Pxy} ${Pxz} ${Pyz}"\n' - return ret + return ensure_atom_map_before_read_data(ret, param) def make_lammps_annealing(conf, type_map, interaction, param, cal_setting): """LAMMPS input for annealing using the same stage controls as annealing/. @@ -777,6 +832,9 @@ def _unfix_stage_analysis(stage): ret += "dimension\t3\n" ret += "boundary\tp p p\n" ret += "atom_style\tatomic\n" + if param["type"] in {"deepmd", "mace"}: + ret += "atom_modify map yes\n" + ret += "newton on\n" ret += "box tilt large\n" ret += "read_data %s\n" % conf ret += "replicate ${nx} ${ny} ${nz}\n" @@ -890,7 +948,7 @@ def _unfix_stage_analysis(stage): ret += 'print "__end_of_lmp_annealing_calculation__"\n' ret += 'label end_of_run\n' - return ret + return ensure_atom_map_before_read_data(ret, param) """ def make_lammps_phonon( diff --git a/apex/core/calculator/lib/vasp_utils.py b/apex/core/calculator/lib/vasp_utils.py index 936b23a6..06570df0 100644 --- a/apex/core/calculator/lib/vasp_utils.py +++ b/apex/core/calculator/lib/vasp_utils.py @@ -37,6 +37,29 @@ def incar_upper(dincar): return Incar(standard_incar) +def _poscar_coordinate_block(lines, natoms): + """Return the coordinate header and atom lines for a VASP 5 POSCAR. + + VASP writes ``Selective dynamics`` between the species/count lines and + the coordinate mode when constraints are present. Gamma workflows that + start from a VASP-relaxed CONTCAR must preserve that extra header and its + per-atom flags instead of assuming that coordinates always start at line + index eight. + """ + header_index = 7 + selective = lines[header_index].strip().lower().startswith("s") + coordinate_mode_index = header_index + int(selective) + coordinate_start = coordinate_mode_index + 1 + positions = lines[coordinate_start : coordinate_start + int(natoms)] + if len(positions) != int(natoms) or any(not line.split() for line in positions): + raise RuntimeError( + f"Malformed POSCAR coordinate block: expected {int(natoms)} atom lines" + ) + header = ["Selective dynamics"] if selective else [] + header.append(lines[coordinate_mode_index].strip()) + return header, positions + + def regulate_poscar(poscar_in, poscar_out): with open(poscar_in, "r") as fp: lines = fp.read().split("\n") @@ -51,7 +74,7 @@ def regulate_poscar(poscar_in, poscar_out): for nn, cc in zip(names, counts): uniq_count[uniq_name.index(nn)] += cc natoms = np.sum(uniq_count) - posis = lines[8 : 8 + natoms] + coordinate_header, posis = _poscar_coordinate_block(lines, natoms) all_lines = [] for ele in uniq_name: ele_lines = [] @@ -64,7 +87,7 @@ def regulate_poscar(poscar_in, poscar_out): ret = lines[0:5] ret.append(" ".join(uniq_name)) ret.append(" ".join([str(ii) for ii in uniq_count])) - ret.append("Direct") + ret.extend(coordinate_header) ret += all_lines with open(poscar_out, "w") as fp: fp.write("\n".join(ret)) @@ -75,11 +98,11 @@ def sort_poscar(poscar_in, poscar_out, new_names): lines = fp.read().split("\n") names = lines[5].split() counts = [int(ii) for ii in lines[6].split()] - new_counts = np.zeros(len(counts), dtype=int) + new_counts = np.zeros(len(new_names), dtype=int) for nn, cc in zip(names, counts): new_counts[new_names.index(nn)] += cc natoms = np.sum(new_counts) - posis = lines[8 : 8 + natoms] + coordinate_header, posis = _poscar_coordinate_block(lines, natoms) all_lines = [] for ele in new_names: ele_lines = [] @@ -92,7 +115,7 @@ def sort_poscar(poscar_in, poscar_out, new_names): ret = lines[0:5] ret.append(" ".join(new_names)) ret.append(" ".join([str(ii) for ii in new_counts])) - ret.append("Direct") + ret.extend(coordinate_header) ret += all_lines with open(poscar_out, "w") as fp: fp.write("\n".join(ret)) @@ -549,4 +572,3 @@ def make_vasp_kpoints_from_incar(work_dir, jdata): kp.write_file("KPOINTS") os.chdir(cwd) - diff --git a/apex/core/common_prop.py b/apex/core/common_prop.py index f0d281c7..114d85c3 100644 --- a/apex/core/common_prop.py +++ b/apex/core/common_prop.py @@ -18,6 +18,7 @@ from apex.core.property.GammaSurface import GammaSurface from apex.core.property.Gruneisen import Gruneisen from apex.core.property.Annealing import Annealing +from apex.core.property.MeltingPoint import MeltingPoint from apex.core.lib.utils import create_path from apex.core.lib.util import collect_task from apex.core.lib.dispatcher import make_submission @@ -69,6 +70,11 @@ def make_property_instance(parameters, inter_param): return Gruneisen(parameters, inter_param) elif prop_type in ["annealing", "Annealing"]: return Annealing(parameters, inter_param) + elif prop_type in ["melting_point", "two_phase_melting"]: + if prop_type == "two_phase_melting": + parameters = dict(parameters) + parameters["type"] = "melting_point" + return MeltingPoint(parameters, inter_param) else: raise RuntimeError(f"unknown APEX type {prop_type}") @@ -202,7 +208,11 @@ def run_property(confs, inter_param, property_list, mdata): # dispatch the tasks # POSCAR here is useless virtual_calculator = make_calculator(inter_param_prop, "POSCAR") - forward_files = virtual_calculator.forward_files(property_type) + inter_type = inter_param_prop["type"] + if inter_type in lammps_task_type: + forward_files = virtual_calculator.forward_files(property_type, jj) + else: + forward_files = virtual_calculator.forward_files(property_type) forward_common_files = virtual_calculator.forward_common_files( property_type ) @@ -210,7 +220,6 @@ def run_property(confs, inter_param, property_list, mdata): # backward_files += logs # ... task_type = get_task_type({"interaction": inter_param}) - inter_type = inter_param_prop["type"] work_path = path_to_work if rerun_finished: all_task = tmp_task_list diff --git a/apex/core/lib/dispatcher.py b/apex/core/lib/dispatcher.py index 3dc07eb3..cd328abe 100644 --- a/apex/core/lib/dispatcher.py +++ b/apex/core/lib/dispatcher.py @@ -1,5 +1,6 @@ import logging import os +import shlex from dpdispatcher import ( Machine, Resources, @@ -7,6 +8,10 @@ Task ) from dflow.python import upload_packages +from apex.core.lib.vasp_runtime import ( + build_kpoint_aware_vasp_command, + is_switchable_vasp_command, +) upload_packages.append(__file__) @@ -39,14 +44,26 @@ def make_submission( task_list = [] for ii in run_tasks: + task_command = command # execute injected run command injected_run_command = os.path.join(work_path, ii, "run_command") - if os.path.isfile(injected_run_command): + has_injected_run_command = os.path.isfile(injected_run_command) + kpoints_path = os.path.join(work_path, ii, "KPOINTS") + if ( + os.path.isfile(kpoints_path) + and is_switchable_vasp_command(command) + ): + task_command = build_kpoint_aware_vasp_command( + command, + staged_run_command=has_injected_run_command, + ) + elif has_injected_run_command: logging.info(msg=f"execute injected run_command file in {injected_run_command}") - with open(injected_run_command, "r") as f: - command = f.read() + task_command = ( + f"APEX_RUN_COMMAND={shlex.quote(command)} bash run_command" + ) task = Task( - command=command, + command=task_command, task_work_path=ii, forward_files=forward_files, backward_files=backward_files, @@ -63,4 +80,3 @@ def make_submission( backward_common_files=[], ) return submission - diff --git a/apex/core/lib/parent_lattice_mapping.py b/apex/core/lib/parent_lattice_mapping.py new file mode 100644 index 00000000..2a5c2993 --- /dev/null +++ b/apex/core/lib/parent_lattice_mapping.py @@ -0,0 +1,395 @@ +"""Parent-lattice mapping helpers for disordered alloy supercells. + +The public Gamma API uses crystallographic indices in the parent lattice. +This module keeps the supercell-specific basis conversion internal so a +chemically disordered (P1) RSS cell is not mistaken for its own parent cell. + +All lattice matrices follow pymatgen's row-vector convention:: + + r_cart = f @ lattice + super_lattice = M @ parent_lattice + +Consequently a parent direction and plane transform as:: + + d_super = d_parent @ inv(M) + h_super = h_parent @ M.T +""" + +from __future__ import annotations + +from dataclasses import dataclass +import itertools +import math + +import numpy as np +from dflow.python import upload_packages +from pymatgen.core import Structure +from scipy.optimize import linear_sum_assignment + +upload_packages.append(__file__) + + +_PARENT_BASES = { + "bcc": np.array([[0.0, 0.0, 0.0], [0.5, 0.5, 0.5]]), + "fcc": np.array( + [ + [0.0, 0.0, 0.0], + [0.0, 0.5, 0.5], + [0.5, 0.0, 0.5], + [0.5, 0.5, 0.0], + ] + ), + "hcp": np.array([[0.0, 0.0, 0.0], [2.0 / 3.0, 1.0 / 3.0, 0.5]]), +} + + +@dataclass(frozen=True) +class ParentLatticeMapping: + """Resolved relationship between a parent conventional cell and a bulk.""" + + parent_lattice: str + supercell_matrix: np.ndarray + relaxed_parent_lattice: np.ndarray + determinant: int + metric_score: float + site_rms: float + site_max: float + candidate_count: int + source: str + + def as_dict(self) -> dict: + return { + "parent_lattice": self.parent_lattice, + "supercell_matrix": self.supercell_matrix.astype(int).tolist(), + "relaxed_parent_lattice": self.relaxed_parent_lattice.tolist(), + "determinant": int(self.determinant), + "metric_score": float(self.metric_score), + "site_rms_angstrom": float(self.site_rms), + "site_max_angstrom": float(self.site_max), + "candidate_count": int(self.candidate_count), + "source": self.source, + "row_vector_convention": True, + } + + +@dataclass(frozen=True) +class ParentSlipGeometry: + """A parent slip system expressed in the actual relaxed bulk geometry.""" + + parent_plane: np.ndarray + parent_direction: np.ndarray + mapped_plane_full: np.ndarray + slab_miller: np.ndarray + mapped_direction: np.ndarray + plane_normal_cart: np.ndarray + direction_cart: np.ndarray + burgers_fraction: float + burgers_vector_cart: np.ndarray + local_frame: np.ndarray + + def as_dict(self) -> dict: + return { + "parent_plane_miller": self.parent_plane.tolist(), + "parent_slip_direction": self.parent_direction.tolist(), + "mapped_plane_full": self.mapped_plane_full.tolist(), + "slab_miller": self.slab_miller.astype(int).tolist(), + "mapped_direction": self.mapped_direction.tolist(), + "plane_normal_cart": self.plane_normal_cart.tolist(), + "direction_cart": self.direction_cart.tolist(), + "burgers_fraction": float(self.burgers_fraction), + "burgers_vector_cart": self.burgers_vector_cart.tolist(), + "burgers_length_angstrom": float( + np.linalg.norm(self.burgers_vector_cart) + ), + "local_frame_rows": self.local_frame.tolist(), + } + + +def _factor_triples(number: int): + for first in range(1, number + 1): + if number % first: + continue + remaining = number // first + for second in range(1, remaining + 1): + if remaining % second: + continue + yield first, second, remaining // second + + +def _diagonal_candidates(determinant: int): + seen = set() + for values in _factor_triples(determinant): + for diagonal in set(itertools.permutations(values)): + matrix = np.diag(diagonal).astype(int) + key = tuple(matrix.ravel()) + if key not in seen: + seen.add(key) + yield matrix + + +def _hnf_candidates(determinant: int): + """Yield row-HNF candidates used as a fail-closed fallback. + + APEX RSS structures normally retain a diagonal/permuted conventional-cell + basis. HNF coverage handles externally supplied non-diagonal supercells + without exposing a matrix parameter to the user. + """ + + for a, e, g in _factor_triples(determinant): + for b, c, f in itertools.product(range(a), range(a), range(e)): + yield np.array([[a, b, c], [0, e, f], [0, 0, g]], dtype=int) + + +def _parent_metric_score(parent_lattice: str, lattice: np.ndarray) -> float: + lengths = np.linalg.norm(lattice, axis=1) + if np.any(lengths <= 0): + return float("inf") + cosines = np.array( + [ + np.dot(lattice[1], lattice[2]) / (lengths[1] * lengths[2]), + np.dot(lattice[0], lattice[2]) / (lengths[0] * lengths[2]), + np.dot(lattice[0], lattice[1]) / (lengths[0] * lengths[1]), + ] + ) + if parent_lattice in {"bcc", "fcc"}: + scale = float(np.mean(lengths)) + length_error = np.linalg.norm(lengths / scale - 1.0) + angle_error = np.linalg.norm(cosines) + return float(np.hypot(length_error, angle_error)) + + # Conventional HCP: a == b, alpha == beta == 90 deg, gamma == 120 deg. + a_scale = float(np.mean(lengths[:2])) + length_error = abs(lengths[0] - lengths[1]) / a_scale + angle_error = np.linalg.norm(cosines - np.array([0.0, 0.0, -0.5])) + return float(np.hypot(length_error, angle_error)) + + +def _site_fit( + structure: Structure, + parent_lattice: str, + matrix: np.ndarray, +) -> tuple[float, float]: + parent_matrix = np.linalg.inv(matrix) @ structure.lattice.matrix + basis = _PARENT_BASES[parent_lattice] + parent = Structure( + parent_matrix, + ["H"] * len(basis), + basis, + ) + parent.make_supercell(matrix) + if len(parent) != len(structure): + return float("inf"), float("inf") + distances = structure.lattice.get_all_distances( + structure.frac_coords, parent.frac_coords + ) + rows, cols = linear_sum_assignment(distances) + matched = distances[rows, cols] + return float(np.sqrt(np.mean(matched**2))), float(np.max(matched)) + + +def resolve_parent_supercell( + structure: Structure, + parent_lattice: str, + *, + max_metric_score: float = 0.45, + max_site_rms: float = 0.35, + max_site_distance: float = 0.65, +) -> ParentLatticeMapping: + """Infer the conventional-parent supercell matrix without user input. + + The resolver first tries the diagonal/permuted matrices emitted by normal + RSS workflows. It falls back to all HNF matrices with the required + determinant. Candidates are accepted only after an anonymous parent-site + bijection succeeds; chemical species are intentionally ignored. + """ + + parent_lattice = str(parent_lattice).strip().lower() + if parent_lattice not in _PARENT_BASES: + raise ValueError("parent_lattice must be one of: bcc, fcc, hcp") + basis_count = len(_PARENT_BASES[parent_lattice]) + if len(structure) % basis_count: + raise RuntimeError( + f"Cannot map {len(structure)} atoms to a {parent_lattice} " + f"conventional cell containing {basis_count} sites" + ) + determinant = len(structure) // basis_count + lattice = np.asarray(structure.lattice.matrix, dtype=float) + + def rank(candidates, source): + ranked = [] + accepted = [] + count = 0 + for matrix in candidates: + count += 1 + try: + parent_matrix = np.linalg.solve(matrix, lattice) + except np.linalg.LinAlgError: + continue + score = _parent_metric_score(parent_lattice, parent_matrix) + if np.isfinite(score): + ranked.append((score, tuple(matrix.ravel()), matrix, parent_matrix)) + ranked.sort(key=lambda item: (item[0], item[1])) + for score, _, matrix, parent_matrix in ranked[:32]: + if score > max_metric_score: + break + site_rms, site_max = _site_fit(structure, parent_lattice, matrix) + if site_rms <= max_site_rms and site_max <= max_site_distance: + accepted.append( + (site_rms, site_max, score, tuple(matrix.ravel()), matrix, parent_matrix) + ) + if not accepted: + return None + accepted.sort(key=lambda item: item[:4]) + best = accepted[0] + if len(accepted) > 1: + second = accepted[1] + if ( + abs(second[0] - best[0]) < 1e-8 + and abs(second[1] - best[1]) < 1e-8 + and abs(second[2] - best[2]) < 1e-8 + and second[3] != best[3] + ): + raise RuntimeError( + "Parent-supercell mapping is ambiguous between equally " + "good integer matrices; APEX will not guess the index basis" + ) + site_rms, site_max, score, _, matrix, parent_matrix = best + return ParentLatticeMapping( + parent_lattice=parent_lattice, + supercell_matrix=matrix, + relaxed_parent_lattice=parent_matrix, + determinant=determinant, + metric_score=score, + site_rms=site_rms, + site_max=site_max, + candidate_count=count, + source=source, + ) + + mapping = rank(_diagonal_candidates(determinant), "automatic_diagonal") + if mapping is None: + mapping = rank(_hnf_candidates(determinant), "automatic_hnf") + if mapping is None: + raise RuntimeError( + "Could not infer a unique, low-residual parent-supercell mapping " + f"for the {parent_lattice} structure. APEX will not interpret " + "parent Miller indices directly in this P1 cell." + ) + return mapping + + +def _reduce_integer_vector(vector: np.ndarray) -> np.ndarray: + rounded = np.rint(vector).astype(int) + if not np.allclose(vector, rounded, atol=1e-8, rtol=0.0): + raise RuntimeError(f"Mapped Miller indices are not integral: {vector}") + divisor = 0 + for value in rounded: + divisor = math.gcd(divisor, abs(int(value))) + if divisor == 0: + raise RuntimeError("Miller plane cannot be the zero vector") + return rounded // divisor + + +def shortest_parent_translation_fraction( + parent_lattice: str, + direction: np.ndarray, + max_denominator: int = 24, +) -> float: + """Return the shortest parent Bravais translation along ``direction``.""" + + basis = _PARENT_BASES[parent_lattice] + direction = np.asarray(direction, dtype=float) + fractions = sorted( + { + numerator / denominator + for denominator in range(1, max_denominator + 1) + for numerator in range(1, denominator + 1) + if math.gcd(numerator, denominator) == 1 + } + ) + for fraction in fractions: + shifted = (basis + fraction * direction) % 1.0 + delta = shifted[:, None, :] - basis[None, :, :] + delta -= np.rint(delta) + distances = np.linalg.norm(delta, axis=2) + rows, cols = linear_sum_assignment(distances) + if np.max(distances[rows, cols]) < 1e-8: + return float(fraction) + raise RuntimeError( + f"Could not derive a parent-lattice translation along {direction.tolist()}" + ) + + +def resolve_parent_slip_geometry( + structure: Structure, + mapping: ParentLatticeMapping, + plane_miller, + slip_direction, + slip_length=None, +) -> ParentSlipGeometry: + """Map a user-facing parent slip system into the relaxed RSS cell.""" + + parent_plane = np.asarray(plane_miller, dtype=float) + parent_direction = np.asarray(slip_direction, dtype=float) + if parent_plane.shape != (3,) or parent_direction.shape != (3,): + raise RuntimeError("Parent-mapped Gamma currently requires 3-index vectors") + incidence = float(parent_plane @ parent_direction) + if not np.isclose(incidence, 0.0, atol=1e-10, rtol=0.0): + raise RuntimeError( + f"slip direction {slip_direction} is not on plane {plane_miller}" + ) + + matrix = np.asarray(mapping.supercell_matrix, dtype=float) + mapped_plane_full = parent_plane @ matrix.T + slab_miller = _reduce_integer_vector(mapped_plane_full) + mapped_direction = parent_direction @ np.linalg.inv(matrix) + lattice = np.asarray(structure.lattice.matrix, dtype=float) + direction_cart = mapped_direction @ lattice + normal_cart = mapped_plane_full @ np.linalg.inv(lattice).T + direction_norm = np.linalg.norm(direction_cart) + normal_norm = np.linalg.norm(normal_cart) + if direction_norm <= 0 or normal_norm <= 0: + raise RuntimeError("Resolved Gamma direction or plane normal is zero") + direction_unit = direction_cart / direction_norm + normal_unit = normal_cart / normal_norm + orthogonality = abs(float(direction_unit @ normal_unit)) + if orthogonality > 1e-8: + raise RuntimeError( + "Resolved parent slip direction is not in the resolved plane: " + f"normalized dot={orthogonality:.3e}" + ) + + if slip_length is None: + burgers_fraction = shortest_parent_translation_fraction( + mapping.parent_lattice, parent_direction + ) + elif isinstance(slip_length, (int, float, np.integer, np.floating)): + # Legacy scalar lengths are expressed in units of the conventional + # parent a. Convert them into a fraction of the supplied direction; + # the relaxed Cartesian vector is still obtained affinely below. + burgers_fraction = float(slip_length) / float( + np.linalg.norm(parent_direction) + ) + else: + raise RuntimeError( + "Parent-mapped Gamma accepts a scalar legacy slip_length or derives " + "the shortest parent translation when it is omitted" + ) + burgers_vector = burgers_fraction * direction_cart + second = np.cross(normal_unit, direction_unit) + second /= np.linalg.norm(second) + normal_unit = np.cross(direction_unit, second) + normal_unit /= np.linalg.norm(normal_unit) + local_frame = np.array([direction_unit, second, normal_unit]) + return ParentSlipGeometry( + parent_plane=parent_plane, + parent_direction=parent_direction, + mapped_plane_full=mapped_plane_full, + slab_miller=slab_miller, + mapped_direction=mapped_direction, + plane_normal_cart=normal_unit, + direction_cart=direction_cart, + burgers_fraction=burgers_fraction, + burgers_vector_cart=burgers_vector, + local_frame=local_frame, + ) diff --git a/apex/core/lib/slab_orientation.py b/apex/core/lib/slab_orientation.py index ffaef51e..0f34c26e 100644 --- a/apex/core/lib/slab_orientation.py +++ b/apex/core/lib/slab_orientation.py @@ -130,15 +130,62 @@ class SlabSlipSystem(object): } } + # Keep the complete legacy registry above for backwards compatibility, but + # use only the physically important slip systems below for automatic + # crystallographic defaults. Other systems fall back to the geometric + # in-plane check in Gamma and GammaSurface. + __recommended_system_keys = { + 'fcc': ( + '111x11-2', + '111x-1-12', + '111x-110', + '111x1-10', + ), + 'bcc': ( + '110x-111', + '110x1-1-1', + '112x11-1', + '112x-1-11', + '123x11-1', + '123x-1-11', + ), + 'hcp': ( + '0001x2-1-10', + '0001x1-100', + '0001x10-10', + '01-10x-2110', + '01-10x0001', + '01-10x-2113', + '-12-10x-1010', + '-12-10x0001', + '01-11x-2110', + '01-11x-12-1-3', + '01-11x0-112', + '-12-12x10-10', + '-12-12x1-213', + '-12-12x-12-1-3', + ), + } + @classmethod def atomic_system_dict(cls): return cls.__dict_atomic_system + @classmethod + def recommended_system_dict(cls): + return { + structure_type: { + key: cls.__dict_atomic_system[structure_type][key] + for key in keys + } + for structure_type, keys in cls.__recommended_system_keys.items() + } + @classmethod def hint_string(cls): print_str = 'structure \tplane_index \tslip_direction\n' - for struct in cls.__dict_atomic_system.keys(): - for orient in cls.__dict_atomic_system[struct].keys(): + for struct, systems in cls.recommended_system_dict().items(): + for orient in systems: plane, slip = orient.split('x') print_str += f'{struct} \t{plane} \t{slip}\n' return print_str diff --git a/apex/core/lib/vasp_runtime.py b/apex/core/lib/vasp_runtime.py new file mode 100644 index 00000000..790db3f0 --- /dev/null +++ b/apex/core/lib/vasp_runtime.py @@ -0,0 +1,62 @@ +"""Build per-task VASP commands from the generated KPOINTS grid.""" + +import re +import shlex + + +_VASP_EXECUTABLE_RE = re.compile(r"\bvasp_(?:std|gam)\b", re.IGNORECASE) +_GAMMA_ONLY_PROBE = ( + "awk 'NF { n++; " + "if (n == 3) style = tolower($1); " + "if (n == 4) ok = " + "(style == \"gamma\" && $1 == 1 && $2 == 1 && $3 == 1) " + "} END { exit !(n >= 4 && ok) }' KPOINTS" +) + + +def is_switchable_vasp_command(run_command: str) -> bool: + return bool( + isinstance(run_command, str) + and _VASP_EXECUTABLE_RE.search(run_command) + ) + + +def _replace_vasp_executable(run_command: str, executable: str) -> str: + if not isinstance(run_command, str) or not run_command.strip(): + raise ValueError("VASP run command must be a non-empty string") + if not _VASP_EXECUTABLE_RE.search(run_command): + raise ValueError( + "VASP run command must contain vasp_std or vasp_gam so APEX can " + "select the executable from the generated KPOINTS grid" + ) + return _VASP_EXECUTABLE_RE.sub(executable, run_command) + + +def build_kpoint_aware_vasp_command( + run_command: str, *, staged_run_command: bool = False +) -> str: + """Select vasp_gam for Gamma-centered 1x1x1, otherwise vasp_std. + + ``staged_run_command`` is used by properties whose task-local + ``run_command`` performs multiple VASP stages. The selected command is + passed through ``APEX_RUN_COMMAND`` so every stage uses the same binary. + """ + gamma_command = _replace_vasp_executable(run_command, "vasp_gam") + standard_command = _replace_vasp_executable(run_command, "vasp_std") + selector = ( + f"if {_GAMMA_ONLY_PROBE}; then " + f"APEX_RUN_COMMAND={shlex.quote(gamma_command)}; " + "echo 'APEX VASP executable: vasp_gam (Gamma 1x1x1)'; " + "else " + f"APEX_RUN_COMMAND={shlex.quote(standard_command)}; " + "echo 'APEX VASP executable: vasp_std (non-Gamma-only grid)'; " + "fi" + ) + if staged_run_command: + return ( + f"{selector}; " + "if [ -f run_command ]; then " + 'APEX_RUN_COMMAND="$APEX_RUN_COMMAND" bash run_command; ' + 'else eval "$APEX_RUN_COMMAND"; fi' + ) + return f'{selector}; eval "$APEX_RUN_COMMAND"' diff --git a/apex/core/lib/vasp_trajectory.py b/apex/core/lib/vasp_trajectory.py new file mode 100644 index 00000000..3abddaf2 --- /dev/null +++ b/apex/core/lib/vasp_trajectory.py @@ -0,0 +1,77 @@ +"""Small helpers for stage-concatenated VASP trajectory output.""" + +from __future__ import annotations + +import re + +from dflow.python import upload_packages + + +upload_packages.append(__file__) + + +_NUMBER = r"[-+]?(?:\d+(?:\.\d*)?|\.\d+)(?:[Ee][-+]?\d+)?" +_STAGE = re.compile(r"^\s*APEX_STAGE\s+(\S+)\s*$") + + +def split_apex_stage_sections(text): + """Group text by APEX_STAGE, preserving repeated sections in file order.""" + sections = {} + stage = "trajectory" + chunks = [] + + def flush(): + if chunks: + sections.setdefault(stage, []).append("".join(chunks)) + + for line in text.splitlines(keepends=True): + match = _STAGE.match(line) + if match: + flush() + stage = match.group(1) + chunks = [] + else: + chunks.append(line) + flush() + return {name: "".join(parts) for name, parts in sections.items()} + + +def parse_outcar_stage_geometry(text): + """Parse independent cell and volume samples from each OUTCAR stage.""" + geometry = {} + for stage, section in split_apex_stage_sections(text).items(): + cells = [] + lines = section.splitlines() + for index, line in enumerate(lines): + if "direct lattice vectors" not in line.lower(): + continue + try: + cell = [ + [float(value) for value in re.findall(_NUMBER, lines[index + row])[:3]] + for row in (1, 2, 3) + ] + except (IndexError, ValueError): + continue + if all(len(vector) == 3 for vector in cell): + cells.append(cell) + volumes = [ + float(value) + for value in re.findall( + rf"volume of cell\s*:\s*({_NUMBER})", section, flags=re.IGNORECASE + ) + ] + geometry[stage] = {"cells": cells, "volumes": volumes} + return geometry + + +def tail_align_stage_geometry(frames, geometry): + """Attach only available tail-aligned OUTCAR geometry to stage frames.""" + for stage, stage_frames in frames.items(): + samples = geometry.get(stage, {}) + for key, frame_key in (("cells", "cell"), ("volumes", "outcar_volume")): + values = samples.get(key, []) + count = min(len(stage_frames), len(values)) + if not count: + continue + for frame, value in zip(stage_frames[-count:], values[-count:]): + frame[frame_key] = value diff --git a/apex/core/property/Annealing.py b/apex/core/property/Annealing.py index bf87d415..5486ad73 100644 --- a/apex/core/property/Annealing.py +++ b/apex/core/property/Annealing.py @@ -1,13 +1,25 @@ import glob +import math import os import logging +import re from typing import List, Dict, Any +import numpy as np from monty.serialization import dumpfn, loadfn +from pymatgen.core.lattice import Lattice from pymatgen.core.structure import Structure +from dflow.python import upload_packages from apex.core.property.Property import Property from apex.core.calculator.lib import vasp_utils +from apex.core.lib.vasp_trajectory import ( + parse_outcar_stage_geometry, + split_apex_stage_sections, + tail_align_stage_geometry, +) + +upload_packages.append(__file__) def _as_bool(value) -> bool: @@ -21,9 +33,14 @@ def _as_bool(value) -> bool: class Annealing(Property): def __init__(self, parameter: Dict[str, Any], inter_param=None): - parameter["cal_type"] = "annealing" - self.parameter = parameter self.inter_param = inter_param if inter_param is not None else {"type": "lammps"} + if self.inter_param.get("type") == "abacus": + raise NotImplementedError( + "annealing does not support the ABACUS backend; " + "use LAMMPS or VASP" + ) + parameter["cal_type"] = "static" + self.parameter = parameter # geometry self.supercell_size = parameter.get("supercell_size", [2, 2, 2]) @@ -31,6 +48,22 @@ def __init__(self, parameter: Dict[str, Any], inter_param=None): # MD controls (independent knobs only) cal = parameter.setdefault("cal_setting", {}) + self.protocol = parameter.get("protocol", cal.get("protocol", "ramp_cool")) + if self.protocol not in {"ramp_cool", "coexistence"}: + raise ValueError( + "annealing protocol must be 'ramp_cool' or 'coexistence'" + ) + if self.protocol == "coexistence" and self.inter_param.get("type") != "vasp": + raise ValueError( + "annealing protocol='coexistence' currently supports only VASP" + ) + dft_backend = self.inter_param.get("type") == "vasp" + equi_default = ( + 5000 if self.protocol == "coexistence" else (100 if dft_backend else 20000) + ) + ramp_default = 200 if dft_backend else 0 + final_equi_default = 100 if dft_backend else 20000 + cool_default = 200 if dft_backend else 0 # Schedule defaults mirror annealing/spec. self.start_temp = float(cal.get("start_temp", 4)) _tgt = cal.get("target_temp", cal.get("temp", 300)) @@ -40,18 +73,37 @@ def __init__(self, parameter: Dict[str, Any], inter_param=None): self._has_cool_rate = "cool_rate" in cal self.temp_ramp_rate = cal.get("temp_ramp_rate", cal.get("ramp_rate", 1000)) self.cool_rate = cal.get("cool_rate", self.temp_ramp_rate) - self.equi_step = int(cal.get("equi_step", cal.get("init_thermo_equil_step", 20000))) - self.init_lgv_thermo_equil_step = int(cal.get("init_lgv_thermo_equil_step", 20000)) + self.equi_step = int( + cal.get("equi_step", cal.get("init_thermo_equil_step", equi_default)) + ) + self.init_lgv_thermo_equil_step = int( + cal.get("init_lgv_thermo_equil_step", equi_default) + ) self.init_thermo_equil_step = int(cal.get("init_thermo_equil_step", self.equi_step)) - self.final_thermo_equil_step = int(cal.get("final_thermo_equil_step", cal.get("hold_step", 20000))) + self.final_thermo_equil_step = int( + cal.get( + "final_thermo_equil_step", + cal.get("hold_step", final_equi_default), + ) + ) # Explicit step counts override rate-derived counts when provided. - self.ramp_step = int(cal.get("ramp_step", cal.get("temp_ramp_step", 0))) - self.cool_step = int(cal.get("cool_step", cal.get("temp_decline_step", 0))) + self.ramp_step = int( + cal.get("ramp_step", cal.get("temp_ramp_step", ramp_default)) + ) + self.cool_step = int( + cal.get("cool_step", cal.get("temp_decline_step", cool_default)) + ) self.hold_step = int(cal.get("hold_step", self.final_thermo_equil_step)) + self.production_step = int(cal.get("production_step", 10000)) # options self.thermostat = cal.get("thermostat", "nose_hoover") self.ensemble = cal.get("ensemble", "npt") - self.timestep = float(cal.get("timestep", 0.001)) + if "timestep_fs" in cal: + self.timestep_fs = float(cal["timestep_fs"]) + self.timestep = self.timestep_fs / 1000.0 + else: + self.timestep = float(cal.get("timestep", 0.001)) + self.timestep_fs = 1000.0 * self.timestep self.tdamp_factor = cal.get("tdamp_factor", 100) self.pdamp_factor = cal.get("pdamp_factor", 1000) self.tdamp = cal.get("tdamp") @@ -99,6 +151,7 @@ def task_param(self): cal.update( { "start_temp": self.start_temp, + "protocol": self.protocol, "target_temp": self.target_temp, "temp": self.target_temp, "end_temp": self.end_temp, @@ -110,6 +163,7 @@ def task_param(self): "ramp_step": self.ramp_step, "temp_ramp_step": self.ramp_step, "hold_step": self.hold_step, + "production_step": self.production_step, "cool_step": self.cool_step, "temp_decline_step": self.cool_step, "thermostat": self.thermostat, @@ -134,6 +188,7 @@ def task_param(self): "init_fmax_tol": self.init_fmax_tol, "init_stress_tol": self.init_stress_tol, "timestep": self.timestep, + "timestep_fs": self.timestep_fs, "req_compute_rdf": self.req_compute_rdf, "rdf_bins": self.rdf_bins, "rdf_cutoff": self.rdf_cutoff, @@ -163,9 +218,8 @@ def make_confs(self, path_to_work: str, path_to_equi: str, refine=False) -> List else: logging.warning("%s already exists" % path_to_work) - # Calculator selection happens downstream via make_calculator; do not hard-code here. - # Annealing is implemented for LAMMPS; other calculators should provide their own impls. - if not os.path.isdir(path_to_equi) or not os.path.isfile(os.path.join(path_to_equi, "CONTCAR")): + equi_structure = os.path.join(path_to_equi, "CONTCAR") + if not os.path.isdir(path_to_equi) or not os.path.isfile(equi_structure): raise RuntimeError("please finish relaxation before annealing") task_list: List[str] = [] @@ -181,17 +235,15 @@ def make_confs(self, path_to_work: str, path_to_equi: str, refine=False) -> List task_dir = os.path.join(path_to_work, f"task.{idx:06d}") os.makedirs(task_dir, exist_ok=True) - # Build POSCAR from relaxation import shutil - equi_contcar = os.path.join(path_to_equi, "CONTCAR") - shutil.copy(equi_contcar, os.path.join(task_dir, "POSCAR")) - # Load back to fetch lattice metrics if needed later + shutil.copy(equi_structure, os.path.join(task_dir, "POSCAR")) s_sorted = Structure.from_file(os.path.join(task_dir, "POSCAR")) + lattice_lengths = list(s_sorted.lattice.abc) # Derive integer replication from physical length if requested if self.supercell_length is not None: try: - a, b, c = s_sorted.lattice.abc + a, b, c = lattice_lengths import math sx, sy, sz = self.supercell_length nx = max(1, int(math.ceil(sx / a))) @@ -200,9 +252,13 @@ def make_confs(self, path_to_work: str, path_to_equi: str, refine=False) -> List self.supercell_size = [nx, ny, nz] except Exception as e: logging.warning(f"Failed to derive supercell_size from supercell_length: {e}") + if self.inter_param.get("type") == "vasp": + s_sorted.make_supercell(self.supercell_size) + s_sorted.to(filename=os.path.join(task_dir, "POSCAR")) # Persist params per task anneal_task = { + "protocol": self.protocol, "start_temp": self.start_temp, "target_temp": float(tgt), "temp": float(tgt), @@ -317,8 +373,25 @@ def make_confs(self, path_to_work: str, path_to_equi: str, refine=False) -> List var.append(f"variable dump_step equal {self.dump_step}") var.append(f"variable thermostat string {self.thermostat}") var.append(f"variable ensemble string {self.ensemble}") - with open(os.path.join(task_dir, "variable_Annealing.in"), "w") as fp: - fp.write("\n".join(var) + "\n") + if self.inter_param.get("type") != "vasp": + with open(os.path.join(task_dir, "variable_Annealing.in"), "w") as fp: + fp.write("\n".join(var) + "\n") + + anneal_task.update( + { + "equi_step": self.init_thermo_equil_step, + "production_step": self.production_step, + "ramp_step": rstep, + "cool_step": cstep, + "final_equi_step": self.final_thermo_equil_step, + "timestep_fs": self.timestep_fs, + "req_compute_rdf": self.req_compute_rdf, + "req_compute_msd": self.req_compute_msd, + "rdf_bins": self.rdf_bins, + "rdf_cutoff": self.rdf_cutoff, + } + ) + dumpfn(anneal_task, os.path.join(task_dir, "Annealing.json"), indent=4) task_list.append(task_dir) @@ -348,18 +421,297 @@ def _collect_task_result(cls, task_dir: str) -> Dict[str, Any]: task_path = os.path.abspath(task_dir) metadata_path = os.path.join(task_path, "Annealing.json") status_path = os.path.join(task_path, "apex_task_status.json") + metadata = cls._safe_load_json(metadata_path) + dft_result = cls._collect_dft_result(task_path, metadata) result = { "task": os.path.basename(task_path), "path": task_path, - "metadata": cls._safe_load_json(metadata_path), + "metadata": metadata, "status": cls._safe_load_json(status_path), - "rdf": cls._collect_rdf(task_path), - "msd": cls._collect_msd(task_path), - "volume_temperature": cls._collect_volume_temperature(task_path), + "rdf": dft_result.get("rdf", cls._collect_rdf(task_path)), + "msd": dft_result.get("msd", cls._collect_msd(task_path)), + "volume_temperature": dft_result.get( + "volume_temperature", cls._collect_volume_temperature(task_path) + ), } result["summary"] = cls._build_summary(result) return result + @classmethod + def _collect_dft_result(cls, task_path, metadata): + xdatcar = os.path.join(task_path, "XDATCAR") + if os.path.isfile(xdatcar): + frames = cls._parse_vasp_xdatcar(xdatcar) + cls._attach_vasp_thermo( + frames, os.path.join(task_path, "OUTCAR") + ) + else: + return {} + if not frames: + return {} + + stage_names = { + "eq": f"eq_{metadata.get('start_temp', 0):g}K", + "equi": f"equi_{metadata.get('target_temp', 0):g}K", + "production": f"production_{metadata.get('target_temp', 0):g}K", + "ramp": ( + f"T_ramp_{metadata.get('start_temp', 0):g}K_" + f"{metadata.get('target_temp', 0):g}K" + ), + "decline": ( + f"T_decline_{metadata.get('target_temp', 0):g}K_" + f"{metadata.get('end_temp', 0):g}K" + ), + "final_eq": f"final_eq_{metadata.get('end_temp', 0):g}K", + } + rdf = {} + msd = {} + volume_temperature = {} + bins = int(metadata.get("rdf_bins", 100)) + cutoff = float(metadata.get("rdf_cutoff", 6.0)) + timestep_fs = float(metadata.get("timestep_fs", 1.0)) + compute_rdf = _as_bool(metadata.get("req_compute_rdf", True)) + compute_msd = _as_bool(metadata.get("req_compute_msd", True)) + for stage, stage_frames in frames.items(): + if not stage_frames: + continue + label = stage_names.get(stage, stage) + first = stage_frames[0] + natoms = len(first["frac"]) + if compute_rdf: + hist = np.zeros(bins, dtype=float) + edges = np.linspace(0.0, cutoff, bins + 1) + pair_indices = np.triu_indices(natoms, 1) + for frame in stage_frames: + structure = Structure( + Lattice(frame["cell"]), + frame["labels"], + frame["frac"], + ) + distances = structure.distance_matrix[pair_indices] + hist += np.histogram(distances, bins=edges)[0] + radius = 0.5 * (edges[:-1] + edges[1:]) + volume = float( + np.mean([abs(np.linalg.det(f["cell"])) for f in stage_frames]) + ) + shell = ( + 4.0 + * math.pi + / 3.0 + * (edges[1:] ** 3 - edges[:-1] ** 3) + ) + expected = ( + 0.5 + * natoms + * max(natoms - 1, 0) + / volume + * shell + * len(stage_frames) + ) + g_r = np.divide( + hist, + expected, + out=np.zeros_like(hist), + where=expected > 0, + ) + coordination = ( + 2.0 * np.cumsum(hist) / (natoms * len(stage_frames)) + ) + rdf[label] = { + "source": os.path.basename( + xdatcar + ), + "timestep": stage_frames[-1]["step"], + "nblocks": len(stage_frames), + "columns": {}, + "radius": radius.tolist(), + "g_r": g_r.tolist(), + "coordination": coordination.tolist(), + } + + if compute_msd: + reference = first["frac"].copy() + previous = reference.copy() + unwrapped = reference.copy() + reference_cart = np.dot(reference, first["cell"]) + msd_x, msd_y, msd_z, msd_total, timesteps = [], [], [], [], [] + for index, frame in enumerate(stage_frames): + if index: + delta = frame["frac"] - previous + delta -= np.rint(delta) + unwrapped += delta + previous = frame["frac"].copy() + displacement = np.dot(unwrapped, frame["cell"]) - reference_cart + components = np.mean(displacement ** 2, axis=0) + timesteps.append(float(frame["step"]) * timestep_fs) + msd_x.append(float(components[0])) + msd_y.append(float(components[1])) + msd_z.append(float(components[2])) + msd_total.append(float(np.sum(components))) + msd[label] = { + "source": os.path.basename( + xdatcar + ), + "timestep": timesteps, + "msd_x": msd_x, + "msd_y": msd_y, + "msd_z": msd_z, + "msd_total": msd_total, + } + + if stage in {"ramp", "decline", "production"}: + vt_stage = { + "ramp": "heating", + "decline": "cooling", + "production": "production", + }[stage] + volume_temperature[vt_stage] = { + "source": os.path.basename( + os.path.join(task_path, "OUTCAR") + ), + "timestep": [frame["step"] for frame in stage_frames], + "temperature": [ + frame.get("temperature") for frame in stage_frames + ], + "volume_per_atom": [ + frame.get( + "outcar_volume", abs(np.linalg.det(frame["cell"])) + ) + / natoms + for frame in stage_frames + ], + "total_volume": [ + frame.get( + "outcar_volume", abs(np.linalg.det(frame["cell"])) + ) + for frame in stage_frames + ], + "potential_energy": [ + frame.get("potential_energy") for frame in stage_frames + ], + "total_energy": [ + frame.get("total_energy") for frame in stage_frames + ], + "pressure": [frame.get("pressure") for frame in stage_frames], + } + return { + "rdf": rdf, + "msd": msd, + "volume_temperature": volume_temperature, + } + + @staticmethod + def _stage_sections(path): + with open(path, encoding="utf-8", errors="ignore") as fp: + return split_apex_stage_sections(fp.read()) + + @classmethod + def _parse_vasp_xdatcar(cls, path): + result = {} + for stage, text in cls._stage_sections(path).items(): + lines = text.splitlines() + frames = [] + idx = 0 + cell = None + labels = [] + natoms = 0 + while idx < len(lines): + if not lines[idx].strip(): + idx += 1 + continue + if "Direct configuration=" not in lines[idx]: + try: + scale = float(lines[idx + 1].split()[0]) + cell = np.asarray( + [ + [float(value) for value in lines[idx + offset].split()[:3]] + for offset in (2, 3, 4) + ] + ) * scale + species = lines[idx + 5].split() + counts = [int(value) for value in lines[idx + 6].split()] + labels = [ + symbol + for symbol, count in zip(species, counts) + for _ in range(count) + ] + natoms = sum(counts) + idx += 7 + except (IndexError, ValueError): + idx += 1 + continue + if idx >= len(lines) or "Direct configuration=" not in lines[idx]: + continue + match = re.search(r"=\s*(\d+)", lines[idx]) + step = int(match.group(1)) if match else len(frames) + try: + frac = np.asarray( + [ + [float(value) for value in lines[idx + offset].split()[:3]] + for offset in range(1, natoms + 1) + ] + ) + except (IndexError, ValueError): + break + frames.append( + { + "step": step, + "cell": cell.copy(), + "frac": frac, + "labels": labels, + } + ) + idx += natoms + 1 + result[stage] = frames + return result + + @classmethod + def _attach_vasp_thermo(cls, frames, outcar): + if not os.path.isfile(outcar): + return + with open(outcar, encoding="utf-8", errors="ignore") as fp: + outcar_text = fp.read() + sections = split_apex_stage_sections(outcar_text) + tail_align_stage_geometry( + frames, parse_outcar_stage_geometry(outcar_text) + ) + for stage, text in sections.items(): + stage_frames = frames.get(stage, []) + if not stage_frames: + continue + values = { + "temperature": [ + float(value) + for value in re.findall( + r"(?:temperature\s*=|\bT=)\s*([-+0-9.eE]+)", text + ) + ], + "pressure": [ + float(value) + for value in re.findall( + r"external pressure\s*=\s*([-+0-9.eE]+)", text + ) + ], + "potential_energy": [ + float(value) + for value in re.findall( + r"free\s+energy\s+TOTEN\s*=\s*([-+0-9.eE]+)", text + ) + ], + "total_energy": [ + float(value) + for value in re.findall( + r"total energy\s+ETOTAL\s*=\s*([-+0-9.eE]+)", text + ) + ], + } + for key, series in values.items(): + if not series: + continue + series = series[-len(stage_frames):] + for frame, value in zip(stage_frames[-len(series):], series): + frame[key] = value + @staticmethod def _safe_load_json(path: str): try: diff --git a/apex/core/property/Cohesive.py b/apex/core/property/Cohesive.py index 4ad93220..52d857c0 100644 --- a/apex/core/property/Cohesive.py +++ b/apex/core/property/Cohesive.py @@ -13,7 +13,7 @@ from apex.core.calculator.lib import abacus_scf from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import vasp_utils -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result from apex.core.reproduce import make_repro, post_repro from dflow.python import upload_packages @@ -214,6 +214,12 @@ def _compute_lower( # Use the last result as the single-atom reference. last_res = loadfn(all_res[-1]) + if is_failed_task_result(last_res): + raise RuntimeError( + "cohesive reference task " + f"{os.path.basename(all_tasks[-1])} failed; " + "cannot compute cohesive energies" + ) single_atom_energy = last_res["energies"][-1] / sum(last_res["atom_numbs"]) for ii, task_path in enumerate(all_tasks): @@ -221,12 +227,14 @@ def _compute_lower( latt = conf["lattice"] scale = conf["scale"] task_result = loadfn(all_res[ii]) - - total_energy = task_result["energies"][-1] - total_atoms = sum(task_result["atom_numbs"]) - - e_per_atom = total_energy / total_atoms - cohesive_energy = e_per_atom - single_atom_energy + if is_failed_task_result(task_result): + e_per_atom = float("nan") + cohesive_energy = float("nan") + else: + total_energy = task_result["energies"][-1] + total_atoms = sum(task_result["atom_numbs"]) + e_per_atom = total_energy / total_atoms + cohesive_energy = e_per_atom - single_atom_energy if getattr(self, "latt_abs", False): ptr_data += "%7.3f %8.4f %8.4f\n" % (latt, e_per_atom, cohesive_energy) diff --git a/apex/core/property/Decohesive.py b/apex/core/property/Decohesive.py index 208b9438..7bc0929a 100644 --- a/apex/core/property/Decohesive.py +++ b/apex/core/property/Decohesive.py @@ -13,7 +13,7 @@ from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import vasp_utils -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result from apex.core.reproduce import make_repro, post_repro from dflow.python import upload_packages @@ -208,18 +208,28 @@ def _compute_lower(self, output_file, all_tasks, all_res): ptr_data += "Vacuum_size(A) \tDecohesion_E(J/m^2) \tDecohesion_S(Pa)\n" first_result = loadfn(os.path.join(all_tasks[0], "result_task.json")) + if is_failed_task_result(first_result): + raise RuntimeError( + "decohesive reference task " + f"{os.path.basename(all_tasks[0])} failed; " + "cannot compute decohesion energies" + ) equi_evac = first_result["energies"][-1] pre_evac = 0.0 CF_EV_TO_J_PER_M2 = 1.60217657e-16 / 1e-20 * 0.001 for task_dir in all_tasks: - task_result = loadfn(os.path.join(task_dir, "result_task.json")) - area = np.linalg.norm( - np.cross(task_result["cells"][0][0], task_result["cells"][0][1]) - ) - evac = (task_result["energies"][-1] - equi_evac) / area * CF_EV_TO_J_PER_M2 vacuum_size = loadfn(os.path.join(task_dir, "decohesive.json"))["vacuum_size"] - stress = (evac - pre_evac) / vacuum_size_step * 1e10 + task_result = loadfn(os.path.join(task_dir, "result_task.json")) + if is_failed_task_result(task_result): + evac = float("nan") + stress = float("nan") + else: + area = np.linalg.norm( + np.cross(task_result["cells"][0][0], task_result["cells"][0][1]) + ) + evac = (task_result["energies"][-1] - equi_evac) / area * CF_EV_TO_J_PER_M2 + stress = (evac - pre_evac) / vacuum_size_step * 1e10 ptr_data += f"{vacuum_size:7.3f} {evac:7.3f} {stress:10.3e} \n" res_data[f"{vacuum_size}_{os.path.basename(task_dir)}"] = [ diff --git a/apex/core/property/EOS.py b/apex/core/property/EOS.py index 9ba72236..a73a8359 100644 --- a/apex/core/property/EOS.py +++ b/apex/core/property/EOS.py @@ -9,7 +9,7 @@ from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import vasp_utils from apex.core.calculator.lib import abacus_scf -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result from apex.core.refine import make_refine from apex.core.reproduce import make_repro, post_repro from dflow.python import upload_packages @@ -225,13 +225,12 @@ def _compute_lower(self, output_file, all_tasks, all_res): # vol = self.vol_start + ii * self.vol_step vol = loadfn(os.path.join(all_tasks[ii], "eos.json"))["volume"] task_result = loadfn(all_res[ii]) - res_data[vol] = task_result["energies"][-1] / sum( - task_result["atom_numbs"] - ) - ptr_data += "%7.3f %8.4f \n" % ( - vol, - task_result["energies"][-1] / sum(task_result["atom_numbs"]), - ) + if is_failed_task_result(task_result): + epa = float("nan") + else: + epa = task_result["energies"][-1] / sum(task_result["atom_numbs"]) + res_data[vol] = epa + ptr_data += "%7.3f %8.4f \n" % (vol, epa) else: if "init_data_path" not in self.parameter: diff --git a/apex/core/property/FiniteTlatt.py b/apex/core/property/FiniteTlatt.py index 04f2fc48..c6d1233d 100644 --- a/apex/core/property/FiniteTlatt.py +++ b/apex/core/property/FiniteTlatt.py @@ -1,4 +1,4 @@ -"""Lattice parameter vs temperature (LAMMPS npt + averaging). Only LAMMPS supported.""" +"""Lattice parameter versus temperature from NpT molecular dynamics.""" import json import logging @@ -7,7 +7,10 @@ from shutil import copyfile from typing import Dict, List, Tuple +import numpy as np from monty.serialization import dumpfn +from pymatgen.core.lattice import Lattice +from pymatgen.core.structure import Structure from apex.core.property.Property import Property from apex.core.refine import make_refine @@ -17,7 +20,7 @@ upload_packages.append(__file__) DEFAULT_SUPERCELL = [2, 2, 2] -DEFAULT_CAL_SETTING: Dict[str, int | List[int]] = { +DEFAULT_CAL_SETTING: Dict[str, float | int | List[int]] = { "temperature": [200, 400, 600, 800], "equi_step": 80000, "N_every": 100, @@ -31,20 +34,32 @@ class FiniteTlatt(Property): """ - Generate LAMMPS tasks to measure lattice parameters at finite temperatures - using NPT runs plus time-averaging. + Generate LAMMPS or VASP tasks to measure finite-temperature lattice + parameters using NpT runs plus production-trajectory statistics. """ def __init__(self, parameter: Dict, inter_param: Dict | None = None): + self.inter_param = inter_param or {"type": "lammps"} + if self.inter_param.get("type") == "abacus": + raise NotImplementedError( + "finite_t_latt does not support the ABACUS backend; " + "use LAMMPS or VASP" + ) parameter["reproduce"] = parameter.get("reproduce", False) self.reprod = parameter["reproduce"] - # Enforce LAMMPS-only workflow - if inter_param is not None and inter_param.get("type") in ["vasp", "abacus"]: - raise TypeError("FiniteTlatt supports only LAMMPS calculations.") - parameter.setdefault("cal_setting", {}) - for key, val in DEFAULT_CAL_SETTING.items(): + default_cal_setting = dict(DEFAULT_CAL_SETTING) + if self.inter_param["type"] == "vasp": + default_cal_setting.update( + { + "temperature": [300, 500, 700, 900, 1100, 1300, 1500], + "equi_step": 5000, + "ave_step": 10000, + "timestep_fs": 1.0, + } + ) + for key, val in default_cal_setting.items(): parameter["cal_setting"].setdefault(key, val) if not self.reprod and not ( @@ -57,11 +72,12 @@ def __init__(self, parameter: Dict, inter_param: Dict | None = None): self.init_from_suffix = parameter["init_from_suffix"] self.supercell_size = parameter.get("supercell_size", DEFAULT_SUPERCELL) - parameter["cal_type"] = "npt+ave/time" + # MD calculators dispatch on the property type. "static" remains a + # valid fallback for calculators that inspect cal_type first and, + # unlike "relaxation", does not require relaxation-only settings. + parameter["cal_type"] = "static" self.cal_setting = parameter["cal_setting"] self.parameter = parameter - # only supports LAMMPS now - self.inter_param = inter_param or {"type": "lammps"} def make_confs(self, path_to_work: str, path_to_equi: str, refine: bool = False): path_to_work = os.path.abspath(path_to_work) @@ -105,11 +121,20 @@ def _compute_lower(self, output_file, all_tasks, all_res): ) else: ptr_data += " Temperature(K) a(A) b(A) c(A)\n" + statistics = {} for idx, task_dir in enumerate(all_tasks): temp = self.cal_setting["temperature"][idx] - a, b, c = self._average_box(task_dir, self.supercell_size) + stats = self._cell_statistics(task_dir, self.supercell_size) + a, b, c = stats["lengths"]["mean"] ptr_data += f"{temp:>10}: {a:7.6f} {b:7.6f} {c:7.6f}\n" res_data[str(temp)] = [a, b, c, temp] + statistics[str(temp)] = stats + # Preserve the historic temperature -> [a, b, c, T] mapping. + # Rich tensor/statistical data lives under a reserved companion key. + res_data["_metadata"] = { + "schema": "apex.finite_t_latt.statistics/v1", + "temperatures": statistics, + } with open(output_file, "w") as fp: json.dump(res_data, fp, indent=4) @@ -149,64 +174,175 @@ def _make_refine(self, path_to_work: str) -> List[str]: return task_list def _make_fresh_tasks(self, path_to_work: str, path_to_equi: str) -> List[str]: - if self.inter_param["type"] in ["vasp", "abacus"]: - raise TypeError("FiniteTlatt only supports LAMMPS calculation") - - equi_contcar = os.path.join(path_to_equi, "CONTCAR") - if not os.path.exists(equi_contcar): + equi_structure = os.path.join(path_to_equi, "CONTCAR") + if not os.path.exists(equi_structure): raise RuntimeError("please do relaxation first") task_list: List[str] = [] for idx, temp in enumerate(self.cal_setting["temperature"]): task_dir = os.path.join(path_to_work, f"task.{idx:06d}") os.makedirs(task_dir, exist_ok=True) - self._write_task(task_dir, equi_contcar, temp) + self._write_task(task_dir, equi_structure, temp) task_list.append(task_dir) return task_list def _symlink_variable(self, init_task: str, out_task: str): os.makedirs(out_task, exist_ok=True) - dst = os.path.join(out_task, "variable_FiniteTlatt.json") - if os.path.exists(dst): + if self.inter_param["type"] == "vasp": + return + src = os.path.join(init_task, "variable_FiniteTlatt.in") + dst = os.path.join(out_task, "variable_FiniteTlatt.in") + if os.path.lexists(dst): os.remove(dst) - os.symlink( - os.path.relpath(os.path.join(init_task, "variable_FiniteTlatt.json"), out_task), - dst, - ) + os.symlink(os.path.relpath(src, out_task), dst) - def _write_task(self, task_dir: str, equi_contcar: str, temp: float): + def _write_task(self, task_dir: str, equi_structure: str, temp: float): os.chdir(task_dir) for fname in ["INCAR", "POTCAR", "POSCAR", "conf.lmp", "in.lammps", "STRU"]: if os.path.exists(fname): os.remove(fname) - copyfile(equi_contcar, "POSCAR") + if self.inter_param["type"] == "vasp": + structure = Structure.from_file(equi_structure) + structure.make_supercell(self.supercell_size) + structure.to(filename="POSCAR") + else: + copyfile(equi_structure, "POSCAR") FiniteTlatt_task = {"temperature": temp, "supercell_size": self.supercell_size} dumpfn(FiniteTlatt_task, "FiniteTlatt.json", indent=4) - with open("variable_FiniteTlatt.in", "w") as fp: - fp.write(self._variable(temp)) + if self.inter_param["type"] != "vasp": + with open("variable_FiniteTlatt.in", "w") as fp: + fp.write(self._variable(temp)) def _average_box(self, task_dir: str, supercell_size: List[int]) -> Tuple[float, float, float]: - a_sum = b_sum = c_sum = count = 0 - box_file = os.path.join(task_dir, "average_box.txt") - with open(box_file, "r") as fh: - for line in fh: - if line.startswith("#") or not line.strip(): - continue - parts = line.split() - if len(parts) == 4: - _, v_lx, v_ly, v_lz = parts - a_sum += float(v_lx) - b_sum += float(v_ly) - c_sum += float(v_lz) - count += 1 + stats = self._cell_statistics(task_dir, supercell_size) + return tuple(stats["lengths"]["mean"]) + + def _cell_statistics(self, task_dir: str, supercell_size: List[int]) -> Dict: + if self.inter_param["type"] == "vasp": + outcar = os.path.join(task_dir, "OUTCAR") + cells = self._vasp_cells(outcar) + ionic_steps = self._vasp_ionic_steps(outcar) + # VASP prints the initial cell before the first ionic step. Keep + # only cell samples aligned with production ionic frames. + if ionic_steps and len(cells) > ionic_steps: + cells = cells[-ionic_steps:] + else: + cells = [] + box_file = os.path.join(task_dir, "average_box.txt") + with open(box_file, "r") as fh: + for line in fh: + if line.startswith("#") or not line.strip(): + continue + parts = line.split() + if len(parts) == 4: + cells.append(np.diag([float(value) for value in parts[1:]])) + return self._summarize_cells(cells, supercell_size) + + @staticmethod + def _series_statistics(values): + array = np.asarray(values, dtype=float) + count = int(array.shape[0]) if count == 0: + shape = list(array.shape[1:]) + zero = np.zeros(shape, dtype=float).tolist() + return { + "mean": zero, + "std": zero, + "block_standard_error": zero, + "sample_count": 0, + "block_count": 0, + } + mean = np.mean(array, axis=0) + std = np.std(array, axis=0, ddof=1) if count > 1 else np.zeros_like(mean) + block_size = max(1, int(np.sqrt(count))) + block_count = count // block_size + if block_count > 1: + trimmed = array[: block_count * block_size] + block_means = trimmed.reshape( + (block_count, block_size) + array.shape[1:] + ).mean(axis=1) + block_error = np.std(block_means, axis=0, ddof=1) / np.sqrt(block_count) + else: + block_error = np.zeros_like(mean) + return { + "mean": np.asarray(mean).tolist(), + "std": np.asarray(std).tolist(), + "block_standard_error": np.asarray(block_error).tolist(), + "sample_count": count, + "block_count": block_count, + } + + @classmethod + def _summarize_cells(cls, cells, supercell_size): + if not cells: + empty = cls._series_statistics(np.empty((0, 3))) + return { + "cell": cls._series_statistics(np.empty((0, 3, 3))), + "lengths": empty, + "angles": empty, + "volume": cls._series_statistics(np.empty((0,))), + "sample_count": 0, + } + scale = np.asarray(supercell_size, dtype=float)[:, None] + normalized = np.asarray(cells, dtype=float) / scale + lengths = np.linalg.norm(normalized, axis=2) + angles = np.asarray( + [Lattice(cell).angles for cell in normalized], dtype=float + ) + volumes = np.abs(np.linalg.det(normalized)) + return { + "cell": cls._series_statistics(normalized), + "lengths": cls._series_statistics(lengths), + "angles": cls._series_statistics(angles), + "volume": cls._series_statistics(volumes), + "sample_count": int(len(normalized)), + } + + @staticmethod + def _mean_cell_lengths(cells, supercell_size): + if not cells: return 0.0, 0.0, 0.0 - a = a_sum / count / supercell_size[0] - b = b_sum / count / supercell_size[1] - c = c_sum / count / supercell_size[2] - return a, b, c + lengths = np.asarray( + [[np.linalg.norm(vector) for vector in cell] for cell in cells], + dtype=float, + ) + return tuple( + float(np.mean(lengths[:, axis]) / supercell_size[axis]) + for axis in range(3) + ) + + @staticmethod + def _vasp_cells(outcar): + cells = [] + with open(outcar, encoding="utf-8", errors="ignore") as fp: + lines = fp.readlines() + for idx, line in enumerate(lines): + if "direct lattice vectors" not in line.lower(): + continue + try: + cell = [ + [float(value) for value in lines[idx + offset].split()[:3]] + for offset in (1, 2, 3) + ] + except (IndexError, ValueError): + continue + cells.append(cell) + return cells + + @staticmethod + def _vasp_ionic_steps(outcar): + count = 0 + with open(outcar, encoding="utf-8", errors="ignore") as fp: + for line in fp: + if re.match( + r"^\s*POSITION\s+TOTAL-FORCE(?:\s|$)", + line, + flags=re.IGNORECASE, + ): + count += 1 + return count def _variable(self, temp: float) -> str: return ( diff --git a/apex/core/property/Gamma.py b/apex/core/property/Gamma.py index 28b941b8..30c43619 100644 --- a/apex/core/property/Gamma.py +++ b/apex/core/property/Gamma.py @@ -1,4 +1,5 @@ import glob +import hashlib import json import os import re @@ -8,16 +9,31 @@ import numpy as np from monty.serialization import dumpfn, loadfn from pymatgen.core.structure import Structure -from pymatgen.core.surface import SlabGenerator -from pymatgen.analysis.diffraction.tem import TEMCalculator from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import vasp_utils -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result +from apex.core.property.gamma_slab import get_first_gamma_slab +from apex.core.property.gamma_slab import make_gamma_slab_generator +from apex.core.property.gamma_slab import validate_gamma_slab_settings +from apex.core.property.gamma_slab import validate_generated_gamma_slab from apex.core.refine import make_refine from apex.core.reproduce import make_repro, post_repro -from apex.core.structure import StructureInfo +from apex.core.structure import ( + StructureInfo, + normalize_parent_lattice_hint, + resolve_parent_lattice_hint, +) from apex.core.lib.slab_orientation import SlabSlipSystem +from apex.core.lib.parent_lattice_mapping import ( + resolve_parent_slip_geometry, + resolve_parent_supercell, +) +from apex.core.property.gamma_geometry import ( + build_parent_gamma_slab, + validate_gamma_cell_geometry, + validate_vacuum_size, +) from apex.core.lib.trans_tools import trans_mat_basis from apex.core.lib.trans_tools import (plane_miller_bravais_to_miller, direction_miller_bravais_to_miller) @@ -26,12 +42,84 @@ upload_packages.append(__file__) +def _sha256_file(path): + digest = hashlib.sha256() + with open(path, "rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _validate_displacement_points(values): + if values is None: + return None + if not isinstance(values, (list, tuple)): + raise ValueError("gamma displacement_points must be a list") + points = [float(value) for value in values] + if ( + not points + or any(not np.isfinite(value) or value < 0.0 or value > 1.0 for value in points) + or len(set(points)) != len(points) + or 0.0 not in points + ): + raise ValueError( + "gamma displacement_points must include 0 and contain unique values in [0, 1]" + ) + return sorted(points) + + +def _validate_n_steps(value): + if ( + isinstance(value, (bool, np.bool_)) + or not isinstance(value, (int, np.integer)) + or value <= 0 + ): + raise ValueError("gamma n_steps must be a positive integer") + return int(value) + + +def _resolve_require_orthogonal_cell(parameter, default=False): + explicit = parameter.get("require_orthogonal_cell", None) + legacy_alias = parameter.get("orthogonalize_cell", None) + for key, candidate in ( + ("require_orthogonal_cell", explicit), + ("orthogonalize_cell", legacy_alias), + ): + if candidate is not None and not isinstance(candidate, (bool, np.bool_)): + raise ValueError(f"gamma {key} must be a boolean") + if explicit is not None and legacy_alias is not None and explicit != legacy_alias: + raise ValueError( + "gamma require_orthogonal_cell and orthogonalize_cell disagree" + ) + value = explicit if explicit is not None else legacy_alias + if value is None: + value = default + if legacy_alias: + logging.info( + "gamma orthogonalize_cell=true is a strict geometry gate; APEX " + "does not Gram-Schmidt periodic cells" + ) + parameter["require_orthogonal_cell"] = bool(value) + return bool(value) + + class Gamma(Property): """ Calculation of gamma lines """ def __init__(self, parameter, inter_param=None): + self.parent_lattice = normalize_parent_lattice_hint( + parameter.get("parent_lattice") + ) + if self.parent_lattice is not None: + parameter["parent_lattice"] = self.parent_lattice + self.parent_max_site_distance = float( + parameter.get("parent_max_site_distance", 0.65) + ) + if not np.isfinite(self.parent_max_site_distance) or self.parent_max_site_distance <= 0: + raise ValueError("gamma parent_max_site_distance must be finite and positive") + parameter["parent_max_site_distance"] = self.parent_max_site_distance parameter["reproduce"] = parameter.get("reproduce", False) self.reprod = parameter["reproduce"] if not self.reprod: @@ -47,8 +135,22 @@ def __init__(self, parameter, inter_param=None): self.plane_shift = parameter["plane_shift"] parameter["supercell_size"] = parameter.get("supercell_size", (1, 1, 5)) self.supercell_size = parameter["supercell_size"] - parameter["vacuum_size"] = parameter.get("vacuum_size", 0) - self.vacuum_size = parameter["vacuum_size"] + parameter["min_slab_height"] = parameter.get( + "min_slab_height", None + ) + self.min_slab_height = parameter["min_slab_height"] + parameter["max_atoms"] = parameter.get("max_atoms", None) + self.max_atoms = parameter["max_atoms"] + parameter["min_distance"] = parameter.get("min_distance", 0.2) + self.min_distance = parameter["min_distance"] + parameter["vacuum_size"] = parameter.get("vacuum_size", 20) + self.vacuum_size = validate_vacuum_size( + parameter["vacuum_size"], "Gamma" + ) + parameter["vacuum_size"] = self.vacuum_size + self.require_orthogonal_cell = _resolve_require_orthogonal_cell( + parameter + ) if parameter["cal_type"] == "static" and "add_fix" not in parameter: parameter["add_fix"] = None else: @@ -57,7 +159,15 @@ def __init__(self, parameter, inter_param=None): ) # standard method self.add_fix = parameter["add_fix"] parameter["n_steps"] = parameter.get("n_steps", 10) - self.n_steps = parameter["n_steps"] + self.n_steps = _validate_n_steps(parameter["n_steps"]) + parameter["n_steps"] = self.n_steps + parameter["displacement_points"] = parameter.get( + "displacement_points", None + ) + self.displacement_points = _validate_displacement_points( + parameter["displacement_points"] + ) + parameter["displacement_points"] = self.displacement_points self.atom_num = None else: parameter["cal_type"] = parameter.get("cal_type", "relaxation") @@ -197,8 +307,26 @@ def make_confs(self, path_to_work, path_to_equi, refine=False): ss = Structure.from_file("CONTCAR.direct") # get structure type st = StructureInfo(ss) - self.structure_type = st.lattice_structure - self.conv_std_structure = st.conventional_structure + self.detected_structure_type = st.lattice_structure + self.structure_type, self.structure_type_source = ( + resolve_parent_lattice_hint( + self.detected_structure_type, self.parent_lattice + ) + ) + if self.structure_type_source == "user_override": + logging.info( + "Gamma parent_lattice=%s overrides automatic structure " + "classification %s without changing the input geometry.", + self.structure_type, + self.detected_structure_type, + ) + # A parent_lattice hint changes the crystallographic index + # basis, not merely the parameter block. Preserve the exact + # relaxed RSS row basis so the inferred integer supercell + # mapping remains traceable from POSCAR to CONTCAR. + self.conv_std_structure = ( + ss if self.parent_lattice is not None else st.conventional_structure + ) relax_a = self.conv_std_structure.lattice.a relax_b = self.conv_std_structure.lattice.b relax_c = self.conv_std_structure.lattice.c @@ -210,9 +338,29 @@ def make_confs(self, path_to_work, path_to_equi, refine=False): self.slip_length = type_param.get("slip_length", self.slip_length) self.plane_shift = type_param.get("plane_shift", self.plane_shift) self.supercell_size = type_param.get("supercell_size", self.supercell_size) + self.min_slab_height = type_param.get( + "min_slab_height", self.min_slab_height + ) + self.max_atoms = type_param.get("max_atoms", self.max_atoms) + self.min_distance = type_param.get( + "min_distance", self.min_distance + ) self.vacuum_size = type_param.get("vacuum_size", self.vacuum_size) + self.require_orthogonal_cell = _resolve_require_orthogonal_cell( + type_param, self.require_orthogonal_cell + ) self.add_fix = type_param.get("add_fix", self.add_fix) self.n_steps = type_param.get("n_steps", self.n_steps) + self.displacement_points = type_param.get( + "displacement_points", self.displacement_points + ) + self.displacement_points = _validate_displacement_points( + self.displacement_points + ) + self.vacuum_size = validate_vacuum_size( + self.vacuum_size, "Gamma" + ) + self.n_steps = _validate_n_steps(self.n_steps) if not (self.plane_miller and self.slip_direction): raise RuntimeError(f'fail to obtain both slip plane ' f'and slip direction info from input json file!') @@ -223,11 +371,124 @@ def make_confs(self, path_to_work, path_to_equi, refine=False): f'is not fully supported so far.\n' f'You may need to double check the generated slab structures' ) - # gen initial slab - (plane_miller, slip_direction, - slip_length, Q) = self.__convert_input_miller(self.conv_std_structure) - slab = self.__gen_slab_pmg(self.conv_std_structure, plane_miller, trans_matrix=Q) + ( + self.supercell_size, + self.min_slab_height, + self.max_atoms, + self.min_distance, + ) = validate_gamma_slab_settings( + self.supercell_size, + self.min_slab_height, + self.max_atoms, + self.min_distance, + ) + # Generate the initial slab. Disordered parent-mapped cells use + # the user-facing parent indices and an automatically inferred + # parent-supercell mapping. Legacy perfect-crystal inputs retain + # the established path. + self.parent_mapping = None + self.parent_slip_geometry = None + self.gamma_geometry = None + if self.parent_lattice is not None: + self.parent_mapping = resolve_parent_supercell( + self.conv_std_structure, + self.parent_lattice, + max_site_distance=self.parent_max_site_distance, + ) + self.parent_slip_geometry = resolve_parent_slip_geometry( + self.conv_std_structure, + self.parent_mapping, + self.plane_miller, + self.slip_direction, + self.slip_length, + ) + built = build_parent_gamma_slab( + self.conv_std_structure, + self.parent_slip_geometry, + self.supercell_size, + self.supercell_size[2], + self.min_slab_height, + self.vacuum_size, + self.plane_shift, + self.require_orthogonal_cell, + ) + slab = built.slab + self._gamma_upper_indices = built.upper_indices + self._gamma_lower_indices = built.lower_indices + self._slab_generation_metadata = built.generation_metadata + self.gamma_geometry = { + "parent_mapping": self.parent_mapping.as_dict(), + "slip_geometry": self.parent_slip_geometry.as_dict(), + "slab_geometry": built.metadata, + "source_structure": { + "path": os.path.abspath(equi_contcar), + "sha256": _sha256_file(equi_contcar), + "atom_count": len(self.conv_std_structure), + }, + } + self.slab_generation = validate_generated_gamma_slab( + slab, + built.generation_metadata, + self.supercell_size[:2], + self.max_atoms, + self.min_distance, + "Gamma", + ) + self.slab_generation.update(built.metadata) + slip_length = float( + np.linalg.norm( + self.parent_slip_geometry.burgers_vector_cart + ) + ) + else: + (plane_miller, slip_direction, + slip_length, Q) = self.__convert_input_miller(self.conv_std_structure) + slab = self.__gen_slab_pmg( + self.conv_std_structure, plane_miller, trans_matrix=Q + ) + cell_geometry = validate_gamma_cell_geometry( + slab, + require_orthogonal=self.require_orthogonal_cell, + property_name="Gamma", + ) + self.slab_generation.update( + { + "interface_count": 1 if self.vacuum_size > 0 else 2, + "cell_geometry": cell_geometry, + "require_orthogonal_cell": self.require_orthogonal_cell, + } + ) + self.gamma_geometry = { + "parent_mapping": None, + "slip_geometry": None, + "slab_geometry": { + "interface_count": self.slab_generation["interface_count"], + "added_vacuum_angstrom": self.vacuum_size, + "cell_geometry": cell_geometry, + "require_orthogonal_cell": self.require_orthogonal_cell, + }, + "source_structure": { + "path": os.path.abspath(equi_contcar), + "sha256": _sha256_file(equi_contcar), + "atom_count": len(self.conv_std_structure), + }, + } self.atom_num = len(slab.sites) + self.slab_generation.update( + { + "detected_structure_type": self.detected_structure_type, + "effective_parent_lattice": self.structure_type, + "structure_type_source": self.structure_type_source, + } + ) + dumpfn( + self.slab_generation, + os.path.join(path_to_work, "slab_generation.json"), + ) + dumpfn( + self.gamma_geometry, + os.path.join(path_to_work, "gamma_geometry.json"), + ) os.chdir(path_to_work) if os.path.exists(POSCAR): @@ -236,7 +497,12 @@ def make_confs(self, path_to_work, path_to_equi, refine=False): # task_poscar = os.path.join(output, 'POSCAR') count = 0 # define slip vector - if type(slip_length) == int or type(slip_length) == float: + if self.parent_slip_geometry is not None: + frac_slip_vec = ( + self.parent_slip_geometry.local_frame + @ self.parent_slip_geometry.burgers_vector_cart + ) + elif type(slip_length) == int or type(slip_length) == float: frac_slip_vec = np.array([slip_length, 0, 0]) * relax_a else: # for Sequence[int|float, int|float, int|float] type @@ -252,9 +518,17 @@ def make_confs(self, path_to_work, path_to_equi, refine=False): ) self.slip_length = frac_slip_vec[0] # get displaced structure - for obtained_slab in self.__displace_slab_generator(slab, - disp_vector=frac_slip_vec, - is_frac=False): + for frac, obtained_slab in self.__displace_slab_generator( + slab, disp_vector=frac_slip_vec, is_frac=False + ): + validate_generated_gamma_slab( + obtained_slab, + self._slab_generation_metadata, + self.supercell_size[:2], + self.max_atoms, + self.min_distance, + "Gamma task", + ) output_task = os.path.join(path_to_work, "task.%06d" % count) os.makedirs(output_task, exist_ok=True) os.chdir(output_task) @@ -277,6 +551,9 @@ def make_confs(self, path_to_work, path_to_equi, refine=False): # record miller dumpfn(self.plane_miller, "miller.json") dumpfn(self.slip_length, 'slip_length.json') + dumpfn(frac, "normalized_displacement.json") + if self.gamma_geometry is not None: + dumpfn(self.gamma_geometry, "gamma_geometry.json") count += 1 os.chdir(cwd) @@ -291,8 +568,10 @@ def __convert_input_miller(self, structure: Structure): slip_str = ''.join([str(i) for i in slip_direction]) combined_key = 'x'.join([plane_str, slip_str]) l2_normalize_1d = lambda v: v / np.linalg.norm(v, 2) - # try to get default slip system from pre-defined dict - dir_dict = SlabSlipSystem.atomic_system_dict() + # Use only physically recommended slip systems for crystallographic + # defaults. Other orthogonal systems remain available through the + # geometry-only fallback below. + dir_dict = SlabSlipSystem.recommended_system_dict() try: system = dir_dict[self.structure_type] (plane_miller, x_miller, @@ -300,10 +579,11 @@ def __convert_input_miller(self, structure: Structure): except KeyError: logging.warning( 'Warning:\n' - 'The input slip system is not pre-defined in the Gamma module!\n' - 'We highly recommend you to double check the slab structure generated' - 'of an undefined slip system, as it may not be what you expected, ' - 'especially for a HCP structure.' + 'The input slip system is not one of the physically recommended ' + 'FCC/BCC/HCP systems in README section 4.10.\n' + 'Gamma is falling back to a geometric construction and will only ' + 'check that slip_direction lies on plane_miller. Double-check the ' + 'generated slab, especially for HCP or structure type "other".' ) x_miller = slip_direction if not slip_length: @@ -334,32 +614,25 @@ def __convert_input_miller(self, structure: Structure): z_cartesian_unit_vector = l2_normalize_1d(np.cross(x_cartesian, xy_cartesian)) y_cartesian_unit_vector = l2_normalize_1d(np.cross(z_cartesian_unit_vector, x_cartesian_unit_vector)) - finally: - reoriented_basis = np.array([x_cartesian_unit_vector, - y_cartesian_unit_vector, - z_cartesian_unit_vector]) - # Transform the lattice vectors of the slab - Q = trans_mat_basis(reoriented_basis) + reoriented_basis = np.array([x_cartesian_unit_vector, + y_cartesian_unit_vector, + z_cartesian_unit_vector]) + # Transform the lattice vectors of the slab + Q = trans_mat_basis(reoriented_basis) return plane_miller, x_miller, slip_length, Q def __gen_slab_pmg(self, structure: Structure, plane_miller, trans_matrix=None) -> Structure: - # Get slab inter-plane distance - tem_calc_obj = TEMCalculator() - spacing_dict = tem_calc_obj.get_interplanar_spacings(self.conv_std_structure, - [plane_miller]) - slab_size = spacing_dict[plane_miller] * self.supercell_size[2] - # Generate slab via Pymatgen - slabGen = SlabGenerator(structure, miller_index=plane_miller, - min_slab_size=slab_size, min_vacuum_size=0, - center_slab=True, in_unit_planes=False, - lll_reduce=True, reorient_lattice=False, - primitive=False) - slabs_pmg = slabGen.get_slabs(ftol=0.001) - slab = [s for s in slabs_pmg if s.miller_index == plane_miller][0] + slabGen, generation_metadata = make_gamma_slab_generator( + structure, + plane_miller, + self.supercell_size[2], + self.min_slab_height, + ) + slab = get_first_gamma_slab(slabGen, ftol=0.001) # If a transform matrix is passed, reorient the slab - if trans_matrix.any(): + if trans_matrix is not None and np.asarray(trans_matrix).any(): reoriented_lattice_vectors = [trans_matrix.dot(v) for v in slab.lattice.matrix] slab = Structure(lattice=np.matrix(reoriented_lattice_vectors), coords=slab.frac_coords, species=slab.species) @@ -400,25 +673,49 @@ def __gen_slab_pmg(self, structure: Structure, slab.make_supercell(scaling_matrix=[self.supercell_size[0], self.supercell_size[1], 1]) + self._slab_generation_metadata = generation_metadata + self.slab_generation = validate_generated_gamma_slab( + slab, + generation_metadata, + self.supercell_size[:2], + self.max_atoms, + self.min_distance, + "Gamma", + ) + self.slab_generation.update( + { + "min_slab_height": self.min_slab_height, + "max_atoms": self.max_atoms, + "min_distance_threshold": self.min_distance, + "vacuum_size": self.vacuum_size, + } + ) + logging.info("Gamma slab generation: %s", self.slab_generation) return slab def __displace_slab_generator(self, slab: Structure, disp_vector, is_frac=True, - to_unit_cell=True) -> Structure: - # generator of displaced slab structures - yield slab.copy() + to_unit_cell=True): + """Yield normalized-displacement and slab pairs.""" # return list of atoms number to be displaced which above 0.5 z - disp_atoms_list = np.where(slab.frac_coords[:, 2] > 0.5)[0] - for _ in list(range(self.n_steps)): - frac_disp = 1 / self.n_steps - unit_vector = frac_disp * np.array(disp_vector) - slab.translate_sites( + disp_atoms_list = getattr( + self, + "_gamma_upper_indices", + np.where(slab.frac_coords[:, 2] > 0.5)[0], + ) + if self.displacement_points is not None: + fractions = self.displacement_points + else: + fractions = [step / self.n_steps for step in range(self.n_steps + 1)] + for frac in fractions: + displaced = slab.copy() + displaced.translate_sites( indices=disp_atoms_list, - vector=unit_vector, + vector=frac * np.array(disp_vector), frac_coords=is_frac, to_unit_cell=to_unit_cell, ) - yield slab.copy() + yield frac, displaced def __poscar_fix(self, poscar) -> None: # add position fix condition of x and y in POSCAR @@ -537,7 +834,23 @@ def _compute_lower(self, output_file, all_tasks, all_res): all_tasks.sort() n_steps = len(all_tasks) - 1 task_result_slab_equi = loadfn(os.path.join(all_tasks[0], "result_task.json")) + if is_failed_task_result(task_result_slab_equi): + raise RuntimeError( + "gamma reference task " + f"{os.path.basename(all_tasks[0])} failed; cannot compute SFE" + ) slip_length = loadfn(os.path.join(all_tasks[0], "slip_length.json")) + geometry_file = os.path.join(all_tasks[0], "gamma_geometry.json") + interface_count = 1 if self.vacuum_size > 0 else 2 + if os.path.isfile(geometry_file): + geometry = loadfn(geometry_file) + interface_count = int( + geometry.get("slab_geometry", {}).get("interface_count", 1) + ) + if interface_count not in (1, 2): + raise RuntimeError( + f"Invalid Gamma interface_count={interface_count}" + ) equi_path = os.path.abspath( os.path.join( os.path.dirname(output_file), "../relaxation/relax_task" @@ -547,27 +860,43 @@ def _compute_lower(self, output_file, all_tasks, all_res): equi_epa = equi_result["energies"][-1] / np.sum( equi_result["atom_numbs"] ) + ref_energy = task_result_slab_equi["energies"][-1] + ref_natoms = np.sum(task_result_slab_equi["atom_numbs"]) + equi_epa_slab = ref_energy / ref_natoms + reference_cell = np.asarray(task_result_slab_equi["cells"][0], dtype=float) + reference_area = float( + np.linalg.norm(np.cross(reference_cell[0], reference_cell[1])) + ) + if not np.isfinite(reference_area) or reference_area <= 0.0: + raise RuntimeError("Gamma reference task has an invalid in-plane area") for ii in all_tasks: - task_result = loadfn(os.path.join(ii, "result_task.json")) - natoms = np.sum(task_result["atom_numbs"]) - epa = task_result["energies"][-1] / natoms - equi_epa_slab = task_result_slab_equi["energies"][-1] / natoms - AA = np.linalg.norm( - np.cross(task_result["cells"][0][0], task_result["cells"][0][1]) - ) - structure_dir = os.path.basename(ii) - Cf = 1.60217657e-16 / 1e-20 * 0.001 - sfe = ( - ( - task_result["energies"][-1] - - task_result_slab_equi["energies"][-1] + frac_file = os.path.join(ii, "normalized_displacement.json") + frac = loadfn(frac_file) if os.path.isfile(frac_file) else int(ii[-4:]) / n_steps + miller_index = loadfn(os.path.join(ii, "miller.json")) + task_result = loadfn(os.path.join(ii, "result_task.json")) + if is_failed_task_result(task_result): + sfe = float("nan") + epa = float("nan") + else: + natoms = np.sum(task_result["atom_numbs"]) + epa = task_result["energies"][-1] / natoms + AA = np.linalg.norm( + np.cross(task_result["cells"][0][0], task_result["cells"][0][1]) + ) + if not np.isclose( + AA, reference_area, rtol=1.0e-8, atol=1.0e-8 + ): + raise RuntimeError( + "Gamma task in-plane area changed relative to the " + f"u=0 reference: {AA:.12g} vs {reference_area:.12g} A^2" ) - / AA + Cf = 1.60217657e-16 / 1e-20 * 0.001 + sfe = ( + (task_result["energies"][-1] - ref_energy) + / (AA * interface_count) * Cf - ) - frac = int(ii[-4:]) / n_steps - miller_index = loadfn(os.path.join(ii, "miller.json")) + ) ptr_data += "%-25s %7.2f %7.3f %7.3f %8.3f %8.3f\n" % ( str(miller_index) + "-" + structure_dir + ":", frac, diff --git a/apex/core/property/GammaSurface.py b/apex/core/property/GammaSurface.py index c4a3868e..c2d55466 100644 --- a/apex/core/property/GammaSurface.py +++ b/apex/core/property/GammaSurface.py @@ -1,4 +1,5 @@ import glob +import hashlib import json import logging import os @@ -7,30 +8,83 @@ import dpdata import numpy as np from monty.serialization import dumpfn, loadfn -from pymatgen.analysis.diffraction.tem import TEMCalculator from pymatgen.core.structure import Structure -from pymatgen.core.surface import SlabGenerator from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import vasp_utils from apex.core.lib.slab_orientation import SlabSlipSystem +from apex.core.lib.parent_lattice_mapping import ( + resolve_parent_slip_geometry, + resolve_parent_supercell, +) from apex.core.lib.trans_tools import direction_miller_bravais_to_miller from apex.core.lib.trans_tools import plane_miller_bravais_to_miller from apex.core.lib.trans_tools import trans_mat_basis -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result +from apex.core.property.gamma_slab import get_first_gamma_slab +from apex.core.property.gamma_slab import make_gamma_slab_generator +from apex.core.property.gamma_slab import validate_gamma_slab_settings +from apex.core.property.gamma_slab import validate_generated_gamma_slab +from apex.core.property.gamma_geometry import ( + build_parent_gamma_slab, + validate_gamma_cell_geometry, + validate_vacuum_size, +) from apex.core.refine import make_refine from apex.core.reproduce import make_repro from apex.core.reproduce import post_repro -from apex.core.structure import StructureInfo +from apex.core.structure import ( + StructureInfo, + normalize_parent_lattice_hint, + resolve_parent_lattice_hint, +) from dflow.python import upload_packages upload_packages.append(__file__) +def _sha256_file(path): + digest = hashlib.sha256() + with open(path, "rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _resolve_require_orthogonal_cell(parameter, default=False): + explicit = parameter.get("require_orthogonal_cell", None) + legacy_alias = parameter.get("orthogonalize_cell", None) + for key, candidate in ( + ("require_orthogonal_cell", explicit), + ("orthogonalize_cell", legacy_alias), + ): + if candidate is not None and not isinstance(candidate, (bool, np.bool_)): + raise ValueError(f"gamma_surface {key} must be a boolean") + if explicit is not None and legacy_alias is not None and explicit != legacy_alias: + raise ValueError( + "gamma_surface require_orthogonal_cell and orthogonalize_cell disagree" + ) + value = explicit if explicit is not None else legacy_alias + if value is None: + value = default + if legacy_alias: + logging.info( + "gamma_surface orthogonalize_cell=true is a strict geometry gate; " + "APEX does not Gram-Schmidt periodic cells" + ) + parameter["require_orthogonal_cell"] = bool(value) + return bool(value) + + class GammaSurface(Property): """Calculation of generalized stacking fault energy surface.""" def __init__(self, parameter, inter_param=None): + self.parent_lattice = normalize_parent_lattice_hint( + parameter.get("parent_lattice") + ) + if self.parent_lattice is not None: + parameter["parent_lattice"] = self.parent_lattice self._add_fix_explicit = "add_fix" in parameter parameter["reproduce"] = parameter.get("reproduce", False) self.reprod = parameter["reproduce"] @@ -52,8 +106,22 @@ def __init__(self, parameter, inter_param=None): self.plane_shift = parameter["plane_shift"] parameter["supercell_size"] = parameter.get("supercell_size", (1, 1, 5)) self.supercell_size = parameter["supercell_size"] - parameter["vacuum_size"] = parameter.get("vacuum_size", 0) - self.vacuum_size = parameter["vacuum_size"] + parameter["min_slab_height"] = parameter.get( + "min_slab_height", None + ) + self.min_slab_height = parameter["min_slab_height"] + parameter["max_atoms"] = parameter.get("max_atoms", None) + self.max_atoms = parameter["max_atoms"] + parameter["min_distance"] = parameter.get("min_distance", 0.2) + self.min_distance = parameter["min_distance"] + parameter["vacuum_size"] = parameter.get("vacuum_size", 20) + self.vacuum_size = validate_vacuum_size( + parameter["vacuum_size"], "GammaSurface" + ) + parameter["vacuum_size"] = self.vacuum_size + self.require_orthogonal_cell = _resolve_require_orthogonal_cell( + parameter + ) parameter["add_fix"] = parameter.get( "add_fix", ["true", "true", "false"] ) @@ -207,6 +275,7 @@ def make_confs( "slip_length_y.json", "slip_vector_x.json", "slip_vector_y.json", + "gamma_geometry.json", ): source = os.path.join(init_from_task, metadata) if not os.path.exists(source): @@ -260,8 +329,26 @@ def make_confs( ss = Structure.from_file("CONTCAR.direct") os.chdir(cwd) st = StructureInfo(ss) - self.structure_type = st.lattice_structure - self.conv_std_structure = st.conventional_structure + self.detected_structure_type = st.lattice_structure + self.structure_type, self.structure_type_source = ( + resolve_parent_lattice_hint( + self.detected_structure_type, self.parent_lattice + ) + ) + if self.structure_type_source == "user_override": + logging.info( + "GammaSurface parent_lattice=%s overrides automatic " + "structure classification %s without changing the input " + "geometry.", + self.structure_type, + self.detected_structure_type, + ) + # A parent hint defines the crystallographic index basis for a + # disordered supercell. Preserve the exact relaxed row basis so + # the integer parent-supercell mapping remains traceable. + self.conv_std_structure = ( + ss if self.parent_lattice is not None else st.conventional_structure + ) relax_a = self.conv_std_structure.lattice.a relax_b = self.conv_std_structure.lattice.b relax_c = self.conv_std_structure.lattice.c @@ -278,7 +365,17 @@ def make_confs( self.closed_loop = bool(closed_loop) self.plane_shift = type_param.get("plane_shift", self.plane_shift) self.supercell_size = type_param.get("supercell_size", self.supercell_size) + self.min_slab_height = type_param.get( + "min_slab_height", self.min_slab_height + ) + self.max_atoms = type_param.get("max_atoms", self.max_atoms) + self.min_distance = type_param.get( + "min_distance", self.min_distance + ) self.vacuum_size = type_param.get("vacuum_size", self.vacuum_size) + self.require_orthogonal_cell = _resolve_require_orthogonal_cell( + type_param, self.require_orthogonal_cell + ) self.add_fix = type_param.get("add_fix", self.add_fix) self.n_steps_x = type_param.get( "n_steps_x", type_param.get("n_steps", self.n_steps_x) @@ -286,6 +383,10 @@ def make_confs( self.n_steps = self.n_steps_x self.n_steps_y = type_param.get("n_steps_y", self.n_steps_y) + self.vacuum_size = validate_vacuum_size( + self.vacuum_size, "GammaSurface" + ) + if not (self.plane_miller and self.slip_direction): raise RuntimeError( "fail to obtain both slip plane and slip direction info from input json file!" @@ -319,14 +420,121 @@ def make_confs( "Please double check generated slab structures.", self.structure_type, ) + ( + self.supercell_size, + self.min_slab_height, + self.max_atoms, + self.min_distance, + ) = validate_gamma_slab_settings( + self.supercell_size, + self.min_slab_height, + self.max_atoms, + self.min_distance, + ) - plane_miller, _, slip_length_x, Q = self.__convert_input_miller( - self.conv_std_structure + self.parent_mapping = None + self.parent_slip_geometry = None + self.gamma_geometry = None + parent_slip_vector_x = None + if self.parent_lattice is not None: + self.parent_mapping = resolve_parent_supercell( + self.conv_std_structure, self.parent_lattice + ) + self.parent_slip_geometry = resolve_parent_slip_geometry( + self.conv_std_structure, + self.parent_mapping, + self.plane_miller, + self.slip_direction, + self.slip_length, + ) + built = build_parent_gamma_slab( + self.conv_std_structure, + self.parent_slip_geometry, + self.supercell_size, + self.supercell_size[2], + self.min_slab_height, + self.vacuum_size, + self.plane_shift, + self.require_orthogonal_cell, + ) + slab = built.slab + self._gamma_upper_indices = built.upper_indices + self._gamma_lower_indices = built.lower_indices + self._slab_generation_metadata = built.generation_metadata + self.slab_generation = validate_generated_gamma_slab( + slab, + built.generation_metadata, + self.supercell_size[:2], + self.max_atoms, + self.min_distance, + "GammaSurface", + ) + self.slab_generation.update(built.metadata) + parent_slip_vector_x = ( + self.parent_slip_geometry.local_frame + @ self.parent_slip_geometry.burgers_vector_cart + ) + slip_length_x = float(np.linalg.norm(parent_slip_vector_x)) + self.gamma_geometry = { + "parent_mapping": self.parent_mapping.as_dict(), + "slip_geometry": self.parent_slip_geometry.as_dict(), + "slab_geometry": built.metadata, + "source_structure": { + "path": os.path.abspath(equi_contcar), + "sha256": _sha256_file(equi_contcar), + "atom_count": len(self.conv_std_structure), + }, + } + else: + plane_miller, _, slip_length_x, Q = self.__convert_input_miller( + self.conv_std_structure + ) + slab = self.__gen_slab_pmg( + self.conv_std_structure, plane_miller, trans_matrix=Q + ) + cell_geometry = validate_gamma_cell_geometry( + slab, + require_orthogonal=self.require_orthogonal_cell, + property_name="GammaSurface", + ) + self.slab_generation.update( + { + "interface_count": 1 if self.vacuum_size > 0 else 2, + "cell_geometry": cell_geometry, + "require_orthogonal_cell": self.require_orthogonal_cell, + } + ) + self.gamma_geometry = { + "parent_mapping": None, + "slip_geometry": None, + "slab_geometry": { + "interface_count": self.slab_generation["interface_count"], + "added_vacuum_angstrom": self.vacuum_size, + "cell_geometry": cell_geometry, + "require_orthogonal_cell": self.require_orthogonal_cell, + }, + "source_structure": { + "path": os.path.abspath(equi_contcar), + "sha256": _sha256_file(equi_contcar), + "atom_count": len(self.conv_std_structure), + }, + } + self.atom_num = len(slab.sites) + self.slab_generation.update( + { + "detected_structure_type": self.detected_structure_type, + "effective_parent_lattice": self.structure_type, + "structure_type_source": self.structure_type_source, + } ) - slab = self.__gen_slab_pmg( - self.conv_std_structure, plane_miller, trans_matrix=Q + dumpfn( + self.slab_generation, + os.path.join(path_to_work, "slab_generation.json"), + ) + dumpfn( + self.gamma_geometry, + os.path.join(path_to_work, "gamma_geometry.json"), ) - self.atom_num = len(slab.sites) os.chdir(path_to_work) if os.path.exists(POSCAR): @@ -339,26 +547,39 @@ def make_confs( slab, slip_vector_x ) self.__validate_closed_loop( - slab, slip_vector_x, slip_vector_y + slab, + slip_vector_x, + slip_vector_y, + getattr(self, "_gamma_upper_indices", None), ) slip_length_x = float(np.linalg.norm(slip_vector_x)) slip_length_y = float(np.linalg.norm(slip_vector_y)) else: - slip_length_x = self.__resolve_slip_length( - slip_length_x, relax_a, relax_b, relax_c - ) + if parent_slip_vector_x is not None: + slip_vector_x = np.asarray( + parent_slip_vector_x, dtype=float + ) + slip_length_x = float(np.linalg.norm(slip_vector_x)) + else: + slip_length_x = self.__resolve_slip_length( + slip_length_x, relax_a, relax_b, relax_c + ) + slip_vector_x = np.array([slip_length_x, 0.0, 0.0]) if self.slip_length_y is None: slip_length_y = slip_length_x else: slip_length_y = self.__resolve_slip_length( self.slip_length_y, relax_a, relax_b, relax_c ) - slip_vector_x = np.array([slip_length_x, 0.0, 0.0]) slip_vector_y = np.array([0.0, slip_length_y, 0.0]) self.slip_length = slip_length_x self.slip_length_y = slip_length_y - top_atoms = np.where(slab.frac_coords[:, 2] > 0.5)[0] + top_atoms = getattr( + self, + "_gamma_upper_indices", + np.where(slab.frac_coords[:, 2] > 0.5)[0], + ) n_steps_x = self.n_steps_x n_steps_y = self.n_steps_y @@ -386,6 +607,14 @@ def make_confs( frac_coords=False, to_unit_cell=True, ) + validate_generated_gamma_slab( + slab_task, + self._slab_generation_metadata, + self.supercell_size[:2], + self.max_atoms, + self.min_distance, + f"GammaSurface task ({idx_x}, {idx_y})", + ) slab_task.to("POSCAR.tmp", "POSCAR") vasp_utils.regulate_poscar("POSCAR.tmp", "POSCAR") @@ -407,6 +636,7 @@ def make_confs( }, "displacement.json", ) + dumpfn(self.gamma_geometry, "gamma_geometry.json") count += 1 os.chdir(cwd) @@ -503,10 +733,15 @@ def __validate_closed_loop( slab: Structure, slip_vector_x: np.ndarray, slip_vector_y: np.ndarray, + upper_indices=None, tol: float = 1e-8, ) -> None: """Verify that translating the upper slab closes all grid corners.""" - top_atoms = np.where(slab.frac_coords[:, 2] > 0.5)[0] + top_atoms = ( + np.asarray(upper_indices, dtype=int) + if upper_indices is not None + else np.where(slab.frac_coords[:, 2] > 0.5)[0] + ) if len(top_atoms) == 0 or len(top_atoms) == len(slab): raise RuntimeError( "Cannot define gamma_surface fault plane: the slab is not split " @@ -557,14 +792,20 @@ def __convert_input_miller(self, structure: Structure): combined_key = "x".join([plane_str, slip_str]) l2_normalize_1d = lambda v: v / np.linalg.norm(v, 2) - dir_dict = SlabSlipSystem.atomic_system_dict() + # Match Gamma line: use physical recommendations when available, and + # otherwise warn before falling back to an in-plane geometry check. + dir_dict = SlabSlipSystem.recommended_system_dict() try: system = dir_dict[self.structure_type] plane_miller, x_miller, xy_miller, stored_slip_length = system[combined_key].values() except KeyError: logging.warning( - "Input slip system is not pre-defined in GammaSurface. " - "Please double check generated slab structure." + "Warning:\n" + "The input slip system is not one of the physically recommended " + "FCC/BCC/HCP systems in README section 4.10.\n" + "GammaSurface is falling back to a geometric construction and will " + "only check that slip_direction lies on plane_miller. Double-check " + "the generated slab, especially for HCP or structure type \"other\"." ) x_miller = slip_direction if not slip_length: @@ -598,38 +839,25 @@ def __convert_input_miller(self, structure: Structure): y_cartesian_unit_vector = l2_normalize_1d( np.cross(z_cartesian_unit_vector, x_cartesian_unit_vector) ) - finally: - reoriented_basis = np.array( - [x_cartesian_unit_vector, y_cartesian_unit_vector, z_cartesian_unit_vector] - ) - Q = trans_mat_basis(reoriented_basis) + reoriented_basis = np.array( + [x_cartesian_unit_vector, y_cartesian_unit_vector, z_cartesian_unit_vector] + ) + Q = trans_mat_basis(reoriented_basis) return plane_miller, x_miller, slip_length, Q def __gen_slab_pmg(self, structure: Structure, plane_miller, trans_matrix=None) -> Structure: - tem_calc_obj = TEMCalculator() - spacing_dict = tem_calc_obj.get_interplanar_spacings(self.conv_std_structure, [plane_miller]) - slab_size = spacing_dict[plane_miller] * self.supercell_size[2] - slab_gen = SlabGenerator( + slab_gen, generation_metadata = make_gamma_slab_generator( structure, - miller_index=plane_miller, - min_slab_size=slab_size, - min_vacuum_size=0, - center_slab=True, - in_unit_planes=False, - lll_reduce=True, - reorient_lattice=False, - primitive=False, + plane_miller, + self.supercell_size[2], + self.min_slab_height, ) - slabs_pmg = slab_gen.get_slabs(ftol=0.001) - matching_slabs = [ - slab for slab in slabs_pmg if slab.miller_index == plane_miller - ] - if not matching_slabs: + slab = get_first_gamma_slab(slab_gen, ftol=0.001) + if slab.miller_index != tuple(plane_miller): raise RuntimeError( f"Cannot generate a gamma_surface slab for Miller plane {plane_miller}" ) - slab = matching_slabs[0] if trans_matrix is not None and np.asarray(trans_matrix).any(): reoriented_lattice_vectors = [trans_matrix.dot(v) for v in slab.lattice.matrix] slab = Structure( @@ -670,6 +898,26 @@ def __gen_slab_pmg(self, structure: Structure, plane_miller, trans_matrix=None) slab.make_supercell( scaling_matrix=[self.supercell_size[0], self.supercell_size[1], 1] ) + self._slab_generation_metadata = generation_metadata + self.slab_generation = validate_generated_gamma_slab( + slab, + generation_metadata, + self.supercell_size[:2], + self.max_atoms, + self.min_distance, + "GammaSurface", + ) + self.slab_generation.update( + { + "min_slab_height": self.min_slab_height, + "max_atoms": self.max_atoms, + "min_distance_threshold": self.min_distance, + "vacuum_size": self.vacuum_size, + } + ) + logging.info( + "GammaSurface slab generation: %s", self.slab_generation + ) return slab def __poscar_fix(self, poscar) -> None: @@ -766,6 +1014,11 @@ def _compute_lower(self, output_file, all_tasks, all_res): ) all_tasks.sort() task_result_slab_equi = loadfn(os.path.join(all_tasks[0], "result_task.json")) + if is_failed_task_result(task_result_slab_equi): + raise RuntimeError( + "gamma_surface reference task " + f"{os.path.basename(all_tasks[0])} failed; cannot compute SFE" + ) slip_length_x = loadfn(os.path.join(all_tasks[0], "slip_length_x.json")) slip_length_y = loadfn(os.path.join(all_tasks[0], "slip_length_y.json")) slip_vector_x_path = os.path.join(all_tasks[0], "slip_vector_x.json") @@ -788,26 +1041,40 @@ def _compute_lower(self, output_file, all_tasks, all_res): ) equi_result = loadfn(os.path.join(equi_path, "result.json")) equi_epa = equi_result["energies"][-1] / np.sum(equi_result["atom_numbs"]) - - for ii in all_tasks: - task_result = loadfn(os.path.join(ii, "result_task.json")) - natoms = np.sum(task_result["atom_numbs"]) - epa = task_result["energies"][-1] / natoms - equi_epa_slab = task_result_slab_equi["energies"][-1] / natoms - area = np.linalg.norm( - np.cross(task_result["cells"][-1][0], task_result["cells"][-1][1]) + ref_energy = task_result_slab_equi["energies"][-1] + ref_natoms = np.sum(task_result_slab_equi["atom_numbs"]) + equi_epa_slab = ref_energy / ref_natoms + geometry_file = os.path.join(all_tasks[0], "gamma_geometry.json") + interface_count = ( + 1 if getattr(self, "vacuum_size", 20.0) > 0 else 2 + ) + if os.path.isfile(geometry_file): + geometry = loadfn(geometry_file) + interface_count = int( + geometry.get("slab_geometry", {}).get( + "interface_count", interface_count + ) + ) + if interface_count not in (1, 2): + raise RuntimeError( + f"Invalid GammaSurface interface_count={interface_count}" + ) + reference_cell = np.asarray( + task_result_slab_equi["cells"][-1], dtype=float + ) + reference_area = float( + np.linalg.norm(np.cross(reference_cell[0], reference_cell[1])) + ) + if not np.isfinite(reference_area) or reference_area <= 0.0: + raise RuntimeError( + "GammaSurface reference task has an invalid in-plane area" ) + for ii in all_tasks: structure_dir = os.path.basename(ii) disp_info = loadfn(os.path.join(ii, "displacement.json")) frac_x = float(disp_info["frac_x"]) frac_y = float(disp_info["frac_y"]) - cf = 1.60217657e-16 / 1e-20 * 0.001 - sfe = ( - (task_result["energies"][-1] - task_result_slab_equi["energies"][-1]) - / area - * cf - ) miller_index = loadfn(os.path.join(ii, "miller.json")) path_x = slip_length_x * frac_x path_y = slip_length_y * frac_y @@ -818,6 +1085,32 @@ def _compute_lower(self, output_file, all_tasks, all_res): ), dtype=float, ) + task_result = loadfn(os.path.join(ii, "result_task.json")) + if is_failed_task_result(task_result): + sfe = float("nan") + epa = float("nan") + else: + natoms = np.sum(task_result["atom_numbs"]) + epa = task_result["energies"][-1] / natoms + area = np.linalg.norm( + np.cross( + task_result["cells"][-1][0], task_result["cells"][-1][1] + ) + ) + if not np.isclose( + area, reference_area, rtol=1.0e-8, atol=1.0e-8 + ): + raise RuntimeError( + "GammaSurface task in-plane area changed relative to " + f"the (0,0) reference: {area:.12g} vs " + f"{reference_area:.12g} A^2" + ) + cf = 1.60217657e-16 / 1e-20 * 0.001 + sfe = ( + (task_result["energies"][-1] - ref_energy) + / (area * interface_count) + * cf + ) ptr_data += ( "%-25s %7.3f %7.3f %7.3f %7.3f " "%7.3f %7.3f %7.3f %7.3f %8.3f %8.3f\n" diff --git a/apex/core/property/Gruneisen.py b/apex/core/property/Gruneisen.py index 1e93c3c2..e777c036 100644 --- a/apex/core/property/Gruneisen.py +++ b/apex/core/property/Gruneisen.py @@ -492,7 +492,7 @@ def post_process(self, task_list): if pair_line_id is not None: contents = contents[: pair_line_id + 1] with open("in.lammps", "w") as f2: - f2.write(helper._ensure_deepmd_plugin_loaded("".join(contents))) + f2.write(helper._strip_legacy_deepmd_plugin("".join(contents))) self._write_fixed_volume_relax_inputs(task_dir) with open("run_command", "w") as f3: f3.write("bash run_gruneisen_task.sh") diff --git a/apex/core/property/Interstitial.py b/apex/core/property/Interstitial.py index dfdf1038..0f30b98a 100644 --- a/apex/core/property/Interstitial.py +++ b/apex/core/property/Interstitial.py @@ -16,7 +16,7 @@ from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import lammps_utils -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result from apex.core.refine import make_refine from apex.core.reproduce import make_repro, post_repro from apex.core.structure import StructureInfo @@ -35,7 +35,12 @@ def __init__(self, parameter, inter_param=None): default_supercell = [1, 1, 1] parameter["supercell"] = parameter.get("supercell", default_supercell) self.supercell = parameter["supercell"] - self.insert_ele = parameter.get("insert_ele", None) + _insert_ele = parameter.get("insert_ele", None) + # Normalize insert_ele to list (fix: iterating a bare string + # like "Cu" yields chars 'C','u' which are not valid elements) + if isinstance(_insert_ele, str): + _insert_ele = [_insert_ele] + self.insert_ele = _insert_ele parameter["lattice_type"] = parameter.get("lattice_type", None) self.lattice_type = parameter["lattice_type"] parameter["voronoi_param"] = parameter.get("voronoi_param", {}) @@ -539,22 +544,28 @@ def _compute_lower(self, output_file, all_tasks, all_res): for idid, ii in enumerate(all_tasks, start=0): # skip task.000000 structure_dir = os.path.basename(ii) - task_result = loadfn(all_res[idid]) interstitial_type = loadfn(os.path.join(ii, 'interstitial_type.json')) - natoms = sum(task_result["atom_numbs"]) - evac = task_result["energies"][-1] - equi_epa * natoms - supercell_index = loadfn(os.path.join(ii, "supercell.json")) # insert_ele = loadfn(os.path.join(ii, 'task.json'))['insert_ele'][0] insert_ele = fc[idid] + task_result = loadfn(all_res[idid]) + if is_failed_task_result(task_result): + evac = float("nan") + task_energy = float("nan") + equi_energy = float("nan") + else: + natoms = sum(task_result["atom_numbs"]) + task_energy = task_result["energies"][-1] + equi_energy = equi_epa * natoms + evac = task_energy - equi_energy ptr_data += "%s: \t%7.3f \t%7.3f \t%7.3f \n" % ( insert_ele + "_" + str(interstitial_type) + "_" + structure_dir, evac, - task_result["energies"][-1], - equi_epa * natoms, + task_energy, + equi_energy, ) res_data[ insert_ele + "_" + str(interstitial_type) + "_" + structure_dir - ] = [evac, task_result["energies"][-1], equi_epa * natoms] + ] = [evac, task_energy, equi_energy] else: if "init_data_path" not in self.parameter: diff --git a/apex/core/property/MeltingPoint.py b/apex/core/property/MeltingPoint.py new file mode 100644 index 00000000..e7b4250c --- /dev/null +++ b/apex/core/property/MeltingPoint.py @@ -0,0 +1,895 @@ +"""Two-phase-coexistence melting-point workflow for LAMMPS. + +The property prepares one solid/liquid coexistence trajectory per target +temperature and velocity-seed replica. The upper part of the simulation cell +is premelted while the lower crystal is pinned, both halves are conditioned at +the target temperature, and the complete cell is released under zero-pressure +NPT dynamics. Melting is bracketed from the sign of the q6-derived interface +velocity, rather than from a single-phase energy or MSD discontinuity. +""" + +from __future__ import annotations + +import csv +import json +import math +import os +import re +import shutil +from collections import defaultdict +from numbers import Integral, Real +from typing import Any, Dict, Iterable, List + +import numpy as np +from monty.serialization import dumpfn, loadfn +from pymatgen.core.periodic_table import Element + +from apex.core.property.Property import Property +from dflow.python import upload_packages + +upload_packages.append(__file__) + +PROPERTY_TYPE = "melting_point" +METADATA_FILE = "MeltingPoint.json" +VARIABLE_FILE = "variable_MeltingPoint.in" +DUMP_FILE = "dump.melting" + +DEFAULT_CAL_SETTING = { + "temperature": [1500, 1600, 1700], + "premelt_temperature": 4500.0, + "premelt_steps": 5000, + "conditioning_steps": 5000, + "production_steps": 100000, + "timestep": 0.001, + "tdamp": 0.1, + "pdamp": 1.0, + "pressure": 0.0, + "barostat": "iso", + "interface_axis": "z", + "liquid_fraction": 0.5, + "dump_step": 100, + "thermo_step": 100, + "restart_interval": 10000, + "q6_cutoff": 3.5, + "q6_neighbors": 12, + "spatial_bins": 12, + "analysis_stride": 2, + "analysis_block_ps": 2.0, + "minimum_q6_gap": 0.03, + "minimum_directional_change": 0.02, + "replicas": 1, + "velocity_seeds": { + "premelt": 324159, + "condition": 271828, + "release": 161803, + }, +} + + +class MeltingPoint(Property): + """LAMMPS-only direct two-phase melting-point bracket.""" + + def __init__(self, parameter: Dict[str, Any], inter_param=None): + inter_param = inter_param or {"type": "deepmd"} + if inter_param.get("type") in {"vasp", "abacus"}: + raise NotImplementedError( + "melting_point method='two_phase' supports only LAMMPS" + ) + parameter = dict(parameter) + method = str(parameter.get("method", "two_phase")).lower().replace("-", "_") + if method in {"coexistence", "two_phase_coexistence", "direct_coexistence"}: + method = "two_phase" + if method != "two_phase": + raise ValueError("melting_point currently supports method='two_phase'") + + parameter["type"] = PROPERTY_TYPE + parameter["method"] = method + parameter["cal_type"] = PROPERTY_TYPE + parameter.setdefault("supercell_size", [1, 1, 1]) + cal = dict(DEFAULT_CAL_SETTING) + cal.update(parameter.get("cal_setting", {})) + cal["velocity_seeds"] = _normalize_velocity_seeds( + cal.get("velocity_seeds"), int(cal.get("replicas", 1)) + ) + _validate_settings(parameter["supercell_size"], cal) + parameter["cal_setting"] = cal + + self.parameter = parameter + self.inter_param = inter_param + self.supercell_size = [int(v) for v in parameter["supercell_size"]] + self.cal_setting = cal + + def task_type(self): + return PROPERTY_TYPE + + def task_param(self): + return self.parameter + + def make_confs(self, path_to_work: str, path_to_equi: str, refine=False): + if refine: + raise NotImplementedError( + "melting_point refinement is expressed as a new temperature list/suffix" + ) + path_to_work = os.path.abspath(path_to_work) + path_to_equi = os.path.abspath(path_to_equi) + os.makedirs(path_to_work, exist_ok=True) + contcar = os.path.join(path_to_equi, "CONTCAR") + if not os.path.isfile(contcar): + raise RuntimeError("please finish relaxation before melting_point") + + restart_files = self.cal_setting.get("restart_files") + if restart_files is not None: + if len(restart_files) != len(self.cal_setting["temperature"]): + raise ValueError( + "melting_point cal_setting.restart_files must have one " + "entry per temperature" + ) + restart_files = [os.path.abspath(path) for path in restart_files] + missing = [path for path in restart_files if not os.path.isfile(path)] + if missing: + raise FileNotFoundError( + "melting_point restart file(s) not found: " + ", ".join(missing) + ) + + tasks = [] + replicas = int(self.cal_setting["replicas"]) + for temperature_index, temperature in enumerate(self.cal_setting["temperature"]): + for replica in range(replicas): + task_dir = os.path.join(path_to_work, f"task.{len(tasks):06d}") + os.makedirs(task_dir, exist_ok=True) + shutil.copyfile(contcar, os.path.join(task_dir, "POSCAR")) + if restart_files is not None: + shutil.copyfile( + restart_files[temperature_index], + os.path.join(task_dir, "restart.coexistence.start"), + ) + metadata = self._metadata(float(temperature), replica) + dumpfn(metadata, os.path.join(task_dir, METADATA_FILE), indent=4) + with open(os.path.join(task_dir, VARIABLE_FILE), "w") as fp: + fp.write(_variable_file(metadata)) + tasks.append(task_dir) + return tasks + + def post_process(self, task_list): + if task_list: + dumpfn( + [os.path.basename(path) for path in task_list], + os.path.join(os.path.dirname(task_list[0]), "task_list.json"), + indent=4, + ) + + def compute(self, output_file, print_file, path_to_work): + tasks = sorted( + path for path in ( + os.path.join(path_to_work, name) for name in os.listdir(path_to_work) + ) + if os.path.isdir(path) and re.match(r"task\.[0-9]+$", os.path.basename(path)) + ) + result, report = self._compute_lower(output_file, tasks, []) + with open(print_file, "w") as fp: + fp.write(report) + _write_tidy_csv(os.path.join(path_to_work, "melting_point_tidy.csv"), result) + _write_plots(path_to_work, result) + + def _compute_lower(self, output_file, all_tasks, all_res): + points = [] + failures = [] + for task in all_tasks: + try: + points.append(_analyse_task(task)) + except Exception as exc: + failures.append({"task": os.path.basename(task), "error": str(exc)}) + points.sort(key=lambda row: (row["temperature_K"], row["replica"])) + temperatures = _aggregate_temperatures( + points, + expected_replicas=int(self.cal_setting["replicas"]), + expected_temperatures=self.cal_setting["temperature"], + ) + bracket = _infer_bracket(temperatures) + result = { + "schema": "apex.melting_point.two_phase/v1", + "property": PROPERTY_TYPE, + "method": "two_phase", + "criterion": ( + "q6-derived solid-fraction Theil-Sen slope; 95% interval must " + "exclude zero and projected release change must exceed threshold" + ), + "points": points, + "temperatures": temperatures, + "bracket": bracket, + "failed_tasks": failures, + "replica_warning": ( + "Replica-to-replica variation is unavailable with one replica per temperature" + if int(self.cal_setting["replicas"]) == 1 else None + ), + "settings": self.parameter, + } + dumpfn(result, output_file, indent=4) + report = _format_report(os.path.dirname(output_file), result) + return result, report + + def _metadata(self, temperature: float, replica: int): + cal = self.cal_setting + seeds = cal["velocity_seeds"][replica] + premelt_steps = int(cal["premelt_steps"]) + conditioning_steps = int(cal["conditioning_steps"]) + restart_mode = cal.get("restart_files") is not None + return { + "schema": "apex.melting_point.task/v1", + "property": PROPERTY_TYPE, + "method": "two_phase", + "temperature_K": temperature, + "replica": replica + 1, + "supercell_size": self.supercell_size, + "interface_axis": cal["interface_axis"], + "liquid_fraction": float(cal["liquid_fraction"]), + "premelt_temperature_K": float(cal["premelt_temperature"]), + "premelt_steps": premelt_steps, + "conditioning_steps": conditioning_steps, + # A restart is an already prepared coexistence state. Reset its + # timestep to zero and use that initial frame as the analysis + # reference instead of repeating premelting/conditioning. + "restart_mode": restart_mode, + "reference_step": 0 if restart_mode else premelt_steps, + "release_step": 0 if restart_mode else premelt_steps + conditioning_steps, + "production_steps": int(cal["production_steps"]), + "timestep_ps": float(cal["timestep"]), + "tdamp_ps": float(cal["tdamp"]), + "pdamp_ps": float(cal["pdamp"]), + "pressure_bar": float(cal["pressure"]), + "barostat": cal["barostat"], + "dump_step": int(cal["dump_step"]), + "thermo_step": int(cal["thermo_step"]), + "restart_interval": int(cal["restart_interval"]), + "q6_cutoff_A": float(cal["q6_cutoff"]), + "q6_neighbors": int(cal["q6_neighbors"]), + "spatial_bins": int(cal["spatial_bins"]), + "analysis_stride": int(cal["analysis_stride"]), + "analysis_block_ps": float(cal["analysis_block_ps"]), + "minimum_q6_gap": float(cal["minimum_q6_gap"]), + "minimum_directional_change": float(cal["minimum_directional_change"]), + "temperature_tolerance_K": float( + cal.get("temperature_tolerance_K", max(50.0, 0.05 * temperature)) + ), + "velocity_seeds": seeds, + } + + +def _normalize_velocity_seeds(value, replicas: int): + if replicas < 1: + raise ValueError("cal_setting.replicas must be >= 1") + if isinstance(value, list): + if len(value) != replicas: + raise ValueError("velocity_seeds list length must equal replicas") + source = value + elif isinstance(value, dict): + for key, seed in value.items(): + if isinstance(seed, (bool, np.bool_)) or not isinstance(seed, Integral): + raise ValueError("velocity_seeds must contain positive integers") + source = [ + {key: seed + 104729 * index for key, seed in value.items()} + for index in range(replicas) + ] + else: + raise ValueError("velocity_seeds must be a mapping or list of mappings") + normalized = [] + for seeds in source: + missing = {"premelt", "condition", "release"} - set(seeds) + if missing: + raise ValueError(f"velocity_seeds missing keys: {sorted(missing)}") + normalized_seeds = {} + for key in ("premelt", "condition", "release"): + seed = seeds[key] + if isinstance(seed, (bool, np.bool_)) or not isinstance(seed, Integral): + raise ValueError("velocity_seeds must contain positive integers") + normalized_seeds[key] = int(seed) + invalid = [key for key, seed in normalized_seeds.items() if seed <= 0] + if invalid: + raise ValueError( + "velocity_seeds must contain positive integers; invalid keys: " + f"{invalid}" + ) + normalized.append(normalized_seeds) + return normalized + + +def _validate_settings(supercell_size, cal): + if ( + not isinstance(supercell_size, (list, tuple)) + or len(supercell_size) != 3 + or any( + isinstance(value, (bool, np.bool_)) + or not isinstance(value, Integral) + or value < 1 + for value in supercell_size + ) + ): + raise ValueError("supercell_size must contain three positive integers") + temperatures = cal.get("temperature") + if not isinstance(temperatures, list) or not temperatures: + raise ValueError("cal_setting.temperature must be a non-empty list") + if any( + isinstance(temp, (bool, np.bool_)) + or not isinstance(temp, Real) + or not np.isfinite(temp) + or temp <= 0 + for temp in temperatures + ): + raise ValueError("all melting-point temperatures must be positive") + if cal["interface_axis"] not in {"x", "y", "z"}: + raise ValueError("interface_axis must be x, y, or z") + if not 0.1 <= float(cal["liquid_fraction"]) <= 0.9: + raise ValueError("liquid_fraction must be between 0.1 and 0.9") + if cal["barostat"] not in {"iso", "aniso", "x", "y", "z"}: + raise ValueError("barostat must be iso, aniso, x, y, or z") + for key in ( + "premelt_steps", + "conditioning_steps", + "production_steps", + "dump_step", + "thermo_step", + "restart_interval", + ): + value = cal[key] + if ( + isinstance(value, (bool, np.bool_)) + or not isinstance(value, Integral) + or value <= 0 + ): + raise ValueError(f"{key} must be positive") + if cal["production_steps"] < 10 * cal["dump_step"]: + raise ValueError("production_steps must contain at least ten dump intervals") + + +def _variable_file(meta): + axis_index = {"x": 0, "y": 1, "z": 2}[meta["interface_axis"]] + seeds = meta["velocity_seeds"] + lines = [ + "# variable_MeltingPoint.in", + f"variable temperature equal {meta['temperature_K']:.10g}", + f"variable premelt_temperature equal {meta['premelt_temperature_K']:.10g}", + f"variable nx equal {meta['supercell_size'][0]}", + f"variable ny equal {meta['supercell_size'][1]}", + f"variable nz equal {meta['supercell_size'][2]}", + f"variable interface_axis string {meta['interface_axis']}", + f"variable axis_index equal {axis_index}", + f"variable liquid_fraction equal {meta['liquid_fraction']:.10g}", + f"variable premelt_steps equal {meta['premelt_steps']}", + f"variable conditioning_steps equal {meta['conditioning_steps']}", + f"variable production_steps equal {meta['production_steps']}", + f"variable timestep equal {meta['timestep_ps']:.10g}", + f"variable tdamp equal {meta['tdamp_ps']:.10g}", + f"variable pdamp equal {meta['pdamp_ps']:.10g}", + f"variable target_pressure equal {meta['pressure_bar']:.10g}", + f"variable dump_step equal {meta['dump_step']}", + f"variable thermo_step equal {meta['thermo_step']}", + f"variable restart_interval equal {meta['restart_interval']}", + f"variable q6_cutoff equal {meta['q6_cutoff_A']:.10g}", + f"variable q6_neighbors equal {meta['q6_neighbors']}", + f"variable premelt_seed equal {seeds['premelt']}", + f"variable condition_seed equal {seeds['condition']}", + f"variable release_seed equal {seeds['release']}", + ] + return "\n".join(lines) + "\n" + + +def render_melting_point_lammps_input(conf, type_map, interaction, model_param, task_param=None): + """Render the validated solid/liquid preparation and NPT release.""" + cal = (task_param or {}).get("cal_setting", {}) + axis = cal.get("interface_axis", "z") + barostat = cal.get("barostat", "iso") + restart_mode = cal.get("restart_files") is not None + axis_lo, axis_hi = f"{axis}lo", f"{axis}hi" + lo = "v_split" if float(cal.get("liquid_fraction", 0.5)) < 1.0 else "INF" + region_args = { + "x": f"${{{lo[2:]}}} INF INF INF INF INF" if lo.startswith("v_") else "INF INF INF INF INF INF", + "y": f"INF INF ${{{lo[2:]}}} INF INF INF" if lo.startswith("v_") else "INF INF INF INF INF INF", + "z": f"INF INF INF INF ${{{lo[2:]}}} INF" if lo.startswith("v_") else "INF INF INF INF INF INF", + }[axis] + type_map_list = [ + element for element, _type_id in sorted(type_map.items(), key=lambda item: item[1]) + ] + ret = [ + f"include {VARIABLE_FILE}", + "clear", + "units metal", + "dimension 3", + "boundary p p p", + "atom_style atomic", + "atom_modify map array", + *( ["newton on"] if model_param.get("type") == "mace" else [] ), + "box tilt large", + ] + if restart_mode: + ret.extend([ + "read_restart restart.coexistence.start", + "reset_timestep 0", + ]) + else: + ret.extend([f"read_data {conf}", "replicate ${nx} ${ny} ${nz}"]) + for index, element in enumerate(type_map_list, 1): + ret.append(f"mass {index} {float(Element(element).atomic_mass):.8f}") + ret.extend([ + "neighbor 2.0 bin", + "neigh_modify every 1 delay 0 check yes", + interaction(model_param).rstrip(), + "timestep ${timestep}", + "restart ${restart_interval} restart.melting.1 restart.melting.2", + "compute mype all pe", + "compute q6 all orientorder/atom degrees 1 6 nnn ${q6_neighbors} cutoff ${q6_cutoff}", + "thermo ${thermo_step}", + "thermo_style custom step temp press pe ke etotal enthalpy density pxx pyy pzz pxy pxz pyz lx ly lz vol", + "thermo_modify flush yes", + f"dump melting all custom ${{dump_step}} {DUMP_FILE} id type xs ys zs c_q6[1]", + "dump_modify melting sort id", + f"variable split equal {axis_lo}+(1.0-${{liquid_fraction}})*({axis_hi}-{axis_lo})", + f"region liquid_region block {region_args} units box", + "group liquid_seed region liquid_region", + "group solid_seed subtract all liquid_seed", + ]) + if not restart_mode: + ret.extend([ + "velocity all set 0.0 0.0 0.0", + "velocity liquid_seed create ${premelt_temperature} ${premelt_seed} mom yes rot yes dist gaussian", + "fix pin_solid solid_seed setforce 0.0 0.0 0.0", + "fix melt_liquid liquid_seed nvt temp ${premelt_temperature} ${premelt_temperature} ${tdamp}", + 'print "APEX_MELTING_STAGE premelt_liquid"', + "run ${premelt_steps}", + "unfix melt_liquid", + "unfix pin_solid", + "velocity all create ${temperature} ${condition_seed} mom yes rot yes dist gaussian", + "fix condition_all all nvt temp ${temperature} ${temperature} ${tdamp}", + 'print "APEX_MELTING_STAGE condition_all"', + "run ${conditioning_steps}", + "unfix condition_all", + "velocity all create ${temperature} ${release_seed} mom yes rot yes dist gaussian", + ]) + else: + # Emit the restored state as the u=0 analysis reference before + # continuing the coexistence trajectory with preserved velocities. + ret.extend([ + 'print "APEX_MELTING_STAGE restart_coexistence"', + "run 0", + ]) + pressure_clause = ( + f"{barostat} ${{target_pressure}} ${{target_pressure}} ${{pdamp}}" + if barostat in {"iso", "aniso"} + else f"{barostat} ${{target_pressure}} ${{target_pressure}} ${{pdamp}}" + ) + ret.extend([ + f"fix coexistence all npt temp ${{temperature}} ${{temperature}} ${{tdamp}} {pressure_clause}", + "fix remove_drift all momentum 100 linear 1 1 1", + "compute msd_solid solid_seed msd com yes", + "compute msd_liquid liquid_seed msd com yes", + "thermo_style custom step temp press pe ke etotal enthalpy density pxx pyy pzz pxy pxz pyz lx ly lz vol c_msd_solid[4] c_msd_liquid[4]", + 'print "APEX_MELTING_STAGE coexistence_release"', + "run ${production_steps}", + "write_restart restart.melting.final", + 'print "All done"', + ]) + return "\n".join(ret) + "\n" + + +def _iter_dump_frames(path, axis): + axis_column = {"x": "xs", "y": "ys", "z": "zs"}[axis] + with open(path, errors="replace") as fp: + while True: + line = fp.readline() + if not line: + return + if line.strip() != "ITEM: TIMESTEP": + continue + step = int(fp.readline()) + if fp.readline().strip() != "ITEM: NUMBER OF ATOMS": + raise RuntimeError("malformed LAMMPS dump atom-count header") + natoms = int(fp.readline()) + bounds_header = fp.readline().strip() + if not bounds_header.startswith("ITEM: BOX BOUNDS"): + raise RuntimeError("malformed LAMMPS dump box header") + bounds = [list(map(float, fp.readline().split()[:2])) for _ in range(3)] + atom_header = fp.readline().split()[2:] + try: + id_col = atom_header.index("id") + axis_col = atom_header.index(axis_column) + q6_col = next(i for i, name in enumerate(atom_header) if name.startswith("c_q6")) + x_col = atom_header.index("xs") + y_col = atom_header.index("ys") + z_col = atom_header.index("zs") + except (ValueError, StopIteration) as exc: + raise RuntimeError("dump must contain id, xs/ys/zs and c_q6") from exc + ids = np.empty(natoms, dtype=int) + scaled = np.empty(natoms, dtype=float) + coords = np.empty((natoms, 3), dtype=float) + q6 = np.empty(natoms, dtype=float) + for row in range(natoms): + fields = fp.readline().split() + ids[row] = int(fields[id_col]) + scaled[row] = float(fields[axis_col]) % 1.0 + coords[row] = [float(fields[x_col]) % 1.0, float(fields[y_col]) % 1.0, float(fields[z_col]) % 1.0] + q6[row] = float(fields[q6_col]) + order = np.argsort(ids) + length = bounds[{"x": 0, "y": 1, "z": 2}[axis]][1] - bounds[{"x": 0, "y": 1, "z": 2}[axis]][0] + yield {"step": step, "scaled": scaled[order], "coords": coords[order], "q6": q6[order], "axis_length_A": length} + + +def _analyse_task(task_dir): + meta = loadfn(os.path.join(task_dir, METADATA_FILE)) + dump_path = os.path.join(task_dir, DUMP_FILE) + if not os.path.isfile(dump_path): + raise RuntimeError(f"missing {DUMP_FILE}") + release_step = int(meta["release_step"]) + reference_step = int(meta.get("reference_step", meta["premelt_steps"])) + reference = None + reference_matches = 0 + for frame in _iter_dump_frames(dump_path, meta["interface_axis"]): + if frame["step"] == reference_step: + reference = frame + reference_matches += 1 + if reference_matches != 1 or reference is None: + raise RuntimeError(f"expected one reference frame at step {reference_step}") + solid_edge = 1.0 - float(meta["liquid_fraction"]) + solid_bulk = (reference["scaled"] >= 0.08 * solid_edge) & (reference["scaled"] <= 0.84 * solid_edge) + liquid_width = 1.0 - solid_edge + liquid_bulk = ( + (reference["scaled"] >= solid_edge + 0.16 * liquid_width) + & (reference["scaled"] <= solid_edge + 0.92 * liquid_width) + ) + if not np.any(solid_bulk) or not np.any(liquid_bulk): + raise RuntimeError("reference frame has no atoms in a solid/liquid bulk core") + q_solid = float(np.mean(reference["q6"][solid_bulk])) + q_liquid = float(np.mean(reference["q6"][liquid_bulk])) + q_gap = q_solid - q_liquid + + stride = int(meta["analysis_stride"]) + bins = int(meta["spatial_bins"]) + times, fractions, profiles, crossings, lengths = [], [], [], [], [] + if q_gap <= 0: + q_gap = float(q_gap) + + final_frame = None + last_sampled_step = None + release_frame_count = 0 + + def append_frame(frame): + nonlocal final_frame, last_sampled_step + times.append((frame["step"] - release_step) * float(meta["timestep_ps"])) + normalized = np.clip((frame["q6"] - q_liquid) / q_gap, 0.0, 1.0) if q_gap > 0 else np.full_like(frame["q6"], np.nan) + fractions.append(float(np.nanmean(normalized))) + bin_index = np.minimum((frame["scaled"] * bins).astype(int), bins - 1) + profile = [float(np.nanmean(normalized[bin_index == index])) if np.any(bin_index == index) else None for index in range(bins)] + profiles.append(profile) + crossings.append(_interface_crossings(profile)) + lengths.append(frame["axis_length_A"]) + final_frame = frame + last_sampled_step = frame["step"] + + last_release_frame = None + for frame in _iter_dump_frames(dump_path, meta["interface_axis"]): + if frame["step"] < release_step: + continue + if release_frame_count % stride == 0: + append_frame(frame) + release_frame_count += 1 + last_release_frame = frame + if last_release_frame is not None and last_sampled_step != last_release_frame["step"]: + append_frame(last_release_frame) + if len(times) < 6 or final_frame is None: + raise RuntimeError("released trajectory contains too few sampled frames") + + times_array = np.asarray(times) + fractions_array = np.asarray(fractions) + block_times, block_values = _block_means(times_array, fractions_array, float(meta["analysis_block_ps"])) + slope, low, high = _theil_sen(block_times, block_values) + duration = float(times_array[-1] - times_array[0]) + projected = float(slope * duration) + minimum_change = float(meta["minimum_directional_change"]) + thermo = _parse_thermo(os.path.join(task_dir, "log.lammps"), release_step) + tail_temperature = thermo.get("tail_temperature_mean_K") + thermodynamics_valid = bool( + thermo.get("sample_count", 0) >= 10 + and tail_temperature is not None + and abs(tail_temperature - float(meta["temperature_K"])) + <= float( + meta.get( + "temperature_tolerance_K", + max(50.0, 0.05 * float(meta["temperature_K"])), + ) + ) + ) + if not thermodynamics_valid: + outcome = "invalid_thermodynamics" + elif q_gap < float(meta["minimum_q6_gap"]): + outcome = "invalid_preparation" + elif low > 0.0 and projected >= minimum_change: + outcome = "solid_growth" + elif high < 0.0 and projected <= -minimum_change: + outcome = "liquid_growth" + else: + outcome = "inconclusive" + mean_length = float(np.mean(lengths)) + return { + "task": os.path.basename(task_dir), + "temperature_K": float(meta["temperature_K"]), + "replica": int(meta["replica"]), + "atom_count": int(len(reference["q6"])), + "release_frame_count": release_frame_count, + "reference_solid_q6": q_solid, + "reference_liquid_q6": q_liquid, + "reference_q6_gap": q_gap, + "release_duration_ps": duration, + "time_ps": times, + "solid_fraction": fractions, + "liquid_fraction": [float(1.0 - value) for value in fractions], + "spatial_solid_fraction": profiles, + "interface_positions_fractional": crossings, + "interface_motion": { + "block_time_ps": block_times.tolist(), + "block_solid_fraction": block_values.tolist(), + "solid_fraction_slope_per_ps": slope, + "slope_95pct_low_per_ps": low, + "slope_95pct_high_per_ps": high, + "projected_solid_fraction_change": projected, + "interface_velocity_A_per_ps": slope * mean_length / 2.0, + "interface_velocity_95pct_low_A_per_ps": low * mean_length / 2.0, + "interface_velocity_95pct_high_A_per_ps": high * mean_length / 2.0, + "outcome": outcome, + }, + "thermodynamics": thermo, + "thermodynamics_valid": thermodynamics_valid, + "snapshot": { + "reference_scaled_coordinates": reference["coords"].tolist(), + "reference_q6": reference["q6"].tolist(), + "final_scaled_coordinates": final_frame["coords"].tolist(), + "final_q6": final_frame["q6"].tolist(), + }, + "metadata": meta, + } + + +def _interface_crossings(profile): + values = np.asarray([np.nan if value is None else value for value in profile], dtype=float) + result = [] + for index in range(len(values)): + left, right = values[index], values[(index + 1) % len(values)] + if not np.isfinite(left) or not np.isfinite(right) or (left - 0.5) * (right - 0.5) > 0: + continue + fraction = 0.0 if right == left else (0.5 - left) / (right - left) + result.append(float(((index + 0.5 + fraction) / len(values)) % 1.0)) + return result + + +def _block_means(times, values, width): + indices = np.floor((times - times[0]) / width).astype(int) + unique = np.unique(indices) + return ( + np.asarray([np.mean(times[indices == index]) for index in unique]), + np.asarray([np.mean(values[indices == index]) for index in unique]), + ) + + +def _theil_sen(x, y): + if len(x) < 3: + return 0.0, -math.inf, math.inf + try: + from scipy.stats import theilslopes + slope, _intercept, low, high = theilslopes(y, x, alpha=0.95) + return float(slope), float(low), float(high) + except Exception: + slopes = [(y[j] - y[i]) / (x[j] - x[i]) for i in range(len(x)) for j in range(i + 1, len(x)) if x[j] != x[i]] + return float(np.median(slopes)), float(np.percentile(slopes, 2.5)), float(np.percentile(slopes, 97.5)) + + +def _parse_thermo(path, release_step): + header = None + rows = [] + if not os.path.isfile(path): + return {"sample_count": 0} + with open(path, errors="replace") as fp: + for line in fp: + fields = line.split() + if fields and fields[0] == "Step" and "Temp" in fields and "Press" in fields: + header = fields + continue + if header is None or len(fields) != len(header): + continue + try: + row = dict(zip(header, map(float, fields))) + except ValueError: + continue + if row["Step"] >= release_step: + rows.append(row) + if not rows: + return {"sample_count": 0} + tail = rows[len(rows) // 2:] + def stats(keys): + key = next((candidate for candidate in keys if candidate in tail[0]), None) + if key is None: + return None, None + values = np.asarray([row[key] for row in tail]) + return float(values.mean()), float(values.std(ddof=1)) if len(values) > 1 else 0.0 + temp_mean, temp_std = stats(["Temp"]) + pressure_mean, pressure_std = stats(["Press"]) + energy_mean, energy_std = stats(["TotEng", "Etotal", "Etot"]) + return { + "sample_count": len(rows), + "tail_temperature_mean_K": temp_mean, + "tail_temperature_std_K": temp_std, + "tail_pressure_mean_bar": pressure_mean, + "tail_pressure_std_bar": pressure_std, + "tail_total_energy_mean_eV": energy_mean, + "tail_total_energy_std_eV": energy_std, + } + + +def _aggregate_temperatures( + points, + expected_replicas=None, + expected_temperatures=None, +): + if expected_replicas is not None and expected_replicas < 1: + raise ValueError("expected_replicas must be >= 1") + grouped = defaultdict(list) + for point in points: + grouped[float(point["temperature_K"])].append(point) + temperatures = set(grouped) + if expected_temperatures is not None: + temperatures.update(float(value) for value in expected_temperatures) + rows = [] + for temperature in sorted(temperatures): + replicas = grouped.get(temperature, []) + velocities = np.asarray([ + row["interface_motion"]["interface_velocity_A_per_ps"] + for row in replicas + ]) + outcomes = [row["interface_motion"]["outcome"] for row in replicas] + replica_ids = [row.get("replica") for row in replicas] + replicas_complete = ( + expected_replicas is None + or ( + len(replicas) == expected_replicas + and len(set(replica_ids)) == expected_replicas + ) + ) + consensus = ( + outcomes[0] + if replicas_complete + and outcomes + and outcomes[0] in {"solid_growth", "liquid_growth"} + and len(set(outcomes)) == 1 + else "inconclusive" + ) + rows.append({ + "temperature_K": temperature, + "replica_count": len(replicas), + "expected_replica_count": expected_replicas, + "replicas_complete": replicas_complete, + "replica_outcomes": outcomes, + "consensus_outcome": consensus, + "interface_velocity_mean_A_per_ps": ( + float(np.mean(velocities)) if len(velocities) else None + ), + "interface_velocity_std_A_per_ps": float(np.std(velocities, ddof=1)) if len(velocities) > 1 else None, + "interface_velocity_standard_error_A_per_ps": float(np.std(velocities, ddof=1) / math.sqrt(len(velocities))) if len(velocities) > 1 else None, + }) + return rows + + +def _infer_bracket(rows): + solid = [row["temperature_K"] for row in rows if row["consensus_outcome"] == "solid_growth"] + liquid = [row["temperature_K"] for row in rows if row["consensus_outcome"] == "liquid_growth"] + if solid and liquid and max(solid) < min(liquid): + low, high = max(solid), min(liquid) + return { + "status": "bracketed", + "lower_solid_growth_K": low, + "upper_liquid_growth_K": high, + "width_K": high - low, + "estimated_melting_temperature_K": 0.5 * (low + high), + "uncertainty_half_width_K": 0.5 * (high - low), + "recommended_refinement_temperature_K": 0.5 * (low + high), + } + if rows and all(row["consensus_outcome"] == "solid_growth" for row in rows): + return {"status": "expand_upper_temperature"} + if rows and all(row["consensus_outcome"] == "liquid_growth" for row in rows): + return {"status": "expand_lower_temperature"} + return {"status": "inconclusive_or_unbracketed"} + + +def _write_tidy_csv(path, result): + with open(path, "w", newline="") as fp: + writer = csv.DictWriter(fp, fieldnames=["temperature_K", "replica", "outcome", "interface_velocity_A_per_ps", "q6_gap", "temperature_tail_K", "pressure_tail_bar", "energy_tail_eV"]) + writer.writeheader() + for point in result["points"]: + thermo = point["thermodynamics"] + writer.writerow({ + "temperature_K": point["temperature_K"], + "replica": point["replica"], + "outcome": point["interface_motion"]["outcome"], + "interface_velocity_A_per_ps": point["interface_motion"]["interface_velocity_A_per_ps"], + "q6_gap": point["reference_q6_gap"], + "temperature_tail_K": thermo.get("tail_temperature_mean_K"), + "pressure_tail_bar": thermo.get("tail_pressure_mean_bar"), + "energy_tail_eV": thermo.get("tail_total_energy_mean_eV"), + }) + + +def _write_plots(work_dir, result): + try: + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + except Exception: + return + points = result["points"] + if not points: + return + fig, ax = plt.subplots(figsize=(6.2, 4.2), dpi=180) + for point in points: + ax.plot(point["time_ps"], point["solid_fraction"], label=f"{point['temperature_K']:g} K r{point['replica']}") + ax.set(xlabel="Released time (ps)", ylabel="q6-normalized solid fraction") + ax.legend(fontsize=7, ncol=2) + fig.tight_layout(); fig.savefig(os.path.join(work_dir, "solid_fraction_vs_time.png")); plt.close(fig) + + fig, ax = plt.subplots(figsize=(5.6, 4.2), dpi=180) + for point in points: + motion = point["interface_motion"] + y = motion["interface_velocity_A_per_ps"] + lo = motion["interface_velocity_95pct_low_A_per_ps"] + hi = motion["interface_velocity_95pct_high_A_per_ps"] + ax.errorbar(point["temperature_K"], y, yerr=[[y - lo], [hi - y]], fmt="o", color="tab:blue", alpha=0.8) + ax.axhline(0.0, color="black", linewidth=1) + ax.set(xlabel="Temperature (K)", ylabel="Interface velocity (A/ps)") + fig.tight_layout(); fig.savefig(os.path.join(work_dir, "interface_velocity_vs_temperature.png")); plt.close(fig) + + chosen = min(points, key=lambda row: abs(row["temperature_K"] - np.mean([p["temperature_K"] for p in points]))) + snap = chosen["snapshot"] + interface_axis = chosen.get("metadata", {}).get("interface_axis", "z") + transverse_index, axis_index, xlabel, ylabel = _snapshot_projection( + interface_axis + ) + fig, axes = plt.subplots(1, 2, figsize=(9.0, 4.0), dpi=180, sharex=True, sharey=True) + for ax, coords_key, q6_key, title in [ + (axes[0], "reference_scaled_coordinates", "reference_q6", "Prepared two-phase state"), + (axes[1], "final_scaled_coordinates", "final_q6", "End of release"), + ]: + coords = np.asarray(snap[coords_key]); q6 = np.asarray(snap[q6_key]) + scatter = ax.scatter( + coords[:, transverse_index], coords[:, axis_index], + c=q6, s=3, cmap="viridis", rasterized=True, + ) + ax.set( + title=title, + xlabel=xlabel, + ylabel=ylabel, + ) + fig.colorbar(scatter, ax=axes, label="local q6", shrink=0.8) + fig.savefig(os.path.join(work_dir, "solid_liquid_interface_snapshots.png"), bbox_inches="tight"); plt.close(fig) + + +def _snapshot_projection(interface_axis): + """Return a projection that always contains the interface normal.""" + axis_index = {"x": 0, "y": 1, "z": 2}[interface_axis] + transverse_index = 0 if axis_index != 0 else 1 + axis_names = ("x", "y", "z") + return ( + transverse_index, + axis_index, + f"fractional {axis_names[transverse_index]}", + f"fractional {interface_axis}", + ) + + +def _format_report(path, result): + bracket = result["bracket"] + lines = [path, "Two-phase coexistence melting point", f"Status: {bracket['status']}"] + if bracket["status"] == "bracketed": + lines.append( + f"Tm = {bracket['estimated_melting_temperature_K']:.6g} +/- " + f"{bracket['uncertainty_half_width_K']:.6g} K " + f"({bracket['lower_solid_growth_K']:.6g}-{bracket['upper_liquid_growth_K']:.6g} K)" + ) + for row in result["temperatures"]: + lines.append(f"{row['temperature_K']:.6g} K: {row['consensus_outcome']}, replicas={row['replica_count']}") + if result.get("replica_warning"): + lines.append("WARNING: " + result["replica_warning"]) + return "\n".join(lines) + "\n" diff --git a/apex/core/property/Phonon.py b/apex/core/property/Phonon.py index f681b683..eadb2c3c 100644 --- a/apex/core/property/Phonon.py +++ b/apex/core/property/Phonon.py @@ -18,6 +18,7 @@ from apex.core.calculator.calculator import LAMMPS_INTER_TYPE from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import vasp_utils +from apex.core.calculator.lib.lammps_utils import is_deepmd_pt2 from apex.core.property.Property import Property from apex.core.refine import make_refine from apex.core.reproduce import make_repro, post_repro @@ -291,15 +292,13 @@ def __init__(self, parameter, inter_param=None): self.parameter = parameter self.inter_param = inter_param if inter_param is not None else {"type": "vasp"} - def _ensure_deepmd_plugin_loaded(self, input_text: str) -> str: - if self.inter_param.get("type") != "deepmd": - return input_text - if "plugin load" in input_text or "pair_style deepmd" not in input_text: - return input_text - return input_text.replace( - "pair_style deepmd", - "plugin load libdeepmd_lmp.so\npair_style deepmd", - 1, + @staticmethod + def _strip_legacy_deepmd_plugin(input_text: str) -> str: + """Remove plugin commands unsupported by integrated USER-DEEPMD builds.""" + return "".join( + line + for line in input_text.splitlines(keepends=True) + if line.strip() != "plugin load libdeepmd_lmp.so" ) def _build_phonolammps_run_command(self) -> str: @@ -310,7 +309,7 @@ def _build_phonolammps_run_command(self) -> str: ) if not command_template: return self._join_phonopy_arguments( - f"phonolammps in.lammps -c POSCAR --dim {dim_x} {dim_y} {dim_z}", + f"phonolammps in.lammps -c POSCAR --dim {dim_x} {dim_y} {dim_z} --logshow", primitive_axes, ) command = command_template.format( @@ -632,7 +631,7 @@ def post_process(self, task_list): del contents[pair_line_id + 1:] with open("in.lammps", 'w') as f2: - f2.write(self._ensure_deepmd_plugin_loaded("".join(contents))) + f2.write(self._strip_legacy_deepmd_plugin("".join(contents))) # dump phonolammps command phonolammps_cmd = self._build_phonolammps_run_command() with open("run_command", 'w') as f3: diff --git a/apex/core/property/Property.py b/apex/core/property/Property.py index 53a3ac03..80b347b6 100644 --- a/apex/core/property/Property.py +++ b/apex/core/property/Property.py @@ -3,13 +3,60 @@ import os from abc import ABC, abstractmethod -from monty.serialization import dumpfn +from monty.serialization import dumpfn, loadfn from apex.core.calculator.calculator import make_calculator from dflow.python import upload_packages upload_packages.append(__file__) +def _result_energies(result): + """Extract energies from a task result payload. + + Supports: + - flat dicts / mapping-like objects with top-level ``energies`` + (rehydrated dpdata LabeledSystem) + - calculator ``as_dict`` payloads with nested ``data.energies`` + """ + try: + return result["energies"] + except Exception: + pass + if not isinstance(result, dict): + return None + data = result.get("data") + if isinstance(data, dict): + return data.get("energies") + return None + + +def is_failed_task_result(result) -> bool: + """Return True when a task result cannot be used for property aggregation. + + Accepts plain dicts, calculator ``as_dict`` payloads (``data.energies``), + and mapping-like objects such as dpdata LabeledSystem (fixture + ``result_task.json`` files often load as the latter). + """ + if result is None: + return True + if isinstance(result, dict) and result.get("failed") is True: + return True + return _result_energies(result) is None + + +def _task_marked_failed(task_dir: str) -> bool: + status_path = os.path.join(task_dir, "apex_task_status.json") + if not os.path.isfile(status_path): + return False + try: + status = loadfn(status_path) + except Exception: + return True + if not isinstance(status, dict): + return True + return status.get("state") != "succeeded" or status.get("exit_code", 0) != 0 + + class Property(ABC): @abstractmethod def __init__(self, parameter): @@ -93,7 +140,9 @@ def compute(self, output_file, print_file, path_to_work): idata = json.load(fp) poscar = os.path.join(ii, "POSCAR") task = make_calculator(idata, poscar) - res = task.compute(ii) + res = None if _task_marked_failed(ii) else task.compute(ii) + if is_failed_task_result(res): + res = {"failed": True} dumpfn(res, os.path.join(ii, "result_task.json"), indent=4) # all_res.append(res) all_res.append(os.path.join(ii, "result_task.json")) diff --git a/apex/core/property/Surface.py b/apex/core/property/Surface.py index b12dc7dd..aaca709b 100644 --- a/apex/core/property/Surface.py +++ b/apex/core/property/Surface.py @@ -11,7 +11,7 @@ from apex.core.calculator.lib import abacus_utils from apex.core.calculator.lib import vasp_utils -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result from apex.core.refine import make_refine from apex.core.reproduce import make_repro, post_repro from dflow.python import upload_packages @@ -219,18 +219,21 @@ def _compute_lower(self, output_file, all_tasks, all_res): ) for ii in all_tasks: - task_result = loadfn(os.path.join(ii, "result_task.json")) - natoms = np.sum(task_result["atom_numbs"]) - epa = task_result["energies"][-1] / natoms - AA = np.linalg.norm( - np.cross(task_result["cells"][0][0], task_result["cells"][0][1]) - ) - structure_dir = os.path.basename(ii) - Cf = 1.60217657e-16 / (1e-20 * 2) * 0.001 - evac = (task_result["energies"][-1] - equi_epa * natoms) / AA * Cf miller_index = loadfn(os.path.join(ii, "miller.json")) - + task_result = loadfn(os.path.join(ii, "result_task.json")) + if is_failed_task_result(task_result): + evac = float("nan") + epa = float("nan") + else: + natoms = np.sum(task_result["atom_numbs"]) + epa = task_result["energies"][-1] / natoms + AA = np.linalg.norm( + np.cross(task_result["cells"][0][0], task_result["cells"][0][1]) + ) + Cf = 1.60217657e-16 / (1e-20 * 2) * 0.001 + evac = (task_result["energies"][-1] - equi_epa * natoms) / AA * Cf + ptr_data += "%-25s %7.3f %8.3f %8.3f\n" % ( str(miller_index) + "-" + structure_dir + ":", evac, diff --git a/apex/core/property/Vacancy.py b/apex/core/property/Vacancy.py index 7fbbc44b..2ba0727f 100644 --- a/apex/core/property/Vacancy.py +++ b/apex/core/property/Vacancy.py @@ -9,7 +9,7 @@ from pymatgen.core.structure import Structure from apex.core.calculator.lib import abacus_utils -from apex.core.property.Property import Property +from apex.core.property.Property import Property, is_failed_task_result from apex.core.refine import make_refine from apex.core.reproduce import make_repro, post_repro from dflow.python import upload_packages @@ -199,21 +199,28 @@ def _compute_lower(self, output_file, all_tasks, all_res): for idid, ii in enumerate(all_tasks): structure_dir = os.path.basename(ii) + supercell_index = loadfn(os.path.join(ii, "supercell.json")) task_result = loadfn(all_res[idid]) - natoms = sum(task_result["atom_numbs"]) - evac = task_result["energies"][-1] - equi_epa * natoms + if is_failed_task_result(task_result): + evac = float("nan") + task_energy = float("nan") + equi_energy = float("nan") + else: + natoms = sum(task_result["atom_numbs"]) + task_energy = task_result["energies"][-1] + equi_energy = equi_epa * natoms + evac = task_energy - equi_energy - supercell_index = loadfn(os.path.join(ii, "supercell.json")) ptr_data += "%s: %7.3f %7.3f %7.3f \n" % ( str(supercell_index) + "-" + structure_dir, evac, - task_result["energies"][-1], - equi_epa * natoms, + task_energy, + equi_energy, ) res_data[str(supercell_index) + "-" + structure_dir] = [ evac, - task_result["energies"][-1], - equi_epa * natoms, + task_energy, + equi_energy, ] else: diff --git a/apex/core/property/gamma_geometry.py b/apex/core/property/gamma_geometry.py new file mode 100644 index 00000000..304fc18e --- /dev/null +++ b/apex/core/property/gamma_geometry.py @@ -0,0 +1,360 @@ +"""Shared, parent-aware geometry construction for Gamma properties.""" + +from __future__ import annotations + +from dataclasses import dataclass + +import numpy as np +from dflow.python import upload_packages +from pymatgen.core import Structure +from scipy.optimize import linear_sum_assignment + +from apex.core.lib.parent_lattice_mapping import ParentSlipGeometry +from apex.core.property.gamma_slab import ( + get_first_gamma_slab, + make_gamma_slab_generator, +) + +upload_packages.append(__file__) + + +@dataclass(frozen=True) +class GammaSlabGeometry: + slab: Structure + upper_indices: np.ndarray + lower_indices: np.ndarray + metadata: dict + generation_metadata: dict + + +def validate_vacuum_size(value, property_name="Gamma") -> float: + """Return a finite, non-negative vacuum thickness.""" + + if isinstance(value, (bool, np.bool_)): + raise ValueError(f"{property_name} vacuum_size must be a finite number >= 0") + try: + vacuum_size = float(value) + except (TypeError, ValueError) as exc: + raise ValueError( + f"{property_name} vacuum_size must be a finite number >= 0" + ) from exc + if not np.isfinite(vacuum_size) or vacuum_size < 0.0: + raise ValueError(f"{property_name} vacuum_size must be a finite number >= 0") + return vacuum_size + + +def validate_gamma_cell_geometry( + structure: Structure, + *, + require_orthogonal: bool = False, + property_name: str = "Gamma", + tolerance: float = 1.0e-8, +) -> dict: + """Record the slab metric and optionally require a zero-tilt cell. + + APEX deliberately does not Gram-Schmidt a periodic cell: doing so changes + its periodic boundaries. The strict option is therefore a fail-closed + validation for workflows that require an orthogonal, Cartesian-z slab. + """ + + matrix = np.asarray(structure.lattice.matrix, dtype=float) + if matrix.shape != (3, 3) or not np.all(np.isfinite(matrix)): + raise RuntimeError(f"{property_name} slab lattice is not a finite 3x3 matrix") + lengths = np.linalg.norm(matrix, axis=1) + if np.any(lengths <= 0.0): + raise RuntimeError(f"{property_name} slab lattice contains a zero vector") + cosine_matrix = matrix @ matrix.T / np.outer(lengths, lengths) + off_diagonal = np.array( + [cosine_matrix[0, 1], cosine_matrix[0, 2], cosine_matrix[1, 2]], + dtype=float, + ) + maximum_cosine = float(np.max(np.abs(off_diagonal))) + orthogonal = bool(maximum_cosine <= tolerance) + cartesian_z_normal = bool( + np.max(np.abs(matrix[:2, 2])) <= tolerance + and np.max(np.abs(matrix[2, :2])) <= tolerance + and matrix[2, 2] > 0.0 + ) + metrics = { + "lattice_matrix_angstrom": matrix.tolist(), + "lattice_lengths_angstrom": lengths.tolist(), + "lattice_angles_degree": list(map(float, structure.lattice.angles)), + "normalized_dot_ab_ac_bc": off_diagonal.tolist(), + "maximum_abs_normalized_dot": maximum_cosine, + "orthogonal": orthogonal, + "cartesian_z_normal": cartesian_z_normal, + "zero_tilt": orthogonal and cartesian_z_normal, + "orthogonality_tolerance": float(tolerance), + "orthogonalization_applied": False, + } + if require_orthogonal and not metrics["zero_tilt"]: + raise RuntimeError( + f"{property_name} require_orthogonal_cell=true, but the generated " + "periodic slab is not an orthogonal Cartesian-z zero-tilt cell " + f"(max normalized dot={maximum_cosine:.3e}). APEX will not " + "Gram-Schmidt the cell because that changes periodic boundaries; " + "rebuild or relax an orthogonal parent bulk instead." + ) + return metrics + + +def _rotate_structure(structure: Structure, local_frame: np.ndarray) -> Structure: + lattice = np.array([local_frame @ vector for vector in structure.lattice.matrix]) + rotated = Structure( + lattice, + structure.species, + structure.frac_coords, + site_properties=structure.site_properties, + ) + if rotated.lattice.matrix[2, 2] < 0: + flipped_lattice = rotated.lattice.matrix.copy() + flipped_lattice[2] *= -1.0 + rotated = Structure( + flipped_lattice, + rotated.species, + rotated.cart_coords, + coords_are_cartesian=True, + to_unit_cell=True, + site_properties=rotated.site_properties, + ) + return rotated + + +def _select_fault_gap( + structure: Structure, + target_fraction: float, +) -> tuple[np.ndarray, np.ndarray, float, float]: + """Select a material-internal layer gap and freeze the atom grouping.""" + + frac_z = np.mod(structure.frac_coords[:, 2], 1.0) + order = np.argsort(frac_z) + values = frac_z[order] + following = np.r_[values[1:], values[0] + 1.0] + gaps = following - values + height = abs(float(structure.lattice.matrix[2, 2])) + gaps_angstrom = gaps * height + # Local alloy relaxation splits nominal layers into nearby z values. Only + # compare the pronounced inter-layer gaps when choosing the fault plane. + threshold = max(0.5, 0.4 * float(np.max(gaps_angstrom))) + candidates = np.where(gaps_angstrom >= threshold)[0] + if not len(candidates): + raise RuntimeError("Could not find a material-internal gap for Gamma") + centers = np.mod((values + following) / 2.0, 1.0) + target_fraction = float(target_fraction) % 1.0 + half_cut = len(values) // 2 - 1 + if ( + len(values) % 2 == 0 + and np.isclose(target_fraction, 0.5, atol=1e-12, rtol=0.0) + and half_cut in candidates + ): + # APEX constructs the normal Gamma slab from two equivalent material + # blocks whenever the central layer gap permits it. Prefer the exact + # half split over a nearby gap whose fractional center was shifted by + # alloy relaxation or cell skew. + cut = half_cut + else: + circular_distance = np.abs( + np.mod(centers[candidates] - target_fraction + 0.5, 1.0) - 0.5 + ) + cut = int(candidates[np.argmin(circular_distance)]) + lower = order[: cut + 1] + upper = order[cut + 1 :] + if not len(lower) or not len(upper): + # If the selected gap crosses the periodic boundary, rotate the order + # to place that gap at the end and split at the closest internal gap. + raise RuntimeError("Gamma fault split selected the periodic boundary") + return lower, upper, float(centers[cut]), float(gaps_angstrom[cut]) + + +def _center_fault_and_add_vacuum( + structure: Structure, + fault_center_fraction: float, + vacuum_size: float, +) -> Structure: + material = structure.copy() + if vacuum_size <= 0: + # In a fully periodic GSFE cell there is no physical surface. Centering + # the selected fault only changes the periodic origin and keeps the two + # interfaces easy to inspect. + shift = 0.5 - fault_center_fraction + material.translate_sites( + list(range(len(material))), + [0.0, 0.0, shift], + frac_coords=True, + to_unit_cell=True, + ) + return material + # With vacuum, do not move the fault gap to z=0.5 before opening the cell. + # For an odd number of atomic layers, the point opposite an inter-layer + # fault gap lies inside a layer; opening vacuum there would split one layer + # across the two free surfaces. The slab generator already places a real + # material gap at its periodic boundary, so retain that boundary and keep + # the independently frozen fault indices at their actual internal gap. + # Once a free-surface vacuum is present, the old bulk-periodic in-plane + # component of c is neither required nor desirable. Retaining it creates + # a strongly tilted vacuum box (often c_xy ~= half an in-plane vector), + # even though the physical surface normal is already local z. Preserve + # the complete Cartesian slab and its in-plane periodicity, but replace c + # by the pure normal repeat before wrapping. Vacuum-free GSFE cells keep + # the exact bulk-periodic c vector above. + lattice = material.lattice.matrix.copy() + material_height = abs(float(lattice[2, 2])) + lattice[2] = np.array( + [0.0, 0.0, material_height + float(vacuum_size)] + ) + coords = material.cart_coords.copy() + z_min = float(np.min(coords[:, 2])) + z_max = float(np.max(coords[:, 2])) + coords[:, 2] += ( + material_height + float(vacuum_size) - (z_max - z_min) + ) / 2.0 - z_min + return Structure( + lattice, + material.species, + coords, + coords_are_cartesian=True, + to_unit_cell=True, + site_properties=material.site_properties, + ) + + +def _anonymous_closure( + reference: Structure, + displaced: Structure, +) -> tuple[float, float]: + distances = reference.lattice.get_all_distances( + reference.frac_coords, displaced.frac_coords + ) + rows, cols = linear_sum_assignment(distances) + matched = distances[rows, cols] + return float(np.sqrt(np.mean(matched**2))), float(np.max(matched)) + + +def _chemical_closure( + reference: Structure, + displaced: Structure, +) -> tuple[float, float]: + matched_distances = [] + reference_symbols = np.array([site.specie.symbol for site in reference]) + displaced_symbols = np.array([site.specie.symbol for site in displaced]) + all_distances = reference.lattice.get_all_distances( + reference.frac_coords, displaced.frac_coords + ) + for symbol in sorted(set(reference_symbols)): + left = np.where(reference_symbols == symbol)[0] + right = np.where(displaced_symbols == symbol)[0] + if len(left) != len(right): + return float("inf"), float("inf") + rows, cols = linear_sum_assignment(all_distances[np.ix_(left, right)]) + matched_distances.extend( + all_distances[np.ix_(left, right)][rows, cols].tolist() + ) + matched = np.asarray(matched_distances) + return float(np.sqrt(np.mean(matched**2))), float(np.max(matched)) + + +def build_parent_gamma_slab( + structure: Structure, + slip_geometry: ParentSlipGeometry, + supercell_size, + plane_target: float, + min_slab_height: float | None, + vacuum_size: float, + plane_shift: float = 0.0, + require_orthogonal_cell: bool = False, +) -> GammaSlabGeometry: + """Build a fault-ready slab while retaining parent crystallography.""" + + generator, generation_metadata = make_gamma_slab_generator( + structure, + tuple(int(value) for value in slip_geometry.slab_miller), + plane_target, + min_slab_height, + ) + # RSS relaxation splits atoms belonging to one parent plane by up to a + # few tenths of an Angstrom. Cluster those sites into the same termination + # plane; an almost-zero tolerance can cut through a relaxed alloy layer. + material = get_first_gamma_slab(generator, ftol=0.2) + material = _rotate_structure(material, slip_geometry.local_frame) + material.make_supercell([int(supercell_size[0]), int(supercell_size[1]), 1]) + + a, b, c = material.lattice.matrix + material_height = abs(float(c[2])) + if material_height <= 0: + raise RuntimeError("Resolved Gamma slab has zero normal thickness") + if abs(a[2]) > 1e-7 or abs(b[2]) > 1e-7: + raise RuntimeError( + "Resolved Gamma in-plane lattice vectors are not perpendicular " + "to the fault normal" + ) + # plane_shift remains an existing API. In parent mode it selects the + # nearest material gap to a shifted fractional target; it never acts on + # the vacuum-extended cell. + target = (0.5 + float(plane_shift)) % 1.0 + lower, upper, fault_center, fault_gap = _select_fault_gap(material, target) + slab = _center_fault_and_add_vacuum(material, fault_center, vacuum_size) + + local_burgers = slip_geometry.local_frame @ slip_geometry.burgers_vector_cart + if np.linalg.norm(local_burgers[1:]) > 1e-7: + raise RuntimeError( + "Resolved Burgers vector is not aligned with the local Gamma x axis" + ) + endpoint = slab.copy() + endpoint.translate_sites( + upper, + local_burgers, + frac_coords=False, + to_unit_cell=True, + ) + geometry_rms, geometry_max = _anonymous_closure(slab, endpoint) + chemical_rms, chemical_max = _chemical_closure(slab, endpoint) + # The parent translation is topologically closed by construction. A + # chemically disordered, locally relaxed RSS is not invariant under one + # elementary parent translation, so its endpoint coordinate mismatch is a + # diagnostic rather than a valid periodicity gate. + + cell_geometry = validate_gamma_cell_geometry( + slab, + require_orthogonal=require_orthogonal_cell, + property_name="parent-aware Gamma", + ) + surface_gap = float(vacuum_size) + if vacuum_size > 0: + frac_z = np.sort(np.mod(slab.frac_coords[:, 2], 1.0)) + boundary_gap = (frac_z[0] + 1.0 - frac_z[-1]) * abs( + float(slab.lattice.matrix[2, 2]) + ) + surface_gap = float(boundary_gap) + metadata = { + "parent_aware_geometry": True, + "atom_count": len(slab), + "lower_count": int(len(lower)), + "upper_count": int(len(upper)), + "lower_indices": lower.astype(int).tolist(), + "upper_indices": upper.astype(int).tolist(), + "moved_count": int(len(upper)), + "material_height_angstrom": material_height, + "fault_gap_angstrom": fault_gap, + "fault_center_material_fraction": fault_center, + "added_vacuum_angstrom": float(vacuum_size), + "surface_gap_angstrom": surface_gap, + "interface_count": 1 if vacuum_size > 0 else 2, + "anonymous_u1_rms_angstrom": geometry_rms, + "anonymous_u1_max_angstrom": geometry_max, + "chemical_u1_rms_angstrom": chemical_rms, + "chemical_u1_max_angstrom": chemical_max, + "anonymous_u1_closed": geometry_max <= 0.65, + "chemical_u1_closed": chemical_max <= 0.2, + "parent_translation_topology_closed": True, + "local_burgers_vector": local_burgers.tolist(), + "slab_lattice": slab.lattice.matrix.tolist(), + "cell_geometry": cell_geometry, + "require_orthogonal_cell": bool(require_orthogonal_cell), + } + return GammaSlabGeometry( + slab=slab, + upper_indices=upper, + lower_indices=lower, + metadata=metadata, + generation_metadata=generation_metadata, + ) diff --git a/apex/core/property/gamma_slab.py b/apex/core/property/gamma_slab.py new file mode 100644 index 00000000..8e361d32 --- /dev/null +++ b/apex/core/property/gamma_slab.py @@ -0,0 +1,281 @@ +"""Shared slab-generation safeguards for gamma-line and gamma-surface jobs.""" + +from __future__ import annotations + +import itertools +import math +from numbers import Integral, Real + +import numpy as np +from dflow.python import upload_packages +from pymatgen.core.structure import Structure +from pymatgen.core.surface import SlabGenerator +from scipy.cluster.hierarchy import fcluster, linkage +from scipy.spatial.distance import squareform + +upload_packages.append(__file__) + + +_LAYER_TOL = 1.0e-10 + + +def _ceil_with_tolerance(value: float) -> int: + """Ceil a layer ratio without promoting round-off above an integer.""" + return int(math.ceil(float(value) - _LAYER_TOL)) + + +def validate_gamma_slab_settings( + supercell_size, + min_slab_height, + max_atoms, + min_distance, +): + """Validate and normalize user-facing gamma slab protection settings.""" + if ( + not isinstance(supercell_size, (list, tuple, np.ndarray)) + or len(supercell_size) != 3 + ): + raise ValueError("gamma supercell_size must contain three values") + + inplane = [] + for axis, value in zip(("x", "y"), supercell_size[:2]): + if ( + isinstance(value, (bool, np.bool_)) + or not isinstance(value, Integral) + or value <= 0 + ): + raise ValueError( + f"gamma supercell_size[{axis}] must be a positive integer" + ) + inplane.append(int(value)) + + plane_target = supercell_size[2] + if ( + isinstance(plane_target, (bool, np.bool_)) + or not isinstance(plane_target, Real) + or not np.isfinite(plane_target) + or plane_target <= 0 + ): + raise ValueError( + "gamma supercell_size[z] must be a positive finite number of " + "Miller-plane spacings" + ) + + if min_slab_height is not None: + if ( + isinstance(min_slab_height, (bool, np.bool_)) + or not isinstance(min_slab_height, Real) + or not np.isfinite(min_slab_height) + or min_slab_height <= 0 + ): + raise ValueError("gamma min_slab_height must be a positive number") + min_slab_height = float(min_slab_height) + + if max_atoms is not None: + if ( + isinstance(max_atoms, (bool, np.bool_)) + or not isinstance(max_atoms, Integral) + or max_atoms <= 0 + ): + raise ValueError("gamma max_atoms must be a positive integer") + max_atoms = int(max_atoms) + + if ( + isinstance(min_distance, (bool, np.bool_)) + or not isinstance(min_distance, Real) + or not np.isfinite(min_distance) + or min_distance < 0 + ): + raise ValueError("gamma min_distance must be a non-negative number") + + return ( + (inplane[0], inplane[1], float(plane_target)), + min_slab_height, + max_atoms, + float(min_distance), + ) + + +def make_gamma_slab_generator( + structure: Structure, + plane_miller, + plane_target: float, + min_slab_height: float | None, +): + """Build a Pymatgen slab generator using plane counts, not Angstrom ratios.""" + + def new_generator(target): + return SlabGenerator( + structure, + miller_index=plane_miller, + min_slab_size=target, + min_vacuum_size=0, + center_slab=True, + in_unit_planes=True, + # A three-dimensional LLL reduction can mix the slab-normal + # lattice vector into the in-plane basis. Gamma geometry is + # reduced only after the physical fault frame is known. + lll_reduce=False, + reorient_lattice=False, + primitive=False, + ) + + slab_generator = new_generator(plane_target) + d_hkl = structure.lattice.d_hkl(plane_miller) + planes_per_oriented_cell = round( + slab_generator._proj_height / d_hkl, 8 + ) + if planes_per_oriented_cell <= 0: + raise RuntimeError( + f"Cannot determine a positive layer count for Miller plane " + f"{tuple(plane_miller)}" + ) + + repeats_for_planes = _ceil_with_tolerance( + plane_target / planes_per_oriented_cell + ) + repeats_for_height = 1 + if min_slab_height is not None: + repeats_for_height = _ceil_with_tolerance( + min_slab_height / slab_generator._proj_height + ) + oriented_cell_repeats = max( + 1, repeats_for_planes, repeats_for_height + ) + + effective_plane_target = ( + oriented_cell_repeats * planes_per_oriented_cell + ) + if not np.isclose( + effective_plane_target, + plane_target, + atol=_LAYER_TOL, + rtol=0.0, + ): + slab_generator = new_generator(effective_plane_target) + + expected_base_atoms = ( + len(slab_generator.oriented_unit_cell) * oriented_cell_repeats + ) + slab_height = ( + slab_generator._proj_height * oriented_cell_repeats + ) + metadata = { + "requested_plane_spacings": float(plane_target), + "effective_plane_spacings": float(effective_plane_target), + "planes_per_oriented_cell": float(planes_per_oriented_cell), + "oriented_cell_repeats": int(oriented_cell_repeats), + "slab_height": float(slab_height), + "expected_base_atoms": int(expected_base_atoms), + } + return slab_generator, metadata + + +def get_first_gamma_slab( + slab_generator: SlabGenerator, + ftol: float = 0.001, +): + """Return the same first termination without building every SQS slab. + + APEX consumes only the first Pymatgen termination. Calling ``get_slabs`` + nevertheless materializes and structure-matches all terminations, which + can exhaust memory for large disordered parent cells. + """ + frac_coords = slab_generator.oriented_unit_cell.frac_coords + n_atoms = len(frac_coords) + if n_atoms == 1: + termination = frac_coords[0][2] + 0.5 + shift = termination - math.floor(termination) + return slab_generator.get_slab(shift=shift) + + distances = np.zeros((n_atoms, n_atoms)) + for ii, jj in itertools.combinations(range(n_atoms), 2): + z_distance = frac_coords[ii][2] - frac_coords[jj][2] + z_distance = ( + abs(z_distance - round(z_distance)) + * slab_generator._proj_height + ) + distances[ii, jj] = z_distance + distances[jj, ii] = z_distance + + clusters = fcluster( + linkage(squareform(distances)), ftol, criterion="distance" + ) + cluster_locations = { + cluster: frac_coords[index][2] + for index, cluster in enumerate(clusters) + } + locations = [ + coordinate - math.floor(coordinate) + for coordinate in sorted(cluster_locations.values()) + ] + terminations = [] + for index, location in enumerate(locations): + if index == len(locations) - 1: + termination = (locations[0] + 1 + location) * 0.5 + else: + termination = (location + locations[index + 1]) * 0.5 + terminations.append(termination - math.floor(termination)) + + return slab_generator.get_slab(shift=sorted(terminations)[0]) + + +def minimum_pair_distance(structure: Structure) -> float: + """Return the shortest periodic pair distance in a structure.""" + if len(structure) < 2: + return float("inf") + distances = structure.distance_matrix + upper = np.triu_indices(len(structure), k=1) + return float(np.min(distances[upper])) + + +def validate_generated_gamma_slab( + slab: Structure, + metadata: dict, + inplane_size, + max_atoms: int | None, + min_distance: float, + property_name: str, +): + """Stop on unexpected layer promotion, excessive size, or atom overlap.""" + expected_atoms = ( + metadata["expected_base_atoms"] + * int(inplane_size[0]) + * int(inplane_size[1]) + ) + actual_atoms = len(slab) + if actual_atoms != expected_atoms: + raise RuntimeError( + f"{property_name} generated {actual_atoms} atoms, but " + f"{expected_atoms} were expected from " + f"{metadata['oriented_cell_repeats']} oriented-cell repeat(s). " + "This may indicate an unintended slab-layer promotion." + ) + if max_atoms is not None and actual_atoms > max_atoms: + raise RuntimeError( + f"{property_name} generated {actual_atoms} atoms, exceeding " + f"max_atoms={max_atoms}. Increase max_atoms only after checking " + "the slab thickness and Miller indices." + ) + + pair_distance = minimum_pair_distance(slab) + if pair_distance < min_distance: + overlap_message = ( + "Generated Gamma surface contains overlapping atoms." + if property_name.startswith("GammaSurface") + else "Generated Gamma line contains overlapping atoms." + ) + raise RuntimeError( + f"{overlap_message} {property_name} minimum pair " + f"distance {pair_distance:.6f} A is below " + f"min_distance={min_distance:.6f} A." + ) + + result = dict(metadata) + result.update( + { + "atom_count": int(actual_atoms), + "minimum_pair_distance": float(pair_distance), + } + ) + return result diff --git a/apex/core/structure.py b/apex/core/structure.py index af53f08a..833eae67 100644 --- a/apex/core/structure.py +++ b/apex/core/structure.py @@ -4,6 +4,30 @@ upload_packages.append(__file__) + +SUPPORTED_PARENT_LATTICE_HINTS = frozenset({"bcc", "fcc", "hcp"}) + + +def normalize_parent_lattice_hint(value): + """Validate an explicit parent-lattice hint for disordered structures.""" + if value is None: + return None + if not isinstance(value, str): + raise ValueError("parent_lattice must be one of: bcc, fcc, hcp") + normalized = value.strip().lower() + if normalized not in SUPPORTED_PARENT_LATTICE_HINTS: + allowed = ", ".join(sorted(SUPPORTED_PARENT_LATTICE_HINTS)) + raise ValueError(f"parent_lattice must be one of: {allowed}") + return normalized + + +def resolve_parent_lattice_hint(detected_structure_type, parent_lattice=None): + """Return the effective lattice family and its provenance.""" + hint = normalize_parent_lattice_hint(parent_lattice) + if hint is None: + return detected_structure_type, "auto_detected" + return hint, "user_override" + class StructureInfo(object): """Analyze structure type Arg: diff --git a/apex/default_config/abacus/param_props.json b/apex/default_config/abacus/param_props.json index 0435790b..0f785b85 100755 --- a/apex/default_config/abacus/param_props.json +++ b/apex/default_config/abacus/param_props.json @@ -38,21 +38,28 @@ "slip_length": 1 }, "supercell_size": [1,1,20], - "vacuum_size": 15, + "vacuum_size": 20, "add_fix": ["true","true","false"], "n_steps": 20 }, - { - "type": "gamma_surface", - "req_calc": false, - "plane_miller": [1,1,0], - "slip_direction": [1,-1,-1], - "supercell_size": [1,1,20], - "vacuum_size": 15, - "closed_loop": false, - "add_fix": ["true","true","false"], - "n_steps_x": 20, - "n_steps_y": 20 + { + "type": "gamma_surface", + "req_calc": false, + "plane_miller": [1,1,1], + "slip_direction": [-1,1,0], + "bcc": { + "plane_miller": [1,1,0], + "slip_direction": [-1,1,1] + }, + "hcp": { + "plane_miller": [0,0,0,1], + "slip_direction": [2,-1,-1,0] + }, + "supercell_size": [1,1,20], + "closed_loop": true, + "add_fix": ["true","true","false"], + "n_steps_x": 20, + "n_steps_y": 20 }, { "type": "phonon", diff --git a/apex/default_config/lammps/param_props.json b/apex/default_config/lammps/param_props.json index 7123ecba..c9c13d93 100755 --- a/apex/default_config/lammps/param_props.json +++ b/apex/default_config/lammps/param_props.json @@ -40,6 +40,37 @@ "tdamp": 0.1, "pdamp": 1.0} }, + { + "type": "melting_point", + "req_calc": false, + "method": "two_phase", + "supercell_size": [1, 1, 2], + "cal_setting": { + "temperature": [1500, 1600, 1700], + "premelt_temperature": 4500, + "premelt_steps": 5000, + "conditioning_steps": 5000, + "production_steps": 100000, + "timestep": 0.001, + "tdamp": 0.1, + "pdamp": 1.0, + "pressure": 0.0, + "barostat": "iso", + "interface_axis": "z", + "liquid_fraction": 0.5, + "dump_step": 100, + "thermo_step": 100, + "restart_interval": 10000, + "q6_cutoff": 3.5, + "q6_neighbors": 12, + "replicas": 1, + "velocity_seeds": { + "premelt": 324159, + "condition": 271828, + "release": 161803 + } + } + }, { "type": "elastic", "req_calc": false, @@ -92,21 +123,29 @@ "plane_miller": [1,1,1], "slip_direction": [1,1,-2], "supercell_size": [2,2,100], - "vacuum_size": 15, + "vacuum_size": 20, "add_fix": ["true","true","false"], "n_steps": 10 }, - { - "type": "gamma_surface", - "req_calc": false, - "plane_miller": [1,1,0], - "slip_direction": [1,-1,-1], - "supercell_size": [1,1,20], - "vacuum_size": 15, - "closed_loop": false, - "add_fix": ["true","true","false"], - "n_steps_x": 20, - "n_steps_y": 20 + { + "type": "gamma_surface", + "req_calc": false, + "plane_miller": [1,1,1], + "slip_direction": [-1,1,0], + "bcc": { + "plane_miller": [1,1,0], + "slip_direction": [-1,1,1] + }, + "hcp": { + "plane_miller": [0,0,0,1], + "slip_direction": [2,-1,-1,0] + }, + "supercell_size": [1,1,20], + "vacuum_size": 20, + "closed_loop": true, + "add_fix": ["true","true","false"], + "n_steps_x": 20, + "n_steps_y": 20 } ] } diff --git a/apex/default_config/vasp/param_props.json b/apex/default_config/vasp/param_props.json index 8d9fee19..504a499e 100755 --- a/apex/default_config/vasp/param_props.json +++ b/apex/default_config/vasp/param_props.json @@ -51,22 +51,82 @@ "slip_length": 1 }, "supercell_size": [1,1,20], - "vacuum_size": 15, + "vacuum_size": 20, "add_fix": ["true","true","false"], "n_steps": 20 }, - { - "type": "gamma_surface", - "req_calc": false, - "plane_miller": [1,1,0], - "slip_direction": [1,-1,-1], - "supercell_size": [1,1,20], - "vacuum_size": 15, - "closed_loop": false, - "add_fix": ["true","true","false"], - "n_steps_x": 20, - "n_steps_y": 20 + { + "type": "gamma_surface", + "req_calc": false, + "plane_miller": [1,1,1], + "slip_direction": [-1,1,0], + "bcc": { + "plane_miller": [1,1,0], + "slip_direction": [-1,1,1] + }, + "hcp": { + "plane_miller": [0,0,0,1], + "slip_direction": [2,-1,-1,0] + }, + "supercell_size": [1,1,20], + "vacuum_size": 20, + "closed_loop": true, + "add_fix": ["true","true","false"], + "n_steps_x": 20, + "n_steps_y": 20 }, + { + "type": "finite_t_latt", + "req_calc": false, + "supercell_size": [2, 2, 2], + "cal_setting": { + "temperature": [300, 500, 700, 900, 1100, 1300, 1500], + "equi_step": 5000, + "ave_step": 10000, + "timestep_fs": 1.0, + "pressure_kbar": 0.0, + "langevin_gamma": 10.0, + "langevin_gamma_l": 10.0, + "pmass": 1000.0 + } + }, + { + "type": "annealing", + "req_calc": false, + "protocol": "ramp_cool", + "supercell_size": [2, 2, 2], + "cal_setting": { + "start_temp": 300, + "target_temp": 900, + "end_temp": 300, + "equi_step": 100, + "ramp_step": 200, + "cool_step": 200, + "hold_step": 100, + "timestep_fs": 1.0, + "pressure_kbar": 0.0, + "langevin_gamma": 10.0, + "langevin_gamma_l": 10.0, + "pmass": 1000.0 + } + }, + { + "type": "annealing", + "suffix": "coexistence", + "req_calc": false, + "protocol": "coexistence", + "supercell_size": [2, 2, 2], + "cal_setting": { + "target_temp": 900, + "equi_step": 5000, + "production_step": 10000, + "timestep_fs": 1.0, + "pressure_kbar": 0.0, + "langevin_gamma": 10.0, + "langevin_gamma_l": 10.0, + "pmass": 1000.0 + } + }, { "type": "phonon", "req_calc": false, diff --git a/apex/flow.py b/apex/flow.py index cd5e70d6..c6d4b3be 100644 --- a/apex/flow.py +++ b/apex/flow.py @@ -229,11 +229,39 @@ def _failure_log_excerpt(paths, max_lines: int = 80, max_chars: int = 12000) -> break if sum(len(chunk) for chunk in chunks) >= max_chars: break - result = "\n".join(chunks) + result = FlowGenerator._redact_log_secrets("\n".join(chunks)) if len(result) > max_chars: result = "\n" + result[-max_chars:] return result + @staticmethod + def _redact_log_secrets(text: str) -> str: + def redact_value(match): + value = match.group(2) + quote = ( + value[0] + if len(value) >= 2 + and value[0] in {'"', "'"} + and value[-1] == value[0] + else "" + ) + return f"{match.group(1)}{quote}[REDACTED]{quote}" + + authorization = re.compile( + r'(?im)(\bauthorization\b["\']?\s*[:=]\s*)' + r'("[^"]*"|\'[^\']*\'|(?:bearer\s+)?[^\s,;]+)' + ) + text = authorization.sub(redact_value, text) + secret_key = re.compile( + r'(?i)(\b(?:access[_-]?key|app[_-]?key|password|ticket)\b["\']?\s*[:=]\s*)' + r'("[^"]*"|\'[^\']*\'|[^\s,;&}\]]+)' + ) + text = secret_key.sub(redact_value, text) + env_secret = re.compile( + r'(?i)(\b(?:BOHRIUM_ACCESS_KEY|BOHRIUM_APP_KEY|BOHR_TICKET)=)([^\s]+)' + ) + return env_secret.sub(r'\1[REDACTED]', text) + @staticmethod def _failure_cause_from_excerpt(excerpt: str) -> str: if not excerpt: diff --git a/apex/gui.py b/apex/gui.py index 47cb3a71..2c8d3d40 100644 --- a/apex/gui.py +++ b/apex/gui.py @@ -747,6 +747,7 @@ def _load_account_state(account_path: Optional[str] = None) -> Dict[str, Any]: "email": str(merged.get("email") or ""), "program_id": str(program_id) if program_id not in (None, "") else "", "password_set": bool(merged.get("password")), + "access_key_set": bool(merged.get("access_key")), } @@ -754,6 +755,7 @@ def _render_account_summary(account_state: Dict[str, Any]) -> str: email = account_state.get("email") or "(未设置)" program_id = account_state.get("program_id") or "(未设置)" password_status = "已设置 (隐藏)" if account_state.get("password_set") else "未设置" + access_key_status = "已设置 (隐藏)" if account_state.get("access_key_set") else "未设置" config_path = account_state.get("path") or "(unknown)" return "\n".join( [ @@ -761,6 +763,7 @@ def _render_account_summary(account_state: Dict[str, Any]) -> str: f"Email: {email}", f"Program ID: {program_id}", f"Password: {password_status}", + f"AccessKey: {access_key_status}", ] ) @@ -778,6 +781,7 @@ def _save_account_overwrite( email: str, password: str, program_id_text: str, + access_key: str = "", account_path: Optional[str] = None, ) -> Tuple[Dict[str, Any], Dict[str, Any]]: path_obj = get_account_config_path(account_path) @@ -790,6 +794,7 @@ def _save_account_overwrite( clean_email = (email or "").strip() clean_password = (password or "").strip() clean_program_id = (program_id_text or "").strip() + clean_access_key = (access_key or "").strip() if clean_email: merged["email"] = clean_email @@ -797,6 +802,9 @@ def _save_account_overwrite( if clean_password: merged["password"] = clean_password updates_applied.append("password") + if clean_access_key: + merged["access_key"] = clean_access_key + updates_applied.append("access_key") if clean_program_id: try: merged["program_id"] = int(clean_program_id) @@ -2780,7 +2788,7 @@ def _build_account_tab() -> dbc.Tab: return dbc.Tab( label="Account", children=[ - html.P("底层对应 `apex account`,密码仅支持覆盖保存,不会在界面显示。", className="text-muted"), + html.P("底层对应 `apex account`,密码和 AccessKey 仅支持覆盖保存,不会在界面显示。", className="text-muted"), dbc.Row( [ dbc.Col( @@ -2807,6 +2815,14 @@ def _build_account_tab() -> dbc.Tab: placeholder="输入新密码以覆盖", ), html.Br(), + dbc.Label("AccessKey (留空表示保持当前 AccessKey 不变)"), + dbc.Input( + id="account-access-key", + type="password", + value="", + placeholder="输入新 AccessKey 以覆盖", + ), + html.Br(), dbc.Button("刷新", id="account-refresh", color="secondary", className="me-2"), dbc.Button("覆盖保存", id="account-save", color="primary"), ], @@ -3635,6 +3651,7 @@ def _refresh_retrieve_progress(_n_intervals, retrieve_state, submit_workdir, sub Output("account-email", "value"), Output("account-program-id", "value"), Output("account-password", "value"), + Output("account-access-key", "value"), Output("account-summary", "children"), Output("account-feedback", "children"), Input("account-refresh", "n_clicks"), @@ -3642,15 +3659,20 @@ def _refresh_retrieve_progress(_n_intervals, retrieve_state, submit_workdir, sub State("account-email", "value"), State("account-program-id", "value"), State("account-password", "value"), + State("account-access-key", "value"), prevent_initial_call=True, ) - def _handle_account(_refresh_clicks, _save_clicks, email_value, program_id_value, password_value): + def _handle_account( + _refresh_clicks, _save_clicks, email_value, program_id_value, + password_value, access_key_value + ): triggered_id = _resolve_triggered_id() if triggered_id == "account-save": feedback, account_state = _save_account_overwrite( email=email_value or "", password=password_value or "", program_id_text=program_id_value or "", + access_key=access_key_value or "", ) else: account_state = _load_account_state() @@ -3659,6 +3681,7 @@ def _handle_account(_refresh_clicks, _save_clicks, email_value, program_id_value account_state.get("email", ""), account_state.get("program_id", ""), "", + "", _render_account_summary(account_state), _brief_feedback(feedback), ) diff --git a/apex/main.py b/apex/main.py index 8fc4b753..02dbcc1e 100644 --- a/apex/main.py +++ b/apex/main.py @@ -19,10 +19,6 @@ __version__, ) from apex.config import Config -from apex.step import do_step_from_args -from apex.submit import submit_from_args -from apex.archive import archive_from_args -from apex.report import report_from_args from apex.utils import load_config_file from apex.task_failure import classify_apex_task_status @@ -552,6 +548,17 @@ def parse_args(): default=0.0, help="Shift the rendered viewport vertically by a fraction of the data span; positive values move the structure downward", ) + parser_preview.add_argument( + "--gif-view", + choices=("auto", "default", "slip-plane", "parent-bc", "both"), + default="auto", + help=( + "Gamma projection: auto writes both scientific views for gamma " + "and gamma_surface; alternatively preserve the legacy Cartesian " + "view, look normal to the slip plane, look normal to the parent " + "bc plane, or explicitly write both views" + ), + ) ########################################## # GUI @@ -612,6 +619,17 @@ def parse_args(): parser_account.add_argument("--context-type", dest="context_type", type=str, default=None) parser_account.add_argument("--email", type=str, default=None) parser_account.add_argument("--password", type=str, default=None) + parser_account.add_argument("--access-key", dest="access_key", type=str, default=None) + parser_account.add_argument( + "--clear", + nargs="?", + const="all", + choices=("all", "access-key", "email"), + help=( + "Clear both login methods when used alone, or clear only " + "'access-key' or the email/password pair" + ), + ) parser_account.add_argument("--program-id", dest="program_id", type=int, default=None) parser_account.add_argument("--apex-image-name", dest="apex_image_name", type=str, default=None) @@ -619,13 +637,16 @@ def parse_args(): # Agent skill parser_skill = subparsers.add_parser( "skill", - help="Print an Agent prompt for installing apex-flow, or build a zip file for uploading to MatMaster", + help=( + "Print the local Agent installation prompt, or build the separate " + "Bohrium Cloud/MatMaster skill zip" + ), formatter_class=argparse.ArgumentDefaultsHelpFormatter, ) parser_skill.add_argument( "--zip", action="store_true", - help="Write a zip of the bundled apex-flow", + help="Write the Bohrium Cloud/MatMaster apex-flow zip", ) parser_skill.add_argument( "-o", "--output", @@ -1274,6 +1295,8 @@ def main(): # parse args parser, args = parse_args() if args.cmd == 'submit': + from apex.submit import submit_from_args + header() try: submit_from_args( @@ -1518,6 +1541,8 @@ def main(): f"under {os.path.join(work_dir, '.failed-artifacts')}" ) elif args.cmd == 'do': + from apex.step import do_step_from_args + header() do_step_from_args( parameter=args.parameter, @@ -1525,6 +1550,8 @@ def main(): step=args.step ) elif args.cmd == 'archive': + from apex.archive import archive_from_args + archive_from_args( parameters=args.json, config_file=args.config, @@ -1536,6 +1563,8 @@ def main(): is_result=args.result ) elif args.cmd == 'report': + from apex.report import report_from_args + header() report_from_args( config_file=args.config, @@ -1564,7 +1593,8 @@ def main(): elif args.cmd == 'preview': from apex.preview import preview_from_args - preview_from_args(args) + for output_path in preview_from_args(args): + print(output_path) elif args.cmd == 'skill': from apex.skill import skill_from_args diff --git a/apex/op/RunLAMMPS.py b/apex/op/RunLAMMPS.py index 44c6c65f..908db901 100644 --- a/apex/op/RunLAMMPS.py +++ b/apex/op/RunLAMMPS.py @@ -12,6 +12,7 @@ ) from apex.task_failure import ( HEADER_ONLY_RETRY_REASON, + TRANSIENT_LAMMPS_RETRY_REASON, classify_lammps_exit_code, is_header_only_lammps_failure, is_lammps_header_only_log, @@ -53,6 +54,9 @@ def _cleanup_model_links(cls, task_dir): logging.warning(f"Failed to load inter.json for symlink cleanup: {exc}") return + if inter_param.get("model_in_image") is True: + return + model_spec = inter_param.get("model", []) if isinstance(model_spec, str): model_list = [model_spec] @@ -269,8 +273,18 @@ def _archive_retry_file(cls, path: Path, attempt: int): @classmethod def _prepare_retry(cls, task_dir: Path, attempt: int): - for name in ["log.lammps", "outlog", "errlog", "run.log", "dump.relax", "stress_timeseries.txt"]: + for name in [ + "log.lammps", + "outlog", + "errlog", + "run.log", + "dump.relax", + "dump.melting", + "stress_timeseries.txt", + ]: cls._archive_retry_file(task_dir / name, attempt) + for path in task_dir.glob("restart.melting.*"): + cls._archive_retry_file(path, attempt) @classmethod def _resource_snapshot(cls) -> str: @@ -392,6 +406,35 @@ def execute(self, op_in: OPIO) -> OPIO: time.sleep(retry_delay) attempts += 1 exit_code = self._run_command(cmd, task_dir) + transient_retries = max( + 0, + self._runtime_int_option( + cmd, + "APEX_LAMMPS_TRANSIENT_RETRY", + 1, + ), + ) + transient_attempts = 0 + while ( + exit_code != 0 + and transient_attempts < transient_retries + and classify_lammps_exit_code(exit_code).get("reason") + in {"killed_or_oom", "terminated", "timeout"} + ): + classification = classify_lammps_exit_code(exit_code)["reason"] + retry_reason = TRANSIENT_LAMMPS_RETRY_REASON + self._append_debug( + debug_file, + f"\n## Retry {attempts + 1}\n" + f"Retrying transient LAMMPS failure because " + f"exit_code={exit_code} and " + f"classification={classification}.", + ) + self._prepare_retry(task_dir, attempts) + time.sleep(retry_delay) + attempts += 1 + transient_attempts += 1 + exit_code = self._run_command(cmd, task_dir) elapsed = time.time() - start finished_at = self._utc_now() self._write_task_status( diff --git a/apex/op/RunVASP.py b/apex/op/RunVASP.py new file mode 100644 index 00000000..7f10a516 --- /dev/null +++ b/apex/op/RunVASP.py @@ -0,0 +1,651 @@ +"""APEX-owned VASP execution OP. + +The upstream fpop RunVasp OP only stages the four conventional VASP input +files. APEX finite-temperature properties also generate ``run_command`` and +multiple ``INCAR.`` files, so those tasks need the complete prepared +input set in a writable working directory. +""" + +import datetime +import json +import logging +import re +import shutil +from pathlib import Path +from typing import Dict, Iterable, List, Optional + +from dflow.python import FatalError, OP, OPIO, TransientError, upload_packages +from dflow.utils import set_directory +from fpop.vasp import RunVasp + +upload_packages.append(__file__) + + +class RunVASP(RunVasp): + """Run VASP while preserving APEX staged-calculation semantics.""" + + _MANDATORY_INPUTS = ("POSCAR", "INCAR", "POTCAR", "KPOINTS") + _STAGE_PLAN = "apex_vasp_stage_plan.json" + _STAGE_STATUS = "apex_vasp_stage_status.json" + _FAILURE_STATUS = "apex_vasp_failure.json" + _EVIDENCE_FILES = ( + "outlog", + "OUTCAR", + "OSZICAR", + "CONTCAR", + "XDATCAR", + "POSCAR", + "INCAR", + "KPOINTS", + "run_command", + "task.json", + "FiniteTlatt.json", + "Annealing.json", + _STAGE_PLAN, + _STAGE_STATUS, + _FAILURE_STATUS, + ) + _EVIDENCE_GLOBS = ( + "OUTCAR.*", + "OSZICAR.*", + "CONTCAR.*", + "XDATCAR.*", + "INCAR.*", + ) + _IONIC_POSITION = re.compile( + r"(?m)^\s*POSITION\s+TOTAL-FORCE(?:\s|$)", re.IGNORECASE + ) + _IONIC_ITERATION = re.compile( + r"(?m)^-*\s*Iteration\s+(\d+)\s*\(", re.IGNORECASE + ) + _OSZICAR_MD_STEP = re.compile(r"(?m)^\s*(\d+)\s+T=") + _FOOTER_MARKERS = { + "general_timing": re.compile( + r"General timing and accounting", re.IGNORECASE + ), + "total_cpu_time": re.compile( + r"Total CPU time used", re.IGNORECASE + ), + "elapsed_time": re.compile(r"Elapsed time", re.IGNORECASE), + } + _FOOTER_TAIL_LINES = 256 + + @staticmethod + def _copy_path(source: Path, destination: Path) -> None: + destination.parent.mkdir(parents=True, exist_ok=True) + if source.is_dir(): + if destination.exists(): + shutil.rmtree(destination) + shutil.copytree(source, destination, symlinks=False) + else: + # Follow task-input symlinks so INCAR and POSCAR become writable. + shutil.copy2(source, destination, follow_symlinks=True) + + @classmethod + def _prepared_inputs( + cls, + task_path: Path, + backward_dir_name: str, + log_name: str, + ) -> Iterable[Path]: + excluded = { + backward_dir_name, + log_name, + cls._FAILURE_STATUS, + cls._STAGE_STATUS, + } + output_prefixes = ("OUTCAR", "OSZICAR", "CONTCAR", "XDATCAR") + for source in task_path.iterdir(): + # Never let stale completion evidence from an earlier attempt make + # a partial rerun appear successful. + if source.name in excluded: + continue + if source.name in output_prefixes or source.name.startswith( + tuple(f"{prefix}." for prefix in output_prefixes) + ): + continue + yield source + + @staticmethod + def _incar_nsw(path: Path) -> Optional[int]: + if not path.is_file(): + return None + matches = re.findall( + r"(?im)^\s*NSW\s*=\s*([+-]?\d+)", path.read_text(errors="replace") + ) + return int(matches[-1]) if matches else None + + @classmethod + def _oszicar_steps(cls, path: Path) -> int: + if not path.is_file(): + return 0 + text = path.read_text(errors="replace") + matches = [int(value) for value in cls._OSZICAR_MD_STEP.findall(text)] + return max(matches, default=0) + + @classmethod + def _inspect_outcar_text( + cls, + text: str, + *, + expected_ionic_steps: Optional[int], + require_exact_steps: bool, + oszicar_path: Optional[Path] = None, + ) -> Dict: + positions = len(cls._IONIC_POSITION.findall(text)) + iterations = [ + int(value) for value in cls._IONIC_ITERATION.findall(text) + ] + footer_tail = "\n".join( + text.splitlines()[-cls._FOOTER_TAIL_LINES:] + ) + footer_markers = { + name: bool(pattern.search(footer_tail)) + for name, pattern in cls._FOOTER_MARKERS.items() + } + footer_complete = all(footer_markers.values()) + failure_reasons = [] + if not footer_complete: + missing = [ + name for name, present in footer_markers.items() if not present + ] + failure_reasons.append( + "missing_footer_markers:" + ",".join(missing) + ) + if require_exact_steps: + if expected_ionic_steps is None: + failure_reasons.append("missing_expected_ionic_steps") + elif positions != expected_ionic_steps: + failure_reasons.append( + "ionic_step_count_mismatch:" + f"expected={expected_ionic_steps},observed={positions}" + ) + return { + "expected_ionic_steps": expected_ionic_steps, + "observed_ionic_steps": positions, + "observed_iteration_max": max(iterations, default=0), + "observed_oszicar_steps": ( + cls._oszicar_steps(oszicar_path) + if oszicar_path is not None + else 0 + ), + "footer_complete": footer_complete, + "footer_markers": footer_markers, + "footer_tail_lines_checked": cls._FOOTER_TAIL_LINES, + "finished": not failure_reasons, + "failure_reasons": failure_reasons, + } + + @classmethod + def _inspect_outcar( + cls, + path: Path, + *, + expected_ionic_steps: Optional[int], + require_exact_steps: bool, + oszicar_path: Optional[Path] = None, + ) -> Dict: + if not path.is_file(): + return { + "expected_ionic_steps": expected_ionic_steps, + "observed_ionic_steps": 0, + "observed_iteration_max": 0, + "observed_oszicar_steps": ( + cls._oszicar_steps(oszicar_path) + if oszicar_path is not None + else 0 + ), + "footer_complete": False, + "footer_markers": { + name: False for name in cls._FOOTER_MARKERS + }, + "finished": False, + "failure_reasons": ["missing_outcar"], + } + return cls._inspect_outcar_text( + path.read_text(errors="replace"), + expected_ionic_steps=expected_ionic_steps, + require_exact_steps=require_exact_steps, + oszicar_path=oszicar_path, + ) + + @staticmethod + def _task_type() -> str: + task_json = Path("task.json") + if task_json.is_file(): + try: + payload = json.loads(task_json.read_text()) + except (OSError, ValueError): + payload = {} + if isinstance(payload, dict): + return str(payload.get("type", "")) + if Path("FiniteTlatt.json").is_file(): + return "finite_t_latt" + if Path("Annealing.json").is_file(): + return "annealing" + return "" + + @staticmethod + def _annealing_stage_names() -> List[str]: + command_path = Path("run_command") + if not command_path.is_file(): + return [] + stages = [] + for line in command_path.read_text(errors="replace").splitlines(): + if "OUTCAR.apex" not in line: + continue + match = re.search(r"APEX_STAGE\s+([A-Za-z0-9_.-]+)", line) + if match and match.group(1) not in stages: + stages.append(match.group(1)) + return stages + + @classmethod + def _load_stage_plan(cls) -> List[Dict]: + path = Path(cls._STAGE_PLAN) + if not path.is_file(): + return [] + try: + payload = json.loads(path.read_text()) + except (OSError, ValueError): + return [] + stages = payload.get("stages", []) if isinstance(payload, dict) else [] + return [stage for stage in stages if isinstance(stage, dict)] + + @classmethod + def _fallback_finite_t_latt_plan(cls) -> List[Dict]: + command = ( + Path("run_command").read_text(errors="replace") + if Path("run_command").is_file() + else "" + ) + stages = [] + if "OUTCAR.nvt" in command: + stages.append({ + "name": "nvt", + "incar": "INCAR.nvt", + "outcar": "OUTCAR.nvt", + "oszicar": "OSZICAR.nvt", + }) + stages.extend([ + { + "name": "equi", + "incar": "INCAR.equi", + "outcar": "OUTCAR.equi", + "oszicar": "OSZICAR.equi", + }, + { + "name": "production", + "incar": "INCAR.production", + "outcar": "OUTCAR", + "oszicar": "OSZICAR", + }, + ]) + return stages + + @classmethod + def _fallback_annealing_plan(cls) -> List[Dict]: + return [ + { + "name": name, + "incar": f"INCAR.{name}", + "outcar": "OUTCAR", + "oszicar": "OSZICAR", + "section": name, + } + for name in cls._annealing_stage_names() + ] + + @staticmethod + def _stage_sections(text: str) -> Dict[str, str]: + markers = list( + re.finditer(r"(?m)^APEX_STAGE\s+([A-Za-z0-9_.-]+)\s*$", text) + ) + sections = {} + for index, marker in enumerate(markers): + start = marker.end() + end = ( + markers[index + 1].start() + if index + 1 < len(markers) + else len(text) + ) + sections[marker.group(1)] = text[start:end] + return sections + + @classmethod + def _validate_stages(cls) -> Dict: + task_type = cls._task_type() + status = { + "schema": "apex.vasp.stage-status/v1", + "task_type": task_type or "single_stage", + "checked_at": datetime.datetime.now( + datetime.timezone.utc + ).isoformat(), + "stages": [], + } + + if task_type in {"finite_t_latt", "annealing"}: + plan = cls._load_stage_plan() + if not plan: + plan = ( + cls._fallback_finite_t_latt_plan() + if task_type == "finite_t_latt" + else cls._fallback_annealing_plan() + ) + if not plan: + status["state"] = "failed" + status["missing_or_incomplete_stages"] = ["stage_plan"] + status["failure_reasons"] = ["missing_stage_plan"] + return status + + section_cache = {} + for stage in plan: + name = str(stage.get("name", "unknown")) + incar = Path(str(stage.get("incar", "INCAR"))) + outcar = Path(str(stage.get("outcar", "OUTCAR"))) + oszicar = Path(str(stage.get("oszicar", "OSZICAR"))) + expected = stage.get("expected_ionic_steps") + if expected is None: + expected = cls._incar_nsw(incar) + else: + expected = int(expected) + + section_name = stage.get("section") + if section_name: + if outcar not in section_cache: + text = ( + outcar.read_text(errors="replace") + if outcar.is_file() + else "" + ) + section_cache[outcar] = cls._stage_sections(text) + section = section_cache[outcar].get(str(section_name)) + if section is None: + inspection = cls._inspect_outcar( + Path("__missing_stage_section__"), + expected_ionic_steps=expected, + require_exact_steps=True, + oszicar_path=oszicar, + ) + inspection["failure_reasons"] = [ + f"missing_stage_section:{section_name}" + ] + else: + inspection = cls._inspect_outcar_text( + section, + expected_ionic_steps=expected, + require_exact_steps=True, + oszicar_path=oszicar, + ) + else: + inspection = cls._inspect_outcar( + outcar, + expected_ionic_steps=expected, + require_exact_steps=True, + oszicar_path=oszicar, + ) + inspection.update({ + "name": name, + "incar": str(incar), + "output": str(outcar), + "oszicar": str(oszicar), + }) + status["stages"].append(inspection) + else: + inspection = cls._inspect_outcar( + Path("OUTCAR"), + expected_ionic_steps=cls._incar_nsw(Path("INCAR")), + require_exact_steps=False, + oszicar_path=Path("OSZICAR"), + ) + inspection.update({ + "name": "vasp", + "incar": "INCAR", + "output": "OUTCAR", + "oszicar": "OSZICAR", + }) + status["stages"] = [inspection] + + missing = [ + item["name"] for item in status["stages"] if not item["finished"] + ] + status["state"] = "failed" if missing else "succeeded" + if missing: + status["missing_or_incomplete_stages"] = missing + status["failure_reasons"] = { + item["name"]: item.get("failure_reasons", []) + for item in status["stages"] + if not item["finished"] + } + return status + + def check_run_success(self): + """Override fpop's fragile last-line ``Voluntary`` check.""" + return self._validate_stages()["state"] == "succeeded" + + @classmethod + def _save_stage_evidence(cls, backward_dir_name: str, status: Dict) -> None: + status_path = Path(cls._STAGE_STATUS) + status_path.write_text(json.dumps(status, indent=4) + "\n") + backward_dir = Path(backward_dir_name) + backward_dir.mkdir(parents=True, exist_ok=True) + shutil.copy2(status_path, backward_dir / status_path.name) + for source in cls._evidence_sources(): + if source.name in {status_path.name, backward_dir.name}: + continue + if source.is_file(): + shutil.copy2(source, backward_dir / source.name) + + @classmethod + def _evidence_sources(cls, log_name: Optional[str] = None) -> List[Path]: + names = list(cls._EVIDENCE_FILES) + if log_name and log_name not in names: + names.append(log_name) + sources = [Path(name) for name in names] + for pattern in cls._EVIDENCE_GLOBS: + sources.extend(Path(".").glob(pattern)) + unique = {} + for source in sources: + unique[source.name] = source + return list(unique.values()) + + @classmethod + def _current_stage(cls, status: Dict) -> str: + for stage in status.get("stages", []): + if not stage.get("finished"): + return str(stage.get("name", "unknown")) + return "unknown" + + @classmethod + def _save_failure_evidence( + cls, + destination: Path, + error: Exception, + log_name: str, + ) -> None: + destination.mkdir(parents=True, exist_ok=True) + status = cls._validate_stages() + status["state"] = "failed" + status["current_or_next_stage"] = cls._current_stage(status) + status["execution_error_type"] = type(error).__name__ + status["execution_error"] = str(error) + status_path = Path(cls._STAGE_STATUS) + status_path.write_text(json.dumps(status, indent=4) + "\n") + + failure = { + "schema": "apex.vasp.failure/v1", + "task_type": status.get("task_type"), + "current_or_next_stage": status["current_or_next_stage"], + "error_type": type(error).__name__, + "error": str(error), + "stage_status": status, + } + failure_path = Path(cls._FAILURE_STATUS) + failure_path.write_text(json.dumps(failure, indent=4) + "\n") + + for source in cls._evidence_sources(log_name): + if source.is_file(): + shutil.copy2(source, destination / source.name) + + @staticmethod + def _dflow_tmp_roots() -> List[Path]: + """Find PythonOP temporary roots from explicit and real cloud layouts.""" + roots = [] + cwd = Path.cwd().resolve() + for candidate in (cwd, *cwd.parents): + if ( + (candidate / "inputs" / "artifacts").is_dir() + and (candidate / "outputs" / "artifacts").is_dir() + ): + roots.append(candidate) + return roots + + def _failure_destinations(self, backward_dir_name: str) -> List[Path]: + destinations = [Path(backward_dir_name)] + tmp_roots = [] + explicit_tmp_root = getattr(self, "tmp_root", None) + if explicit_tmp_root is not None: + tmp_roots.append(Path(explicit_tmp_root).resolve()) + tmp_roots.extend(self._dflow_tmp_roots()) + + # PythonOPTemplate pre-creates these directories. In production the OP + # runs below ``/tmp`` without setting ``self.tmp_root``, so + # discover that ancestor from cwd and mirror evidence there before the + # raised exception causes dflow to package output artifacts. + seen_roots = set() + for tmp_root in tmp_roots: + if tmp_root in seen_roots: + continue + seen_roots.add(tmp_root) + if not ( + (tmp_root / "inputs" / "artifacts").is_dir() + and (tmp_root / "outputs" / "artifacts").is_dir() + ): + continue + dflow_output = ( + tmp_root / "outputs" / "artifacts" / "backward_dir" + ) + if dflow_output.resolve() not in { + destination.resolve() for destination in destinations + }: + destinations.append(dflow_output) + return destinations + + def run_task( + self, + backward_dir_name, + log_name, + backward_list: List[str], + run_image_config: Optional[Dict] = None, + optional_input: Optional[Dict] = None, + ) -> str: + try: + backward_dir_name = super().run_task( + backward_dir_name, + log_name, + backward_list, + run_image_config, + optional_input, + ) + except FileNotFoundError as error: + # Some dflow/fpop combinations pass a runtime ``log_name`` that + # differs from the calculator's declared ``outlog`` backward + # file. VASP has already finished successfully at this point, + # but upstream fpop then aborts while packaging the nonexistent + # alias and discards every expensive output. Recover only when + # the scientific completion gate passes and the missing path is + # exactly a declared backward file; genuine VASP/input failures + # still propagate unchanged. + missing = Path(error.filename).name if error.filename else "" + status = self._validate_stages() + if ( + status.get("state") != "succeeded" + or missing not in set(backward_list) + or missing != "outlog" + or not Path(log_name).is_file() + ): + raise + shutil.copy2(Path(log_name), Path(missing)) + backward_dir = Path(backward_dir_name) + backward_dir.mkdir(parents=True, exist_ok=True) + for name in dict.fromkeys([log_name, *backward_list]): + source = Path(name) + if source.is_file(): + shutil.copy2(source, backward_dir / source.name) + backward_dir_name = str(backward_dir) + except TransientError as error: + if "could not check the exact cause" not in str(error): + raise + status = self._validate_stages() + details = [] + for stage in status.get("stages", []): + if stage.get("finished"): + continue + reasons = ",".join(stage.get("failure_reasons", [])) + details.append(f"{stage.get('name')}: {reasons}") + raise TransientError( + "APEX VASP completion validation failed; " + + "; ".join(details) + ) from error + status = self._validate_stages() + self._save_stage_evidence(backward_dir_name, status) + if status["state"] != "succeeded": + failed = ", ".join(status.get("missing_or_incomplete_stages", [])) + raise TransientError( + "APEX VASP staged run is incomplete; missing or unfinished " + f"stage(s): {failed}" + ) + return backward_dir_name + + @OP.exec_sign_check + def execute(self, op_in: OPIO) -> OPIO: + task_path = Path(op_in["task_path"]).resolve() + if not task_path.is_dir(): + raise FatalError(f"cannot find VASP task directory {task_path}") + for name in self._MANDATORY_INPUTS: + source = task_path / name + if not source.exists(): + raise FatalError(f"cannot find VASP input file {source}") + + task_name = op_in["task_name"] + backward_dir_name = op_in["backward_dir_name"] + log_name = op_in["log_name"] + work_dir = Path(task_name) + + with set_directory(work_dir, mkdir=True): + for source in self._prepared_inputs( + task_path, backward_dir_name, log_name + ): + self._copy_path(source, Path(source.name)) + + optional_artifact = op_in.get("optional_artifact") or {} + for name, artifact_path in optional_artifact.items(): + source = Path(artifact_path) + if not source.exists(): + fallback = task_path / name + if fallback.exists(): + source = fallback + else: + logging.warning( + "Optional VASP artifact %s does not exist", source + ) + continue + self._copy_path(source, Path(name)) + + try: + backward_dir_name = self.run_task( + backward_dir_name, + log_name, + op_in["backward_list"], + op_in["run_image_config"], + op_in["optional_input"], + ) + except Exception as error: + for destination in self._failure_destinations( + backward_dir_name + ): + self._save_failure_evidence( + destination, error, log_name + ) + raise + + return OPIO({"backward_dir": work_dir / backward_dir_name}) diff --git a/apex/op/property_ops.py b/apex/op/property_ops.py index bf4e3883..48a18a54 100644 --- a/apex/op/property_ops.py +++ b/apex/op/property_ops.py @@ -12,7 +12,7 @@ from monty.serialization import dumpfn from apex.utils import recursive_search, apex_task_succeeded from apex.core.lib.utils import create_path -from apex.core.calculator import LAMMPS_INTER_TYPE +from apex.core.calculator import LAMMPS_INTER_TYPE, lammps_model_files_for_cleanup from apex.task_failure import ( REMOTE_LAMMPS_STARTUP_FAILURE, classify_apex_task_status, @@ -21,6 +21,18 @@ upload_packages.append(__file__) +TASK_FAILURE_TOLERANT_TYPES = { + "gamma_surface", + "gamma", + "eos", + "surface", + "vacancy", + "interstitial", + "cohesive", + "decohesive", +} + + def _load_task_status(status_path: Path): if not status_path.is_file(): return None @@ -159,7 +171,8 @@ def get_output_sign(cls): 'output_work_path': Artifact(Path), 'task_names': List[str], 'njobs': int, - 'task_paths': Artifact(List[Path]) + 'task_paths': Artifact(List[Path]), + 'backward_list': List[str], }) @OP.exec_sign_check @@ -200,7 +213,8 @@ def execute( 'output_work_path': abs_path_to_prop, 'task_names': [], 'njobs': 0, - 'task_paths': [] + 'task_paths': [], + 'backward_list': [], }) inter_param_prop = inter_param @@ -208,6 +222,9 @@ def execute( inter_param_prop = prop_param["cal_setting"]["overwrite_interaction"] prop = make_property_instance(prop_param, inter_param_prop) + backward_list = make_calculator( + inter_param_prop, "POSCAR" + ).backward_files(prop.task_type()) task_list = prop.make_confs(abs_path_to_prop, path_to_equi, do_refine) for kk in task_list: if (not rerun_finished) and apex_task_succeeded(kk): @@ -243,7 +260,8 @@ def execute( "output_work_path": input_work_path, "task_names": run_task_names, "njobs": njobs, - "task_paths": jobs + "task_paths": jobs, + "backward_list": backward_list, }) return op_out @@ -329,13 +347,21 @@ def execute(self, op_in: OPIO) -> OPIO: abs_path_to_prop / "failed_lammps_tasks.json", indent=4, ) - raise RuntimeError( - "LAMMPS failed for property task(s): " - + ", ".join(item["task"] for item in lammps_failures) - + ". Retrieved task directories contain apex_task_status.json " - "with failed status records, .debug.log, log.lammps, outlog, " - "and any partial output files." - ) + failed_task_names = ", ".join(item["task"] for item in lammps_failures) + if prop_param.get("type") in TASK_FAILURE_TOLERANT_TYPES: + logging.warning( + "LAMMPS failed for property task(s): %s. " + "Continuing post-process with NaN placeholders for failed points.", + failed_task_names, + ) + else: + raise RuntimeError( + "LAMMPS failed for property task(s): " + + failed_task_names + + ". Retrieved task directories contain apex_task_status.json " + "with failed status records, .debug.log, log.lammps, outlog, " + "and any partial output files." + ) prop = make_property_instance(prop_param, inter_param) param_json = os.path.join(abs_path_to_prop, "param.json") @@ -351,11 +377,7 @@ def execute(self, op_in: OPIO) -> OPIO: # remove potential files in each task if inter_type in LAMMPS_INTER_TYPE: os.chdir(abs_path_to_prop) - inter_files_name = [] - if type(inter_param["model"]) is str: - inter_files_name = [inter_param["model"]] - elif type(inter_param["model"]) is list: - inter_files_name.extend(inter_param["model"]) + inter_files_name = lammps_model_files_for_cleanup(inter_param) for file in inter_files_name: cmd = f"rm -f ../{file}" subprocess.call(cmd, shell=True) diff --git a/apex/op/relaxation_ops.py b/apex/op/relaxation_ops.py index 26502892..7e469149 100644 --- a/apex/op/relaxation_ops.py +++ b/apex/op/relaxation_ops.py @@ -9,7 +9,7 @@ Artifact, upload_packages ) -from apex.core.calculator import LAMMPS_INTER_TYPE +from apex.core.calculator import LAMMPS_INTER_TYPE, lammps_model_files_for_cleanup from apex.utils import recursive_search, apex_task_succeeded upload_packages.append(__file__) @@ -199,23 +199,21 @@ def execute(self, op_in: OPIO) -> OPIO: # remove potential files inter_files_name = [] if inter_type in LAMMPS_INTER_TYPE: - if type(inter_param["model"]) is str: - inter_files_name = [inter_param["model"]] - elif type(inter_param["model"]) is list: - inter_files_name.extend(inter_param["model"]) + inter_files_name = lammps_model_files_for_cleanup(inter_param) elif inter_type == 'vasp': inter_files_name = ['POTCAR'] - for ii in conf_dirs: - cmd = 'rm -f' - for jj in inter_files_name: - cmd += f' {jj}' - os.chdir(ii) - subprocess.call(cmd, shell=True) - os.chdir(op_in['input_all']) - os.chdir(os.path.join(ii, 'relaxation/relax_task')) - subprocess.call(cmd, shell=True) - os.chdir(op_in['input_all']) + if inter_files_name: + for ii in conf_dirs: + cmd = 'rm -f' + for jj in inter_files_name: + cmd += f' {jj}' + os.chdir(ii) + subprocess.call(cmd, shell=True) + os.chdir(op_in['input_all']) + os.chdir(os.path.join(ii, 'relaxation/relax_task')) + subprocess.call(cmd, shell=True) + os.chdir(op_in['input_all']) os.chdir(cwd) for ii in copy_dir_list: diff --git a/apex/preview.py b/apex/preview.py index 9d71ad5d..0980cde8 100644 --- a/apex/preview.py +++ b/apex/preview.py @@ -56,20 +56,33 @@ def _resolve_structure_path(base_dir: Path, structure_glob: str) -> str: return matches[0] -def _prepare_equilibrium_dir(structure_dir: str, temp_root: Path, label: str) -> str: +def _find_structure_file(structure_dir: str) -> Path: src_dir = Path(structure_dir) - equi_dir = temp_root / label / "relaxation" / "relax_task" - equi_dir.mkdir(parents=True, exist_ok=True) - candidates = [src_dir / "CONTCAR", src_dir / "POSCAR", src_dir / "STRU"] source_file = next((path for path in candidates if path.is_file()), None) if source_file is None: raise FileNotFoundError( f"Cannot find CONTCAR/POSCAR/STRU under {structure_dir}" ) + return source_file + +def _prepare_equilibrium_dir(structure_dir: str, temp_root: Path, label: str) -> str: + src_dir = Path(structure_dir) + equi_dir = temp_root / label / "relaxation" / "relax_task" + equi_dir.mkdir(parents=True, exist_ok=True) + + source_file = _find_structure_file(structure_dir) target_file = equi_dir / "CONTCAR" shutil.copy2(source_file, target_file) + source_result = src_dir / "result.json" + target_result = equi_dir / "result.json" + if source_result.is_file(): + shutil.copy2(source_result, target_result) + else: + # Gamma.make_confs requires the baseline result to exist, while preview + # only generates geometry and never runs property post-processing. + target_result.write_text('{"preview_only": true}\n') return str(equi_dir) @@ -95,6 +108,24 @@ def _structure_bounds(atoms, radii_scale: float): radii = covalent_radii[atoms.get_atomic_numbers()] * radii_scale low = (xy - radii[:, None]).min(axis=0) high = (xy + radii[:, None]).max(axis=0) + + # Keep the complete projected unit cell in the viewport. In particular, + # a Gamma slab's vacuum is represented by empty cell volume; atom-only + # bounds crop that volume even when plot_atoms(show_unit_cell=1) is used. + cell = np.asarray(atoms.cell.array, dtype=float) + if cell.shape == (3, 3) and np.isfinite(cell).all() and np.any(cell): + cell_corners = np.asarray( + [ + [i, j, k] + for i in (0.0, 1.0) + for j in (0.0, 1.0) + for k in (0.0, 1.0) + ], + dtype=float, + ) @ cell + low = np.minimum(low, cell_corners[:, :2].min(axis=0)) + high = np.maximum(high, cell_corners[:, :2].max(axis=0)) + x_min, y_min = low x_max, y_max = high return x_min, x_max, y_min, y_max @@ -273,6 +304,250 @@ def _load_frames(poscar_files: Iterable[str]): return frames +def _unit_vector(vector, *, label: str): + vector = np.asarray(vector, dtype=float) + norm = float(np.linalg.norm(vector)) + if norm <= 1.0e-12: + raise RuntimeError(f"Cannot define preview direction from zero {label}") + return vector / norm + + +def _screen_transform(first_axis, second_axis): + screen_x = _unit_vector(first_axis, label="screen x axis") + second_axis = np.asarray(second_axis, dtype=float) + screen_y = _unit_vector( + second_axis - np.dot(second_axis, screen_x) * screen_x, + label="screen y axis", + ) + view_normal = _unit_vector( + np.cross(screen_x, screen_y), + label="view normal", + ) + return np.vstack([screen_x, screen_y, view_normal]) + + +def _transform_frames(frames, transform): + transformed = [] + transform = np.asarray(transform, dtype=float) + for source in frames: + atoms = source.copy() + atoms.positions = atoms.positions @ transform.T + atoms.set_cell(atoms.cell.array @ transform.T, scale_atoms=False) + transformed.append(atoms) + return transformed + + +def _observed_gamma_slip(frames): + if not frames: + raise RuntimeError( + "Gamma view projection requires at least one displacement frame" + ) + first = frames[0] + if len(frames) < 2: + return _unit_vector( + first.cell.array[0], + label="generated Gamma slip axis", + ) + first_scaled = first.get_scaled_positions(wrap=False) + for candidate in frames[1:]: + if len(first) != len(candidate): + raise RuntimeError("Gamma preview frames have inconsistent atom counts") + raw_fractional_delta = ( + candidate.get_scaled_positions(wrap=False) - first_scaled + ) + minimum_image_delta = ( + raw_fractional_delta - np.round(raw_fractional_delta) + ) + for fractional_delta in (minimum_image_delta, raw_fractional_delta): + cartesian_delta = fractional_delta @ first.cell.array + moving = cartesian_delta[ + np.linalg.norm(cartesian_delta, axis=1) > 1.0e-8 + ] + if len(moving): + mean_slip = np.mean(moving, axis=0) + if np.linalg.norm(mean_slip) > 1.0e-12: + return _unit_vector( + mean_slip, + label="observed Gamma slip", + ) + + # A one-step scan can end at a periodically equivalent structure whose + # coordinates have already been wrapped. Gamma orients the generated slab + # a axis along the slip direction, so it remains an unambiguous fallback. + return _unit_vector( + first.cell.array[0], + label="generated Gamma slip axis", + ) + + +def _gamma_slip_axis(frames, *, use_cell_axis: bool = False): + if not frames: + raise RuntimeError( + "Gamma view projection requires at least one displacement frame" + ) + if use_cell_axis: + # GammaSurface traverses two independent displacement directions, so + # frame-to-frame motion is not a stable definition of its primary slip + # axis. Its generated slab a axis is the resolved x slip direction. + return _unit_vector( + frames[0].cell.array[0], + label="generated Gamma slip axis", + ) + return _observed_gamma_slip(frames) + + +def _slip_plane_transform(frames, *, use_cell_axis: bool = False): + slip = _gamma_slip_axis(frames, use_cell_axis=use_cell_axis) + cell = np.asarray(frames[0].cell.array, dtype=float) + normal = _unit_vector( + np.cross(cell[0], cell[1]), + label="generated slip-plane normal", + ) + if np.dot(normal, cell[2]) < 0: + normal *= -1 + normal = _unit_vector( + normal - np.dot(normal, slip) * slip, + label="orthogonalized slip-plane normal", + ) + in_plane_y = _unit_vector( + np.cross(normal, slip), + label="second slip-plane axis", + ) + return np.vstack([slip, in_plane_y, normal]) + + +def _resolved_gamma_view_context(prop_obj): + from apex.core.lib.trans_tools import direction_miller_bravais_to_miller + from apex.core.lib.trans_tools import plane_miller_bravais_to_miller + + parent = getattr(prop_obj, "conv_std_structure", None) + plane = getattr(prop_obj, "plane_miller", None) + slip_direction = getattr(prop_obj, "slip_direction", None) + if parent is None or plane is None or slip_direction is None: + raise RuntimeError( + "Gamma view projection requires resolved structure and slip-system data" + ) + + if getattr(prop_obj, "structure_type", None) == "hcp": + if len(plane) == 4: + plane = plane_miller_bravais_to_miller(plane) + if len(slip_direction) == 4: + slip_direction = direction_miller_bravais_to_miller(slip_direction) + + plane = np.asarray(plane, dtype=float) + slip_direction = np.asarray(slip_direction, dtype=float) + if plane.shape != (3,) or slip_direction.shape != (3,): + raise RuntimeError( + "Gamma view projection requires three-index plane_miller and " + "slip_direction values after crystallographic conversion" + ) + return parent, plane, slip_direction + + +def _parent_bc_transform( + parent, + frames, + plane, + slip_direction, + *, + use_cell_axis: bool = False, +): + plane = np.asarray(plane, dtype=float) + slip_direction = np.asarray(slip_direction, dtype=float) + + slip_parent = slip_direction @ parent.lattice.matrix + normal_parent = plane @ parent.lattice.reciprocal_lattice.matrix + ex_parent = _unit_vector(slip_parent, label="parent slip direction") + ez_parent = _unit_vector( + normal_parent - np.dot(normal_parent, ex_parent) * ex_parent, + label="parent slip-plane normal", + ) + ey_parent = _unit_vector( + np.cross(ez_parent, ex_parent), + label="parent second in-plane axis", + ) + + ex_slab = _gamma_slip_axis(frames, use_cell_axis=use_cell_axis) + cell = np.asarray(frames[0].cell.array, dtype=float) + ez_slab = _unit_vector( + np.cross(cell[0], cell[1]), + label="generated slip-plane normal", + ) + if np.dot(ez_slab, cell[2]) < 0: + ez_slab *= -1 + ez_slab = _unit_vector( + ez_slab - np.dot(ez_slab, ex_slab) * ex_slab, + label="orthogonalized generated slip-plane normal", + ) + ey_slab = _unit_vector( + np.cross(ez_slab, ex_slab), + label="generated second in-plane axis", + ) + + parent_basis = np.vstack([ex_parent, ey_parent, ez_parent]) + slab_basis = np.vstack([ex_slab, ey_slab, ez_slab]) + + def map_parent_vector(vector): + components = parent_basis @ np.asarray(vector, dtype=float) + return components @ slab_basis + + parent_b_slab = map_parent_vector(parent.lattice.matrix[1]) + parent_c_slab = map_parent_vector(parent.lattice.matrix[2]) + return _screen_transform(parent_b_slab, parent_c_slab) + + +def _gamma_frames_for_view( + frames, + view: str, + *, + parent_structure, + plane_miller, + slip_direction, + use_cell_axis: bool = False, +): + if view == "default": + return [frame.copy() for frame in frames] + if view == "slip-plane": + return _transform_frames( + frames, + _slip_plane_transform(frames, use_cell_axis=use_cell_axis), + ) + if view == "parent-bc": + return _transform_frames( + frames, + _parent_bc_transform( + parent_structure, + frames, + plane_miller, + slip_direction, + use_cell_axis=use_cell_axis, + ), + ) + raise ValueError(f"Unknown Gamma preview view: {view}") + + +def _requested_gif_views( + gif_view: str, + property_type: Optional[str] = None, +) -> List[str]: + if gif_view == "auto": + if property_type in {"gamma", "gamma_surface"}: + return ["slip-plane", "parent-bc"] + return ["default"] + if gif_view == "both": + return ["slip-plane", "parent-bc"] + if gif_view in {"default", "slip-plane", "parent-bc"}: + return [gif_view] + raise ValueError(f"Unknown GIF view: {gif_view}") + + +def _view_output_gif_path(output_gif: Path, view: str) -> Path: + if view == "default": + return output_gif + suffix = view.replace("-", "_") + return output_gif.with_name(f"{output_gif.stem}_{suffix}.gif") + + def _arrange_gamma_surface_tasks(task_list: List[str]) -> List[str]: indexed_tasks = [] for fallback_index, task_dir in enumerate(task_list): @@ -295,6 +570,31 @@ def surface_slide_key(item): return [task_dir for _, _, _, task_dir in sorted(indexed_tasks, key=surface_slide_key)] +def _min_pair_distance(structure) -> float: + dmat = structure.distance_matrix + n = dmat.shape[0] + if n < 2: + return float("inf") + iu = np.triu_indices(n, k=1) + return float(dmat[iu].min()) + + +def _warn_gamma_surface_overlaps(task_list: List[str], threshold: float = 0.2) -> None: + from pymatgen.core import Structure + + for task_dir in task_list: + poscar = os.path.join(task_dir, "POSCAR") + if not os.path.isfile(poscar): + continue + structure = Structure.from_file(poscar) + if _min_pair_distance(structure) < threshold: + print( + "Generated Gamma surface contains overlapping atoms.", + file=sys.stderr, + ) + return + + def _derive_output_gif_path(parameter_path: Path, structures_count: int, prop_label: str) -> Path: if structures_count == 1 and prop_label == "": return parameter_path.with_suffix(".gif") @@ -330,6 +630,7 @@ def preview_parameter_file( gif_padding: float = 0.30, gif_xshift: float = 0.0, gif_yshift: float = 0.0, + gif_view: str = "auto", ) -> List[str]: parameter_path = Path(parameter_file).resolve() payload = loadfn(str(parameter_path)) @@ -368,7 +669,6 @@ def preview_parameter_file( for structure_index, structure_glob in enumerate(structures): structure_dir = _resolve_structure_path(parameter_path.parent, structure_glob) - for prop_index, (prop, suffix, do_refine) in enumerate(runnable_properties): prop_obj = make_property_instance( {**deepcopy(prop), "type": prop["type"]}, @@ -401,23 +701,54 @@ def preview_parameter_file( task_list = prop_obj.make_confs( str(work_dir), equi_dir, refine=do_refine, **make_kwargs ) - prop_obj.post_process(task_list) if prop.get("type") == "gamma_surface": + _warn_gamma_surface_overlaps(task_list) task_list = _arrange_gamma_surface_tasks(task_list) poscar_files = [os.path.join(task_dir, "POSCAR") for task_dir in task_list] frames = _load_frames(poscar_files) - _write_gif( - frames, - str(output_gif), - fps=gif_fps, - dpi=gif_dpi, - padding=gif_padding, - xshift=gif_xshift, - yshift=gif_yshift, - progress_label="Loading...", - ) - output_paths.append(str(output_gif)) + property_type = prop.get("type") + is_gamma_preview = property_type in {"gamma", "gamma_surface"} + views = _requested_gif_views(gif_view, property_type) + if not is_gamma_preview and views != ["default"]: + raise RuntimeError( + "--gif-view slip-plane, parent-bc, and both are " + "supported only for type=gamma or type=gamma_surface" + ) + parent_structure = None + plane_miller = None + slip_direction = None + if is_gamma_preview and "parent-bc" in views: + ( + parent_structure, + plane_miller, + slip_direction, + ) = _resolved_gamma_view_context(prop_obj) + for view in views: + view_frames = ( + _gamma_frames_for_view( + frames, + view, + parent_structure=parent_structure, + plane_miller=plane_miller, + slip_direction=slip_direction, + use_cell_axis=property_type == "gamma_surface", + ) + if is_gamma_preview + else frames + ) + view_output_gif = _view_output_gif_path(output_gif, view) + _write_gif( + view_frames, + str(view_output_gif), + fps=gif_fps, + dpi=gif_dpi, + padding=gif_padding, + xshift=gif_xshift, + yshift=gif_yshift, + progress_label=f"Loading {view} view...", + ) + output_paths.append(str(view_output_gif)) return output_paths @@ -433,6 +764,7 @@ def preview_from_args(args: argparse.Namespace) -> List[str]: gif_padding=args.gif_padding, gif_xshift=args.gif_xshift, gif_yshift=args.gif_yshift, + gif_view=getattr(args, "gif_view", "auto"), ) ) return outputs @@ -467,6 +799,17 @@ def build_parser() -> argparse.ArgumentParser: default=0.0, help="Shift the rendered viewport vertically by a fraction of the data span; positive values move the structure downward", ) + parser.add_argument( + "--gif-view", + choices=("auto", "default", "slip-plane", "parent-bc", "both"), + default="auto", + help=( + "Gamma projection: auto writes both scientific views for gamma " + "and gamma_surface; alternatively preserve the legacy Cartesian " + "view, look normal to the slip plane, look normal to the parent " + "bc plane, or explicitly write both views" + ), + ) return parser diff --git a/apex/reporter/DashReportApp.py b/apex/reporter/DashReportApp.py index 69262779..d291b933 100644 --- a/apex/reporter/DashReportApp.py +++ b/apex/reporter/DashReportApp.py @@ -46,6 +46,8 @@ def return_prop_class(prop_type: str): return FiniteTelasticReport elif prop_type == 'annealing': return AnnealingReport + elif prop_type == 'melting_point': + return MeltingPointReport def return_prop_type(prop: str): @@ -56,6 +58,8 @@ def return_prop_type(prop: str): return 'finite_t_latt' if prop.startswith('finite_t_elastic'): return 'finite_t_elastic' + if prop.startswith('melting_point'): + return 'melting_point' prop_type = prop.split('_')[0] except AttributeError: return None diff --git a/apex/reporter/property_report.py b/apex/reporter/property_report.py index dbdb7319..dd482b00 100644 --- a/apex/reporter/property_report.py +++ b/apex/reporter/property_report.py @@ -281,13 +281,68 @@ def dash_table(res_data: dict, decimal: int = 3, **kwargs) -> dash_table.DataTab return build_table(df), df +class MeltingPointReport(PropertyReport): + """Report a q6/interface-velocity two-phase melting bracket.""" + + @staticmethod + def plotly_graph(res_data: dict, name: str, **kwargs): + rows = res_data.get("temperatures", []) + x_values = [row.get("temperature_K") for row in rows] + y_values = [row.get("interface_velocity_mean_A_per_ps") for row in rows] + error_values = [ + row.get("interface_velocity_standard_error_A_per_ps") or 0.0 + for row in rows + ] + trace = go.Scatter( + name=name, + x=x_values, + y=y_values, + error_y={"type": "data", "array": error_values, "visible": True}, + mode="lines+markers", + ) + layout = go.Layout( + title="Two-phase Coexistence Melting Point", + xaxis={"title": "Temperature (K)"}, + yaxis={"title": "Interface velocity (Å/ps)", "zeroline": True}, + ) + return [trace], layout + + @staticmethod + def dash_table(res_data: dict, decimal: int = 6, **kwargs): + bracket = res_data.get("bracket", {}) + rows = [] + for item in res_data.get("temperatures", []): + mean_velocity = item.get("interface_velocity_mean_A_per_ps") + rows.append({ + "Temperature (K)": item.get("temperature_K"), + "Replicas": item.get("replica_count"), + "Outcome": item.get("consensus_outcome"), + "Interface velocity (Å/ps)": ( + None + if mean_velocity is None + else round(mean_velocity, decimal) + ), + "Replica std (Å/ps)": ( + None + if item.get("interface_velocity_std_A_per_ps") is None + else round(item["interface_velocity_std_A_per_ps"], decimal) + ), + "Estimated Tm (K)": bracket.get("estimated_melting_temperature_K"), + "Half-width (K)": bracket.get("uncertainty_half_width_K"), + }) + df = pd.DataFrame(rows) + return build_table(df), df + + class FiniteTlattReport(PropertyReport): """Report lattice parameters as a function of temperature.""" @staticmethod def _normalized_data(res_data, relax_abc=None): data = {} - for value in res_data.values(): + for key, value in res_data.items(): + if str(key).startswith("_"): + continue if isinstance(value, (list, tuple)) and len(value) >= 4: a, b, c, temp = ( float(value[0]), diff --git a/apex/skill.py b/apex/skill.py index aaefddf5..d51702d3 100644 --- a/apex/skill.py +++ b/apex/skill.py @@ -8,94 +8,97 @@ from apex.skills import SKILL_NAME, get_skill_root # Skip noise when packing for MatMaster / agent skill upload. -_ZIP_SKIP_DIR_NAMES = {"__pycache__", ".git"} -_ZIP_SKIP_FILE_NAMES = {".DS_Store"} +_ZIP_SKIP_DIR_NAMES = {"__pycache__", ".git", "variants"} +_ZIP_SKIP_FILE_NAMES = { + ".DS_Store", + "global_bohrium_direct.json", + "global_local_cluster_slurm.json", + "global_local_debug.json", +} _ZIP_SKIP_FILE_SUFFIXES = {".pyc", ".pyo", ".pt", ".partial"} +_ZIP_INCLUDED_PT_FILES = { + Path("models") / "DPA4-alloytongqi" / "model.pt", +} -def _agent_install_prompt(skill_root: Path) -> str: - skill_md = skill_root / "SKILL.md" +def _agent_install_prompt() -> str: + """Return a portable local-Agent installation prompt.""" return f"""\ -# Agent prompt: install the APEX skill ({SKILL_NAME}) +# Agent prompt: install the local APEX skill ({SKILL_NAME}) You are an AI coding agent. Install the bundled APEX agent skill so future -sessions can orchestrate APEX (Alloy Property EXplorer) workflows. +sessions can run APEX from this machine. -## Preferred: MatMaster zip upload +## Required: choose the execution profile -Ask the user (or run yourself) to build an uploadable archive: +Before copying files, ask the user exactly where APEX calculations will run: -```bash -apex skill --zip -``` - -This writes `./{SKILL_NAME}.zip` with top-level folder `{SKILL_NAME}/` -(containing `SKILL.md`, `reference/`, `scripts/`, small frozen models under -`models/`, etc.). Upload that zip in MatMaster's skill upload UI. +1. **Bohrium cloud** — this local Agent runs `apex submit` directly and uses + credentials saved by `apex account`; do not require a ticket or outer job. +2. **local** — run calculators on this workstation with `apex submit -d`. +3. **local cluster** — run from a cluster login node and dispatch calculator + tasks through Slurm/PBS with DPDispatcher. -Optional output path: +Do not guess. Wait for the user's answer, then map it to one profile file: -```bash -apex skill --zip -o /tmp/{SKILL_NAME}.zip -``` +- Bohrium cloud: `bohrium-direct.md` +- local: `local-debug.md` +- local cluster: `local-cluster.md` -## Alternative: copy from the local package +## Install from this package -- Skill directory: `{skill_root}` -- Entry file: `{skill_md}` - -Confirm `{skill_md}` exists before copying. Do not invent or regenerate the -skill content; copy the directory as-is (including `reference/`, `scripts/`, -`data/`, `models/DPA-3.2-5M`, and `plugin.yaml`). - -For DeePMD/DPA jobs, use the bundled frozen DPA-3.2 model: -- `models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth` - -The source multi-head checkpoint (`.pt`) is **not** bundled. Download it only -when the user explicitly needs a different task head: +Resolve the bundled skill directory at runtime. Do not hardcode or copy a +host-absolute path into a shared prompt: ```bash -python scripts/fetch_models.py --source-checkpoint +SKILL_ROOT="$(python -c 'from apex.skills import get_skill_root; print(get_skill_root())')" +LOCAL_VARIANT="$SKILL_ROOT/variants/local" ``` -Install destinations (create parents if missing): - -1. Cursor (user-global): `~/.cursor/skills/{SKILL_NAME}/` -2. Codex / OpenAI agents (user-global): `~/.codex/skills/{SKILL_NAME}/` -3. Cursor (project-local, optional): `/.cursor/skills/{SKILL_NAME}/` +Confirm `$LOCAL_VARIANT/SKILL.md` exists. Copy the shared directory, then +overlay the local entry/reference and the selected execution profile: ```bash -mkdir -p ~/.cursor/skills ~/.codex/skills -cp -R "{skill_root}" ~/.cursor/skills/{SKILL_NAME} -cp -R "{skill_root}" ~/.codex/skills/{SKILL_NAME} +SKILL_ROOT="$(python -c 'from apex.skills import get_skill_root; print(get_skill_root())')" +LOCAL_VARIANT="$SKILL_ROOT/variants/local" +DEST="$HOME/.cursor/skills/{SKILL_NAME}" # or ~/.codex/skills/{SKILL_NAME} +PROFILE="" +mkdir -p "$(dirname "$DEST")" +rm -rf "$DEST" +cp -R "$SKILL_ROOT" "$DEST" +cp "$LOCAL_VARIANT/SKILL.md" "$DEST/SKILL.md" +cp "$LOCAL_VARIANT/reference/submission.md" \ + "$DEST/reference/submission.md" +cp "$LOCAL_VARIANT/profiles/$PROFILE" \ + "$DEST/reference/execution-profile.md" +rm -rf "$DEST/variants" ``` ## Verification -- MatMaster: skill appears after zip upload, name `{SKILL_NAME}` -- Local: `~/.cursor/skills/{SKILL_NAME}/SKILL.md` exists +- `$DEST/SKILL.md` exists and contains `Local Agent Edition` +- `$DEST/reference/execution-profile.md` exists and names the selected profile - YAML frontmatter `name:` is `{SKILL_NAME}` ## After install - Tell the user the skill is available as `{SKILL_NAME}` / `/apex-flow` +- State which execution profile was installed and how to change it - For APEX calculation requests, read `SKILL.md` first, then load referenced docs under `reference/` only as needed -## Notes -- The PyPI / pip package and agent skill both use the name `{SKILL_NAME}`. -- Re-run `apex skill` to reprint this prompt, or `apex skill --zip` to rebuild - the MatMaster upload archive. +`apex skill --zip` is a separate Bohrium Cloud/MatMaster distribution. Do not +install that ticket/outer-job variant for a local Agent. """ def build_skill_zip(output: Path | None = None) -> Path: """ - Pack the bundled apex-flow directory into a zip for MatMaster upload. + Pack the Bohrium Cloud apex-flow variant into a zip for MatMaster upload. - Large DeePMD source checkpoints (``*.pt``) are excluded on purpose. - Ready-to-run frozen ``*.pb`` and ``*.pth`` models under ``models/`` are - included. + Arbitrary DeePMD source/training checkpoints (``*.pt``) are excluded on + purpose. The explicitly allow-listed, ready-to-run single-task DPA4 + checkpoint bundled under ``models/`` is included. """ skill_root = get_skill_root() if not (skill_root / "SKILL.md").is_file(): @@ -115,22 +118,26 @@ def build_skill_zip(output: Path | None = None) -> Path: for path in sorted(skill_root.rglob("*")): if not path.is_file(): continue + relative = path.relative_to(skill_root) if any(part in _ZIP_SKIP_DIR_NAMES for part in path.parts): continue if path.name in _ZIP_SKIP_FILE_NAMES: continue - if path.suffix in _ZIP_SKIP_FILE_SUFFIXES: + if ( + path.suffix in _ZIP_SKIP_FILE_SUFFIXES + and relative not in _ZIP_INCLUDED_PT_FILES + ): continue if path.name.endswith(".partial"): continue - arcname = Path(SKILL_NAME) / path.relative_to(skill_root) + arcname = Path(SKILL_NAME) / relative zf.write(path, arcname.as_posix()) return out def skill_from_args(args) -> None: - """Print the Agent install prompt, or write a MatMaster-ready skill zip.""" + """Print the local Agent prompt, or write the Bohrium Cloud skill zip.""" skill_root = get_skill_root() if not (skill_root / "SKILL.md").is_file(): raise FileNotFoundError( @@ -140,7 +147,7 @@ def skill_from_args(args) -> None: if getattr(args, "zip", False): output = getattr(args, "output", None) zip_path = build_skill_zip(Path(output) if output else None) - print(f"Wrote MatMaster skill archive: {zip_path}") + print(f"Wrote Bohrium Cloud/MatMaster skill archive: {zip_path}") print(f"Upload this zip in MatMaster (top-level folder: {SKILL_NAME}/).") return - print(_agent_install_prompt(skill_root).rstrip()) + print(_agent_install_prompt().rstrip()) diff --git a/apex/skills/apex-flow/SKILL.md b/apex/skills/apex-flow/SKILL.md index eb826d64..15568dae 100644 --- a/apex/skills/apex-flow/SKILL.md +++ b/apex/skills/apex-flow/SKILL.md @@ -1,6 +1,6 @@ --- name: apex-flow -description: Batch multi-property materials calculations (EOS, 0K elastic constant, surface energy, phonon, finite temperature elastic constant, gamma surface, gamma line, cohesive energy) and random-solid-solution structure generation via APEX calculator backends VASP/ABACUS/LAMMPS. The bundled DPA-3.2-5M OMat24 model is a DeePMD potential for LAMMPS. Use when the user mentions APEX, apex, alloy property, generate/give random solid solution, generate/give solid solution, generate/give high-entropy alloy, generate/give high-entropy oxide, generate/give high-entropy material, or multi-property DFT/MLIP screening. +description: Run APEX relaxation and 15-property materials workflows with VASP, ABACUS, or LAMMPS, including EOS, elastic, surface/defect, phonon/Grüneisen, parent-aware Gamma line/surface, finite-temperature lattice/elastic/annealing, two-phase melting-point brackets, previews, reporting, and RSS/high-entropy generation. Supports the validated general DPA4 phonoLAMMPS runtime and a bundled alloytongqi checkpoint with a separate fail-closed T4 profile. Use for APEX, alloy properties, melting/coexistence, stacking faults, random solid solutions, or multi-property DFT/MLIP screening. --- # APEX Flow — Alloy Properties EXplorer @@ -41,9 +41,9 @@ Options to offer via AskQuestion: ## High-Level Workflow (5 Steps) -1. **Prepare inputs** — Check `BOHRIUM_ACCESS_KEY`, then generate `param.json` + `global.json` (including a fresh ticket) and copy structure/model files into a job directory using `scripts/generate_config.py create ...`. Never hand-write either JSON file. -2. **Submit outer Bohrium job** — A lightweight client (`c1_m2_cpu`, recommended) that runs `apex submit ...` without `-s`, connects to the dflow orchestration server, and waits for completion -3. **dflow executes** — Inner containers (LAMMPS/ABACUS/VASP) run the actual calculations, managed by `workflows.deepmodeling.com` +1. **Select the execution profile and prepare inputs** — For the Cloud/MatMaster zip profile, check `BOHRIUM_ACCESS_KEY`, then use `scripts/generate_config.py create ...` to generate `param.json` + ticket-bearing `global.json` and stage structures/models. For an installed local edition, read `reference/execution-profile.md`: Bohrium direct uses the masked `apex account` credentials, while local/debug and Slurm/PBS use their own profile and do not require a Bohrium ticket. Never mix profile instructions. +2. **Submit from the selected client** — Cloud/MatMaster uses a lightweight outer Bohrium client (`c1_m2_cpu`, recommended). The installed Bohrium-direct profile runs `apex submit` on the Agent machine and must not create an outer job. Local/debug and Slurm/PBS follow their installed profile. +3. **The selected runtime executes** — Bohrium uses dflow-managed inner containers; local debug runs on the workstation, and local cluster profiles dispatch through Slurm/PBS. 4. **Monitor and retrieve results** — `apex submit` monitors the inner workflow and retrieves results after completion; parse `confs//_00/result.json` 5. **Present results** — Summarize in a table with physical units (GPa for elastic, J/m² for surface, eV for energies) @@ -60,63 +60,98 @@ Options to offer via AskQuestion: - LAMMPS + classical (EAM / MEAM / SNAP): fast, CPU - ABACUS (DFT) - VASP (DFT; license-gated — resolve image via Bohrium `list_images` keyword=`vasp` or a user-known authorized address; otherwise stop) - **Step B — only if Step A is LAMMPS + DeePMD/DPA: use the model bundled with this skill.** - - Copy `models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth` into the job directory. - - This is the frozen, single-task `OMat24` branch of DPA-3.2-5M. It is - ready for APEX/LAMMPS, includes oxygen, and has 89 observed elements. - - Set `interaction.model` to the copied filename and set `"type_map": "auto"`. - APEX reads the structure at submission time and writes a zero-based, - contiguous element map. Do not derive indices from atomic numbers or the - model's internal type order. - - Do not invent a model path, download another model, or use the source multi-head `.pt` directly. Use a user-provided or separately frozen task head only when the user explicitly requests it or the bundled OMat24 model is unsuitable; explain the choice and obtain confirmation first. + **Step B — only if Step A is LAMMPS + DeePMD/DPA: identify the model and runtime separately.** + - The skill bundles `models/DPA4-alloytongqi/model.pt` as source-checkpoint + provenance. Never pass this `.pt` file directly to LAMMPS for the DPA4 + production profile; that profile uses a hashed image-resident `.pt2`. + - This is the DPA4 single-task `alloytongqi` model supplied by the user. + Static checkpoint inspection confirms DPA4 and an empty branch-alias list; + the `alloytongqi` branch provenance is user-supplied rather than embedded. + Do not infer training-domain coverage merely from the checkpoint type map. + - **Compatibility stop:** The same legacy repository path/tag was tested from + the accessible Bohrium registry mirror at digest + `sha256:43a27ca4a7bba7f774bbd56104d205a6a80cd9d65928f249f6109e9ef37b8402`. + Its DeepMD-kit 3.1.3 loader rejects this checkpoint with + `RuntimeError: Unknown model type: dpa4`; LAMMPS also aborts while + initializing `pair_style deepmd`. The configured `registry.dp.tech` + endpoint itself was pull-denied, so retain the tested mirror digest in any + report. Keep this old image as the general APEX default and the legacy + phonon/Grüneisen forced image; do not submit DPA4 through it. + - For DPA4, use `--runtime-profile dpa4-alloytongqi-t4`. It remains locked + while its image ref/digest placeholders or `pre_snapshot_only` status are + present. The recorded `c4_m15_1 * NVIDIA T4` run is candidate evidence, + not exact-image acceptance or a current recommendation. + Once published, the generator must write + `/usr/local/bin/dpa4-lmp -in in.lammps` and the full + `/usr/local/bin/dpa4-phonolammps {input_file} -c {poscar} --dim {dim} + {primitive_axes}` template; never shorten or override these entrypoints. + - Do not invent a model path or download another default model. Use a + different user-provided compatible model only when the user explicitly + requests it; explain the change and obtain confirmation first. **Skip Step A/B ONLY if** the user already stated them in THIS message - (e.g. “用 EAM”, “用 ABACUS 做 EOS”, “用 DPA-3.2-5M-OMat24.pth”). + (e.g. “用 EAM”, “用 ABACUS 做 EOS”, “用 DPA4 alloytongqi model.pt”). **If AskQuestion times out or fails**: state the intended APEX backend and bundled model selection (if LAMMPS+DPA) in plain text and WAIT. Never silently submit. 2. **STOP: Confirm property parameters before submission — DO NOT PROCEED WITHOUT USER ANSWER.** Before submitting, present the full `properties` configuration (JSON) to the user. Show the defaults that will be used and highlight: - Miller indices / slip systems (for surface/gamma/gamma_surface/decohesive) - Supercell sizes (for vacancy/interstitial/phonon/gruneisen/finite-T) - - Temperature ranges (for finite_t_latt/finite_t_elastic/annealing) + - Temperature ranges (for finite_t_latt/finite_t_elastic/annealing/melting_point) + - For `melting_point`: relaxed and expanded atom counts, temperatures, + replicas, total task count, stage lengths, restart inputs, and resources - Number of deformation/step points For crystallographic planes: - - `gamma` / `gamma_surface`: pick from the canonical FCC/BCC/HCP table in - repository **README §4.10** (see also `reference/properties.md` §8–9). Do not - invent slip systems; do not silently change an approved plane/direction. + - `gamma` / `gamma_surface`: start from the physically recommended FCC/BCC/HCP + table in repository **README §4.10** (see also + `reference/properties.md` §8–9). A non-tabulated system is allowed only when + the direction is in-plane; APEX warns and uses geometric construction, so + report the warning and inspect the generated geometry. Never silently change + an approved plane/direction. - `decohesive`: pick `miller_index` from the crystal-family table in **README §4.5** / `reference/properties.md` §10 (FCC/BCC/Diamond/ZB/Rocksalt/ HCP/Perovskite). HCP must use **3-index** only. Let the user approve or modify. **Skip ONLY if** the user provided explicit property parameters already. **If AskQuestion times out or fails**: display the parameters in your message and WAIT for confirmation before submitting. -3. **Two-layer architecture.** The outer Bohrium job is a thin submission client only. Never attempt `apex do` for production workflows — use `apex submit` which delegates to dflow. See `reference/submission.md` for the full architecture diagram. - For agent-managed Bohrium runs, always use `apex submit ...` without `-s`. - The outer client must remain active while dflow runs so APEX can monitor the +3. **Profile-aware submission architecture.** Never attempt `apex do` for production workflows — use `apex submit` which delegates to dflow. In the Cloud/MatMaster profile, the outer Bohrium job is a thin submission client; in the installed Bohrium-direct profile, the Agent machine is the client and there is no outer job. See the applicable `reference/submission.md`. + For agent-managed Bohrium runs, use `apex submit ...` without `-s`. + The active submit client must remain active while dflow runs so APEX can monitor the workflow and retrieve results automatically. Immediately preserve the exact inner dflow workflow ID printed by `apex submit` and report it to the user. Keep this ID available for all later - monitoring, retrieval, and workflow-control actions; do not confuse it with - the outer Bohrium job ID. Follow the exact status-query and reporting + monitoring, retrieval, and workflow-control actions. In Cloud/MatMaster, + do not confuse it with the outer Bohrium job ID. Follow the exact status-query and reporting protocol in `reference/workflow-control.md`; never infer workflow identity or material identity from the outer job name alone. Use the returned - workflow/step phases and durations to track progress. A + workflow/step phases and durations to track progress. In the Cloud/MatMaster + profile, also retain the outer job ID; do not invent one for direct mode. A long-running or failed step should be investigated by its step ID/key rather than treated as successful completion. After the workflow reaches `Succeeded`, verify that automatic result retrieval completed as described in `reference/submission.md`. -4. **Kill = inner FIRST, outer SECOND.** If you only kill the outer Bohrium node, the dflow workflow continues consuming resources silently. Always terminate the inner dflow workflow first. See `reference/workflow-control.md`. -5. **MUST use `generate_config.py`; never hand-write `param.json` or `global.json`.** +4. **Kill = inner workflow first.** Terminate the inner dflow workflow before stopping any Cloud/MatMaster outer Bohrium node; otherwise compute can continue silently. Bohrium-direct has no outer node. See `reference/workflow-control.md`. +5. **Cloud/MatMaster MUST use `generate_config.py`; never hand-write `param.json` or `global.json`.** Installed local editions follow their selected profile and audited templates. - Create the complete job with `python /scripts/generate_config.py create ...`. - For multiple structures, pass repeated/space-separated `--structure` and/or `--structure-dir` to `create` (it copies each into `confs//` and fills `structures`); do not hand-edit `structures` after create. - To preserve an approved `param.json` while refreshing credentials, run `python /scripts/generate_config.py refresh-global --global global.json` from the task directory. This updates only `global.json`. - Do not invent unsupported flags or call the ticket API directly. -6. **Generate the ticket before packaging the job; never refresh it in `run.sh`.** +6. **Cloud/MatMaster only: generate the ticket before packaging; never refresh it in `run.sh`.** - First inspect the agent/local environment for `BOHRIUM_ACCESS_KEY`. - If it is missing, STOP and ask the user to provide/configure it. `generate_config.py` cannot generate a ticket without an access key. - If it exists, use `create` for a new job or `refresh-global` for an existing job; both convert the key to a fresh ticket and write it to `global.json`. + - **Ticket API**: `GET https://openapi.dp.tech/openapi/v1/ticket/get?accessKey=&expiration=` + - Header: `x-app-key: ""` (empty string) + - `expiration` 单位为**小时**,默认值 `168`(7 天)。`generate_config.py` 已内置此默认值。 + - 返回: `{"code": 0, "data": {"ticket": "UUID"}}` - Verify that `global.json` contains a non-empty `bohrium_config.ticket` before submission. - - `run.sh` must only install/verify APEX and call `apex submit`. Do not add ticket API calls or depend on `BOHRIUM_ACCESS_KEY` inside the APEX container. + - `run.sh` must only install/verify APEX and call `apex submit`. Do not add ticket API calls or depend on `BOHRIUM_ACCESS_KEY` inside the APEX container — 容器内没有此环境变量。 - Install with `python3 -m pip install --upgrade --no-cache-dir apex-flow`. See `reference/submission.md`. + - **Installed local edition exception:** follow + `variants/local/profiles/bohrium-direct.md`; configure AccessKey or + email/password with `apex account`. The saved AccessKey is passed to dflow + for short-lived ticket exchange, is masked by `apex account --show`, and is + not serialized into `global.json`. Do not require `BOHRIUM_ACCESS_KEY`, + `refresh-global`, or an outer job for that profile. 7. **Project ID from environment only.** `generate_config.py` reads `BOHRIUM_PROJECT_ID` (or `--project-id`). Never hardcode a project ID (including old examples like `13529`) into `global.json`, docs, or prompts. 8. **Hard-validate inside the task directory before every upload.** Run: ```bash @@ -124,26 +159,39 @@ Options to offer via AskQuestion: python /scripts/validate_inputs.py \ --param param.json --global global.json ``` - Do not upload or submit unless it prints `Validation PASSED` and reports both - `program_id` and `bohrium_config.project_id` with `type=int`. A quoted numeric - string is invalid. Upload the newly validated directory as a new outer job; - never retry an outer job whose input snapshot was invalid. + Do not upload or submit unless it prints `Validation PASSED`. In + Cloud/MatMaster ticket mode it must also report both `program_id` and + `bohrium_config.project_id` with `type=int`; a quoted numeric string is + invalid. Upload that profile's newly validated directory as a new outer job; + never retry an outer job whose input snapshot was invalid. Direct/local + profiles do not invent these outer-job checks. + For Gamma properties, also read the representative-slab report: parent/final + atom count, thickness, layer repeats, minimum distance, task count, and + generated KPOINTS. Stop on any explicit limit violation. 9. **Screen image × machine before submit.** Before writing `global.json` or submitting, run: ```bash python scripts/validate_apex_combo.py list-combos --backend lammps --prefer gpu python scripts/validate_apex_combo.py check \ - --image registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3 \ - --scass "c8_m31_1 * NVIDIA T4" + --image registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2 \ + --scass "c16_m120_1 * NVIDIA L20" ``` - Do **not** hardcode an unverified `scass_type`. Prefer `recommend` / `list-combos` output. Known failures include `deepmd-kit:3.1.0`, `3.1.1-cuda12.1`, `3.1.2`, the combination `deepmd-kit:3.1.1` × `NVIDIA T4`, `c4_m16_cpu`, and `c12_m46_1 * NVIDIA T4`. The default LAMMPS image is `registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3`; `apex submit` enforces it for LAMMPS phonon and Grüneisen workflows. -10. **MUST use the bundled frozen DPA model under** `models/` **for LAMMPS + DeePMD unless the user explicitly requests another compatible model.** The skill ships - `models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth`, a ready-to-run frozen - DPA-3.2-5M OMat24 model. Copy it into the job directory before generating - `param.json`. The multi-head source checkpoint is **not** in the skill zip — - fetch it only when the user explicitly needs another task head - (`scripts/fetch_models.py --source-checkpoint` or - `dp --pt pretrained download DPA-3.2-5M`) and freeze that head before use. - See `models/README.md`. + Do **not** hardcode an unverified `scass_type`. Prefer `recommend` / `list-combos` output. Known failures include `deepmd-kit:3.1.0`, `3.1.1-cuda12.1`, `3.1.2`, the combination `deepmd-kit:3.1.1` × `NVIDIA T4`, `c4_m16_cpu`, and `c12_m46_1 * NVIDIA T4`. GPU LAMMPS potentials (`deepmd`, `mace`, `nep`) use `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2` with NVIDIA L20 by default; RTX 4090 remains a validated compatible option. CPU LAMMPS potentials use `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post`; never pair 0.0.2 with a `*_cpu` machine because sequential CPU validation stalled before the container command started. `apex submit` only enforces 0.0.2 for GPU-potential phonon and Grüneisen workflows. Do not emit `plugin load libdeepmd_lmp.so` for 0.0.2. + Do **not** hardcode an unverified `scass_type`. Prefer `recommend` / `list-combos` output. GPU LAMMPS potentials (`deepmd`, `mace`, `nep`) use `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2` with NVIDIA L20 by default; RTX 4090 is also validated. CPU LAMMPS potentials use the APEX CPU image. Never pair 0.0.2 with a `*_cpu` machine, and do not emit `plugin load libdeepmd_lmp.so` for its integrated USER-DEEPMD build. + For DPA4, inspect the locked candidate matrix with + `list-combos --runtime-profile dpa4-alloytongqi-t4`; `recommend` must fail + until exact `ref@sha256` post-snapshot qualification is recorded. After + publication, only one rank/one GPU on exact `c4_m15_1 * NVIDIA T4` is + eligible. Other T4 SKUs (including c8/c16), every non-T4 GPU, CPU, + multi-rank, multi-GPU, and cross-architecture PT2 reuse remain unverified or + prohibited and therefore fail closed. V100/SM 7.0 and older GPUs, and an + NVIDIA Linux driver below 580.65.06, are prohibited by the CUDA 13 runtime. +10. **Treat the bundled DPA4 checkpoint as provenance, not a LAMMPS input.** The + skill ships only `models/DPA4-alloytongqi/model.pt`, the user-supplied + single-task `alloytongqi` checkpoint. Generate the image-resident contract + only with `generate_config.py create --runtime-profile + dpa4-alloytongqi-t4`; never hand-write its paths/hashes. The command remains + blocked until publication. No alternate-model downloader is bundled. See + `models/README.md`. 11. **STOP: Check atom count / cell size before property submit — decide whether to expand.** APEX does not require a conventional cell. Do not convert a primitive cell or user-provided supercell to a conventional cell merely because an example uses @@ -214,26 +262,72 @@ Options to offer via AskQuestion: required. Only after a confirmed image exists, pass `--vasp-image ` to `generate_config.py create` (writes `vasp_image_name`). -15. **MUST use a Bohrium-safe VASP `vasp_run_command` — never bare `vasp_std`.** +15. **MUST use a Bohrium-safe VASP `vasp_run_command`.** After a licensed VASP image is resolved (Rule 14), `global.json` should use a command that sources Intel oneAPI, raises stack limit, and calls an absolute binary. Typical Bohrium layout: ```text - bash -c "source /opt/intel/oneapi/setvars.sh && ulimit -s unlimited && mpirun -n /opt/vasp.5.4.4/bin/vasp_std" + bash -c "source && ulimit -s unlimited && mpirun -n " ``` Constraints: - Always `source /opt/intel/oneapi/setvars.sh` (Intel MPI / MKL env). - Always `ulimit -s unlimited` (avoids stack overflow on large cells). - - Prefer absolute binary path (PATH `vasp_std` is unreliable); adjust path if - the user-approved image differs. - - Align `` with `scass_type` CPU count (`c32_*` → `-n 32`, `c16_*` → `-n 16`). - - Do **not** use bare `mpirun -n 16 vasp_std`. + - Prefer an absolute `vasp_std`/`vasp_gam` binary path; adjust it for the + user-approved image. + - Align `` with the CPU count encoded by `scass_type`. + - APEX selects the executable per generated task: Gamma-centered `1x1x1` + uses `vasp_gam`; every other grid uses `vasp_std`. This applies to every + property and relaxation, not only Gamma/GammaSurface workflows. + `KGAMMA=True` alone is not proof—the generated `KPOINTS` is authoritative. + - Any task that resolves to `vasp_gam` requires `KPAR=1`. In general, + `KPAR` must divide ranks, and `NCORE` must divide ranks/KPAR. Do not + combine `NCORE` and `NPAR`; missing `NCORE` is a warning. `generate_config.py` writes the run_command template for `--backend vasp` and sets `vasp_image_name` only from `--vasp-image`. - - - -## Supported Properties (14 types) +16. **STOP: Before submitting `gamma_surface`, run `apex preview` to check for overlapping atoms.** + Disordered / RSS / non-standard cells often produce slab displacements with + unphysically close atom pairs. Preview generates the displacement POSCARs + and prints a text warning when any pair distance is `< 0.2` Å: + ```bash + apex preview param.json + ``` + - **Must check stderr** for exactly: + `Generated Gamma surface contains overlapping atoms.` + - If that line appears → **STOP**, report it to the user, and do not submit + until the slip system / cell / `supercell_size` / `closed_loop` choice is + fixed and preview is clean. + - **Do not open, read, or visually inspect the generated GIF.** The Agent + only needs the stderr overlap warning (or its absence). The GIF is for + optional human viewing only. + - For human-requested diagnostics, `--gif-view` accepts `auto`, `default`, + `slip-plane`, `parent-bc`, or `both`. `auto` is the default and writes the + slip-plane and parent-`bc` projections for both Gamma lines and Gamma + surfaces. The viewport retains projected unit-cell boundaries so vacuum + is not cropped from side-facing projections. These views do not replace + geometry validation. + Skip ONLY if the job has no `gamma_surface` property. +17. **Gamma uses current parent-aware geometry and 20 Å vacuum defaults.** For + RSS/SQS or other disordered cells classified as `other`, set + `parent_lattice` to `bcc`, `fcc`, or `hcp`; APEX interprets the Miller plane + and direction in that parent basis without symmetrizing the supplied cell. + Both `gamma` and `gamma_surface` default to `vacuum_size=20`. For endpoint + Gamma lines, use `displacement_points` (for example `[0.0, 0.5]`); values + must be unique, finite, within `[0,1]`, and include the zero-energy + reference. Read `gamma_geometry.json` and `slab_generation.json`; mapping, + layer split, minimum distance, and parent-translation topology fail closed. +18. **Melting restart and evidence boundaries.** `melting_point` is the + LAMMPS-only q6/interface-velocity workflow. If `cal_setting.restart_files` + is provided, supply exactly one existing file per temperature; APEX copies + the temperature-matched file into every replica as + `restart.coexistence.start` and forwards it. This is transport only—the + generated input does not automatically issue `read_restart`. + `finite_t_latt` never receives this file. A temperature contributes to a + bracket only when all configured, distinct replicas are present and agree; + missing/duplicate/mixed replicas remain `inconclusive`. + + + +## Supported Properties (15 types) | Type | JSON `type` value | Backend | Description | @@ -248,10 +342,11 @@ Options to offer via AskQuestion: | Gamma surface | `gamma_surface` | All | 2D GSFE map | | Cohesive | `cohesive` | All | Cohesive energy curve | | Decohesive | `decohesive` | All | Ideal work of separation | -| Finite-T lattice | `finite_t_latt` | All | Lattice parameter vs temperature (NPT MD) | +| Finite-T lattice | `finite_t_latt` | LAMMPS/VASP | Lattice parameter vs temperature (NPT MD) | | Finite-T elastic | `finite_t_elastic` | LAMMPS only | Elastic constants at finite temperature | | Grüneisen | `gruneisen` | All | Grüneisen parameters & thermal expansion | -| Annealing | `annealing` | All | Heat-hold-quench MD cycle | +| Annealing | `annealing` | LAMMPS/VASP | Heat-hold-quench MD cycle | +| Melting point | `melting_point` | LAMMPS only | Two-phase coexistence melting bracket | > See `reference/properties.md` for full parameter details of each property. @@ -263,7 +358,7 @@ Options to offer via AskQuestion: | `interaction.type` | pair_style | Model file | GPU? | | ------------------ | ------------------------------ | --------------- | ---- | -| `deepmd` | `deepmd` | `.pb` or `.pth` | Yes | +| `deepmd` | `deepmd` | `.pb`, `.pth`, or compatible single-task `.pt` | Yes | | `mace` | `mace no_domain_decomposition` | `.model` | Yes | | `nep` | `nep` | `nep.txt` | Yes | | `eam_alloy` | `eam/alloy` | `.eam.alloy` | No | @@ -307,10 +402,14 @@ See `reference/submission.md` for the full validated template. ## Key Additional Rules -1. **LAMMPS-only properties**: `finite_t_elastic` only works with LAMMPS. `finite_t_latt` and `annealing` also support VASP Langevin–Parrinello–Rahman NpT and ABACUS Nose–Hoover-style NpT. -2. **Model files must be in job directory.** For MLIP workflows, the model file (`.pb`, `.pth`, `.model`, etc.) must be present in the submitted directory. Use relative paths in `param.json`. For DeePMD/DPA, copy `models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth`. Default to `"type_map": "auto"` for every LAMMPS interaction; specify a dictionary only when the user explicitly needs a fixed custom ordering. +1. **Finite-temperature backend limits**: `finite_t_elastic` and `melting_point` are LAMMPS-only. `finite_t_latt` and `annealing` support LAMMPS and VASP, but not ABACUS. VASP uses `MDALGO=3` and requires a binary compiled with `-Dtbdyn`; annealing `protocol="coexistence"` is a fixed-temperature equilibration plus production run and is not the q6/interface-velocity `melting_point` method. Only `melting_point` transports `restart_files` as `restart.coexistence.start`. +2. **Model paths must match the runtime contract.** Ordinary MLIP model files + must be in the job directory and use relative paths. The DPA4 T4 profile is + the exception: do not stage or execute the bundled `.pt`; its exact contract + uses a hashed image-resident `.pt2` and is generated only after publication. + Default to `"type_map": "auto"`. 3. **Joint workflow recommended.** Use `joint` flow (relaxation + properties) for most use cases to ensure proper relaxation before property calculations. -4. **GPU for ML potentials.** DeePMD, MACE, and NEP benefit from GPU acceleration. Set `scass_type` to a validated GPU SKU from `validate_apex_combo.py recommend --prefer gpu` (default: `"c8_m31_1 * NVIDIA T4"`). +4. **GPU for ML potentials.** DeePMD, MACE, and NEP benefit from GPU acceleration. Set `scass_type` to a validated GPU SKU from `validate_apex_combo.py recommend --prefer gpu` (default: `"c16_m120_1 * NVIDIA L20"`; RTX 4090 remains compatible). 5. **Supercell sizing depends on the input atom count, not only the default JSON.** Treat defaults as targets for **unit-cell inputs**. First inspect the user's structure; if it is already large enough, prefer `[1,1,1]` after confirmation. @@ -320,7 +419,7 @@ See `reference/submission.md` for the full validated template. - surface / gamma / decohesive: ensure slab thickness / in-plane size, not bulk supercell If the cell is too small, ask the user to expand before submit. See `reference/workflow-control.md`. -6. **Outer job machine.** Use `c1_m2_cpu` for the outer Bohrium job since it only calls `apex submit` and waits. Don't waste larger CPU or GPU resources on the submission client. +6. **Cloud/MatMaster outer job machine.** Use `c1_m2_cpu` for that profile's outer Bohrium job since it only calls `apex submit` and waits. The installed Bohrium-direct profile has no outer job. @@ -362,7 +461,6 @@ Successfully validated workflow (ID: `cu-fcc-elastic-v3-joint-sdfml`): | `generate_config.py` | `create` a complete job or `refresh-global` credentials without changing param.json | | `list_bohrium_images.py` | List private Bohrium images by keyword (MatMaster `list_images` equivalent) | | `validate_apex_combo.py` | List / check / recommend safe image × scass_type combos | -| `fetch_models.py` | Optional: download the DPA-3.2-5M multi-head source `.pt` for freezing another head | | `parse_results.py` | Parse APEX output into summary | | `validate_inputs.py` | Validate configuration before submission | @@ -374,12 +472,10 @@ Successfully validated workflow (ID: `cu-fcc-elastic-v3-joint-sdfml`): | File | Content | | -------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | -| `reference/submission.md` | Authentication (ticket API + refresh), run.sh template, Bohrium config (images/machines), global.json template, RFC 1123 naming, submission lifecycle | +| `reference/submission.md` | Cloud/MatMaster outer-client authentication (ticket API + refresh), run.sh template, Bohrium config, RFC 1123 naming, and lifecycle; installed local editions use their profile-specific reference | | `reference/workflow-control.md` | Running-task status/count format, live Argo link, stopping/killing procedure, and structure validation | -| `reference/properties.md` | Complete parameter reference for all 14 property types | +| `reference/properties.md` | Complete parameter reference for all 15 property types | | `reference/calculators.md` | Detailed backend configuration (VASP, ABACUS, LAMMPS) | | `reference/lammps_potentials.md` | LAMMPS potential type details and examples | | `reference/rss_workflow.md` | RSS structure generation workflow | | `reference/examples.md` | Complete worked examples for common scenarios | - - diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/README.md b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/README.md new file mode 100644 index 00000000..e237d174 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/README.md @@ -0,0 +1,264 @@ +# DPA4 alloytongqi small-cell compatibility benchmark + +This directory is a reproducible, fail-closed runtime benchmark for the exact +bundled training checkpoint: + +- model: `../../models/DPA4-alloytongqi/model.pt` +- SHA-256: `c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad` +- benchmark ID: `dpa4-alloytongqi-small-cell-gpu-compat-v2` + +The `.pt` checkpoint is **never** passed to LAMMPS. The current DPA4 LAMMPS +backend accepts a device-specific AOTI `.pt2` runtime artifact; CPU and T4 GPU +legs therefore require independently frozen `.pt2` files with distinct hashes. + +It tests checkpoint inspection, the DeepMD LAMMPS plugin, short LAMMPS `run 0` and NVE +MD, CPU–GPU numerical parity, and the APEX-shaped phonoLAMMPS path. It does not +claim that a tiny two-atom cell is scientifically converged, and it does not by +itself prove a complete dflow/Bohrium APEX workflow. + +`manifest.example.json` is intentionally `untested`. Do not change it to +`passed`; create a new evidence workspace and let `build_manifest.py` derive the +status from files produced by actual execution. + +## Evidence states + +- `untested`: no runtime command has produced evidence. +- `inconclusive`: a command may have run, but image digest, package version, + GPU-use proof, or another required identity field is absent. +- `failed`: an executed command failed, emitted a fatal signature, produced no + parseable finite observables or complete finite `FORCE_CONSTANTS`, or exceeded + a parity gate. +- `passed`: all 12 LAMMPS runs, all six CPU–GPU comparisons, and the Ti T4 GPU + phonoLAMMPS smoke passed with a full image digest, one consistent CPU `.pt2`, + one consistent T4 `.pt2`, distinct runtime hashes, and complete evidence. + +Only `passed` is positive compatibility evidence. Never promote an +`inconclusive` run based on a successful shell exit alone. + +## What is generated + +`generate_cases.py` writes independent POSCAR, LAMMPS data, and input files; +it does not depend on `tests/confs/std-fcc` or any other test fixture. + +| Case | Atoms | Purpose | +| --- | ---: | --- | +| `Ti_hcp` | 2 | single-element hcp model load; 2×2×2 phonoLAMMPS smoke | +| `V_bcc` | 2 | single-element bcc model load | +| `TiV_B2` | 2 | mixed Ti/V type mapping in the B2 prototype | + +The NVE input uses 20 steps at 1 fs only as a deterministic execution smoke. +It is not a thermodynamic or stability benchmark. + +Every generated DPA4 LAMMPS input contains `atom_modify map yes` immediately +after `atom_style atomic` and before `read_data`. This is required by the +single-rank periodic DPA4 graph path; removing or moving it invalidates the +benchmark. + +The generated inputs deliberately contain no `plugin load` command. The exact +b95 `BUILD_PY_IF=ON` build installs `libdeepmd_lmpplugin.so`; the candidate +image must expose its containing directory through `LAMMPS_PLUGIN_PATH` so +LAMMPS auto-loads it. The runner rejects a missing/invalid path or a path that +does not contain that exact plugin filename. + +## Required runtime + +Run these scripts **inside the candidate image**, using that image's Python and +executables. The image must provide: + +- DeepMD-kit with DPA4 support; +- LAMMPS plus `libdeepmd_lmpplugin.so` discoverable through + `LAMMPS_PLUGIN_PATH` auto-loading; +- `phonolammps`, `phonopy`, Python `lammps`, `dpdata`, PyTorch, and NumPy; +- `nvidia-smi` on GPU runs. + +The scripts never install packages. They record package metadata for +`deepmd-kit`, `phonolammps`, `phonopy`, `lammps`, `dpdata`, `torch`, and +`numpy`; a missing version keeps the result non-passing. + +This v2 qualification is intentionally limited to one selected NVIDIA T4 +(compute capability 7.5) and one LAMMPS rank with deterministic thread settings (`OMP_NUM_THREADS=1`, DeepMD +intra/inter-op threads = 1). The runners reject a non-T4 selected GPU and an MPI +launcher that does not explicitly request exactly one rank. CPU and GPU parity +must use the same exact image digest and checkpoint hash but different +device-specific `.pt2` hashes. Run the CPU leg on the same T4 node/image; the +script hides GPUs with an empty `CUDA_VISIBLE_DEVICES` and fails if its process +is nevertheless observed on a GPU. + +## Reproducible procedure + +Start in this benchmark directory. Use a new workspace name for every image × +GPU qualification; scripts refuse to overwrite evidence. + +```bash +BENCH_ROOT="$PWD" +WORKSPACE="$BENCH_ROOT/workspace" +CHECKPOINT="$BENCH_ROOT/../../models/DPA4-alloytongqi/model.pt" +CPU_PT2="/absolute/path/from-cpu-freeze/model.cpu.pt2" +GPU_PT2="/absolute/path/from-this-t4-freeze/model.t4.pt2" +PLUGIN_DIR="/absolute/path/containing/libdeepmd_lmpplugin.so" +IMAGE_REF="registry.example.invalid/owner/dpa4-apex:replace-me" +IMAGE_DIGEST="sha256:replace-with-64-lowercase-hex-characters" + +python generate_cases.py --output "$WORKSPACE/cases" +sha256sum "$CHECKPOINT" "$CPU_PT2" "$GPU_PT2" +dp --pt show "$CHECKPOINT" type-map descriptor fitting-net size +``` + +Generate both `.pt2` artifacts with the candidate image's DPA4 AOTI freeze +implementation: freeze the CPU artifact with GPUs hidden, then freeze the GPU +artifact on the exact T4 being qualified. Do not rename a `.pt` checkpoint to +`.pt2`, reuse the CPU artifact on GPU, or reuse an artifact frozen for another +GPU family. The runner copies and hashes the supplied `.pt2`; it does not infer +device compatibility from the suffix. + +The checkpoint hash must equal the value at the top of this document. The +attribute-bearing `dp --pt show` command above must exit zero; a bare +`dp show ` is not this probe and may be rejected by the current +parser because it supplies no attributes. A tag alone +is not an image identity: provide the registry-resolved `sha256:` digest. If the +digest is missing, the scripts still preserve diagnostic output but set the +result to `inconclusive`. + +Run the complete CPU/GPU × `run0`/`md` matrix: + +```bash +for case in Ti_hcp V_bcc TiV_B2; do + for device in cpu gpu; do + if [ "$device" = cpu ]; then + runtime_model="$CPU_PT2" + else + runtime_model="$GPU_PT2" + fi + for mode in run0 md; do + python run_lammps_case.py \ + --case-dir "$WORKSPACE/cases/$case" \ + --checkpoint "$CHECKPOINT" \ + --runtime-model "$runtime_model" \ + --mode "$mode" \ + --device "$device" \ + --output "$WORKSPACE/runs/$case/$device/$mode" \ + --image-ref "$IMAGE_REF" \ + --image-digest "$IMAGE_DIGEST" \ + --lammps-plugin-path "$PLUGIN_DIR" \ + --command-json '["lmp"]' + done + done +done +``` + +For a launcher, pass a shell-free JSON array, for example +`--command-json '["mpirun","-n","1","lmp"]'`. Do not put a shell pipeline or +credential in the command array. Multi-rank launchers are rejected. + +Compare both run modes for all cases: + +```bash +for case in Ti_hcp V_bcc TiV_B2; do + for mode in run0 md; do + python compare_parity.py \ + --cpu "$WORKSPACE/runs/$case/cpu/$mode/result.json" \ + --gpu "$WORKSPACE/runs/$case/gpu/$mode/result.json" \ + --output "$WORKSPACE/parity/$case/$mode.json" + done +done +``` + +Default absolute gates are: + +| Observable | Gate | +| --- | ---: | +| energy per atom | ≤ `1e-4 eV/atom` | +| force component RMS | ≤ `1e-3 eV/Å` | +| maximum absolute force component | ≤ `5e-3 eV/Å` | +| maximum absolute stress component | ≤ `0.01 GPa` | + +LAMMPS `metal` pressure is recorded in bar and converted to GPa by the parity +script. The scripts reject missing or non-finite energy, stress, or force data. + +Run the phonoLAMMPS smoke on the Ti hcp case. This benchmark keeps `read_data`, +relies on b95 `LAMMPS_PLUGIN_PATH` auto-loading, truncates the ordinary input +after `pair_coeff`, passes a POSCAR, uses `--dim 2 2 2`, and expands +`PRIMITIVE_AXES=P` to the identity matrix. It does not change APEX's legacy +phonon input path for the old image: + +```bash +python run_phonolammps_smoke.py \ + --case-dir "$WORKSPACE/cases/Ti_hcp" \ + --checkpoint "$CHECKPOINT" \ + --runtime-model "$GPU_PT2" \ + --device gpu \ + --output "$WORKSPACE/phonolammps/Ti_hcp/gpu" \ + --image-ref "$IMAGE_REF" \ + --image-digest "$IMAGE_DIGEST" \ + --lammps-plugin-path "$PLUGIN_DIR" \ + --command-json '["phonolammps"]' +``` + +The smoke passes only when the command exits zero, no fatal signature is found, +GPU use is directly observed, and `FORCE_CONSTANTS` parses completely: its atom +count must equal the POSCAR atom count multiplied by the requested supercell, +all `N^2` indexed 3×3 blocks must be present in order, every matrix value must +be finite, and no trailing non-empty data is allowed. + +Finally aggregate and verify all evidence: + +```bash +python build_manifest.py \ + --workspace "$WORKSPACE" \ + --checkpoint "$CHECKPOINT" \ + --image-ref "$IMAGE_REF" \ + --image-digest "$IMAGE_DIGEST" \ + --output "$WORKSPACE/manifest.json" + +python verify_manifest.py \ + "$WORKSPACE/manifest.json" \ + --root "$WORKSPACE" +``` + +`verify_manifest.py` re-hashes every declared workspace artifact and enforces +the pass matrix. It uses only Python's standard library. `manifest.schema.json` +is the machine-readable Draft 2020-12 schema; schema validation is an optional +additional check, not a substitute for hash verification. + +## Recorded evidence + +Each LAMMPS and phonoLAMMPS result records: + +- exact training-checkpoint `model.pt` SHA-256 and size; +- exact device-specific runtime `.pt2` SHA-256, byte size, requested device, + and every run record that used it; +- image reference and immutable digest; +- GPU name, UUID, memory, compute capability, NVIDIA driver, and CUDA version; +- effective `LAMMPS_PLUGIN_PATH` plus the exact + `libdeepmd_lmpplugin.so` SHA-256 and size; +- relevant package versions and resolved executable paths; +- command, selected thread/device environment, start/end time, timeout and exit + code; +- attribute-bearing `dp --pt show type-map descriptor fitting-net + size` exit code and complete stdout/stderr before LAMMPS is allowed to start; +- complete stdout/stderr files and their hashes; +- child/descendant GPU-process samples from `nvidia-smi`; +- finite energy/stress/forces or a fully parsed, dimension-consistent, finite + `FORCE_CONSTANTS`; +- hashes and byte sizes for all generated inputs and outputs. + +Short runs can finish before `nvidia-smi` observes their process. That is not a +pass: it is `inconclusive`. Increase the workload in a new, versioned benchmark +rather than editing an existing evidence directory or manually changing the +status. + +## GPU recommendation and prohibition boundary + +Use a passed manifest as one piece of evidence for the exact image digest, +checkpoint hash, CPU/T4 `.pt2` hashes, T4 compute capability, driver, CUDA, +single-rank command, and thread tuple. Do not generalize this v1 benchmark to +another GPU or mutable image tag. A single transient +failure is not enough to prohibit a GPU combination: reproduce a deterministic +hard failure in two fresh workspaces and preserve both manifests. Missing or +mixed evidence is `untested`/`inconclusive`, not “compatible” and not +“forbidden.” + +Before calling the image ready for latest APEX, also run a real APEX make → run +→ post workflow with that exact digest. This benchmark verifies the underlying +runtime and APEX-shaped phonoLAMMPS command, not cloud orchestration, scheduling, +result retrieval, or scientific convergence. diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/benchmark_lib.py b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/benchmark_lib.py new file mode 100644 index 00000000..3ad2675f --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/benchmark_lib.py @@ -0,0 +1,538 @@ +#!/usr/bin/env python3 +"""Shared, dependency-free helpers for the DPA4 alloytongqi benchmark.""" + +from __future__ import annotations + +import hashlib +import importlib.metadata +import json +import math +import os +import re +import signal +import shutil +import subprocess +import sys +import threading +import time +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Sequence + + +SCHEMA_VERSION = "2.0.0" +BENCHMARK_ID = "dpa4-alloytongqi-small-cell-gpu-compat-v2" +EXPECTED_MODEL_SHA256 = ( + "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad" +) +SHA256_RE = re.compile(r"^sha256:[0-9a-f]{64}$") +HEX_SHA256_RE = re.compile(r"^[0-9a-f]{64}$") +REQUIRED_PACKAGES = ( + "deepmd-kit", + "phonolammps", + "phonopy", + "lammps", + "dpdata", + "torch", + "numpy", +) +FATAL_OUTPUT_PATTERNS = ( + re.compile(r"Unknown model type", re.IGNORECASE), + re.compile(r"(?:^|\n)\s*ERROR(?:\s|:)", re.IGNORECASE), + re.compile(r"Segmentation fault", re.IGNORECASE), + re.compile(r"CUDA error", re.IGNORECASE), + re.compile(r"lost atoms", re.IGNORECASE), + re.compile(r"Traceback \(most recent call last\)", re.IGNORECASE), +) + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") + + +def sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def describe_file(path: Path, root: Path | None = None) -> dict[str, Any]: + resolved = path.resolve() + display = str(resolved) + if root is not None: + try: + display = str(resolved.relative_to(root.resolve())) + except ValueError: + pass + return { + "path": display, + "sha256": sha256_file(resolved), + "bytes": resolved.stat().st_size, + } + + +def load_json(path: Path) -> dict[str, Any]: + with path.open("r", encoding="utf-8") as handle: + value = json.load(handle) + if not isinstance(value, dict): + raise ValueError(f"JSON root must be an object: {path}") + return value + + +def write_json_new(path: Path, value: dict[str, Any]) -> None: + """Write a JSON file, refusing to replace any existing evidence.""" + + path.parent.mkdir(parents=True, exist_ok=True) + if path.exists(): + raise FileExistsError(f"refusing to overwrite existing evidence: {path}") + text = json.dumps(value, indent=2, sort_keys=True, ensure_ascii=False) + "\n" + path.write_text(text, encoding="utf-8") + + +def write_text_if_identical(path: Path, text: str) -> None: + """Create deterministic input, or accept an already-identical file.""" + + path.parent.mkdir(parents=True, exist_ok=True) + if path.exists(): + if path.read_text(encoding="utf-8") != text: + raise FileExistsError(f"existing generated input differs: {path}") + return + path.write_text(text, encoding="utf-8") + + +def normalize_image_identity( + image_ref: str | None, image_digest: str | None +) -> tuple[dict[str, Any], list[str]]: + reasons: list[str] = [] + reference = (image_ref or "").strip() or None + digest = (image_digest or "").strip().lower() or None + if reference and "@sha256:" in reference: + ref_digest = "sha256:" + reference.rsplit("@sha256:", 1)[1].lower() + if digest and digest != ref_digest: + reasons.append("IMAGE_DIGEST_CONFLICT") + digest = digest or ref_digest + if reference is None: + reasons.append("IMAGE_REFERENCE_MISSING") + if digest is None or not SHA256_RE.fullmatch(digest): + reasons.append("IMAGE_DIGEST_MISSING_OR_INVALID") + return {"reference": reference, "digest": digest}, reasons + + +def package_versions() -> dict[str, str | None]: + versions: dict[str, str | None] = {} + for name in REQUIRED_PACKAGES: + try: + versions[name] = importlib.metadata.version(name) + except importlib.metadata.PackageNotFoundError: + versions[name] = None + return versions + + +def _run_probe(command: Sequence[str], timeout: float = 8.0) -> dict[str, Any]: + try: + completed = subprocess.run( + list(command), + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + timeout=timeout, + check=False, + ) + return { + "command": list(command), + "exit_code": completed.returncode, + "stdout": completed.stdout, + "stderr": completed.stderr, + } + except (FileNotFoundError, subprocess.TimeoutExpired) as exc: + return { + "command": list(command), + "exit_code": None, + "stdout": "", + "stderr": f"{type(exc).__name__}: {exc}", + } + + +def capture_gpu_inventory() -> tuple[list[dict[str, Any]], dict[str, Any]]: + fields = "index,name,uuid,driver_version,memory.total,compute_cap" + probe = _run_probe( + [ + "nvidia-smi", + f"--query-gpu={fields}", + "--format=csv,noheader,nounits", + ] + ) + inventory: list[dict[str, Any]] = [] + if probe["exit_code"] == 0: + for line in probe["stdout"].splitlines(): + parts = [part.strip() for part in line.split(",")] + if len(parts) != 6: + continue + index, name, uuid, driver, memory_mib, compute_capability = parts + try: + memory_value: int | None = int(memory_mib) + except ValueError: + memory_value = None + inventory.append( + { + "index": index, + "name": name, + "uuid": uuid, + "driver_version": driver, + "memory_mib": memory_value, + "compute_capability": compute_capability, + } + ) + header_probe = _run_probe(["nvidia-smi"]) + header = header_probe.get("stdout", "") + header_probe.get("stderr", "") + match = re.search(r"CUDA Version:\s*([0-9.]+)", header) + probe["cuda_version_from_nvidia_smi"] = match.group(1) if match else None + probe["header_exit_code"] = header_probe.get("exit_code") + probe["header_stderr"] = header_probe.get("stderr", "") + return inventory, probe + + +def capture_runtime_environment() -> dict[str, Any]: + inventory, nvidia_probe = capture_gpu_inventory() + drivers = sorted( + { + item["driver_version"] + for item in inventory + if item.get("driver_version") + } + ) + nvcc_probe = _run_probe(["nvcc", "--version"]) + nvcc_text = nvcc_probe.get("stdout", "") + nvcc_probe.get("stderr", "") + nvcc_match = re.search(r"release\s+([0-9.]+)", nvcc_text) + torch_cuda_probe = _run_probe( + [ + sys.executable, + "-c", + "import torch; print(torch.version.cuda or '')", + ] + ) + torch_cuda = ( + torch_cuda_probe.get("stdout", "").strip() + if torch_cuda_probe.get("exit_code") == 0 + else None + ) + return { + "gpu_inventory": inventory, + "driver_version": drivers[0] if len(drivers) == 1 else None, + "cuda_version": nvidia_probe.get("cuda_version_from_nvidia_smi"), + "cuda_versions": { + "driver_supported": nvidia_probe.get("cuda_version_from_nvidia_smi"), + "toolkit_nvcc": nvcc_match.group(1) if nvcc_match else None, + "torch_built_against": torch_cuda or None, + }, + "cuda_probes": { + "nvcc": nvcc_probe, + "torch": torch_cuda_probe, + }, + "packages": package_versions(), + "executables": { + name: shutil.which(name) + for name in ("python", "dp", "lmp", "phonolammps", "nvidia-smi") + }, + "nvidia_smi_probe": nvidia_probe, + "selected_environment": { + "CONDA_DEFAULT_ENV": os.environ.get("CONDA_DEFAULT_ENV"), + "CUDA_VISIBLE_DEVICES": os.environ.get("CUDA_VISIBLE_DEVICES"), + "LAMMPS_PLUGIN_PATH": os.environ.get("LAMMPS_PLUGIN_PATH"), + "DP_INTRA_OP_PARALLELISM_THREADS": os.environ.get( + "DP_INTRA_OP_PARALLELISM_THREADS" + ), + "DP_INTER_OP_PARALLELISM_THREADS": os.environ.get( + "DP_INTER_OP_PARALLELISM_THREADS" + ), + "OMP_NUM_THREADS": os.environ.get("OMP_NUM_THREADS"), + }, + } + + +def _descendant_pids(root_pid: int) -> set[int]: + """Return root plus Linux /proc descendants; root-only elsewhere.""" + + descendants = {root_pid} + proc = Path("/proc") + if not proc.is_dir(): + return descendants + parent_map: dict[int, int] = {} + for status_path in proc.glob("[0-9]*/status"): + try: + pid = int(status_path.parent.name) + ppid = None + for line in status_path.read_text(encoding="utf-8", errors="replace").splitlines(): + if line.startswith("PPid:"): + ppid = int(line.split()[1]) + break + if ppid is not None: + parent_map[pid] = ppid + except (OSError, ValueError, IndexError): + continue + changed = True + while changed: + changed = False + for pid, ppid in parent_map.items(): + if ppid in descendants and pid not in descendants: + descendants.add(pid) + changed = True + return descendants + + +def _gpu_compute_processes() -> list[dict[str, Any]]: + probe = _run_probe( + [ + "nvidia-smi", + "--query-compute-apps=pid,process_name,gpu_uuid,used_gpu_memory", + "--format=csv,noheader,nounits", + ], + timeout=2.0, + ) + if probe["exit_code"] != 0: + return [] + rows: list[dict[str, Any]] = [] + for line in probe["stdout"].splitlines(): + parts = [part.strip() for part in line.split(",")] + if len(parts) != 4: + continue + try: + pid = int(parts[0]) + except ValueError: + continue + try: + memory_mib: int | None = int(parts[3]) + except ValueError: + memory_mib = None + rows.append( + { + "pid": pid, + "process_name": parts[1], + "gpu_uuid": parts[2], + "used_gpu_memory_mib": memory_mib, + } + ) + return rows + + +def run_monitored( + command: Sequence[str], + cwd: Path, + env: dict[str, str], + timeout: float, +) -> dict[str, Any]: + process = subprocess.Popen( + list(command), + cwd=str(cwd), + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + start_new_session=True, + ) + observed: list[dict[str, Any]] = [] + stop = threading.Event() + + def monitor() -> None: + while not stop.is_set(): + family = _descendant_pids(process.pid) + for row in _gpu_compute_processes(): + if row["pid"] in family: + sample = dict(row) + sample["observed_at"] = utc_now() + observed.append(sample) + stop.wait(0.05) + + thread = threading.Thread(target=monitor, name="gpu-process-monitor", daemon=True) + thread.start() + # AOTI model loading is brief enough that a process can disappear between + # two 50 ms monitor ticks. Capture the root process immediately as well. + family = _descendant_pids(process.pid) + for row in _gpu_compute_processes(): + if row["pid"] in family: + sample = dict(row) + sample["observed_at"] = utc_now() + observed.append(sample) + timed_out = False + try: + stdout, stderr = process.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + timed_out = True + try: + os.killpg(process.pid, signal.SIGKILL) + except (ProcessLookupError, PermissionError): + process.kill() + stdout, stderr = process.communicate() + finally: + stop.set() + thread.join(timeout=3.0) + unique_samples: dict[tuple[Any, ...], dict[str, Any]] = {} + for sample in observed: + key = ( + sample.get("pid"), + sample.get("gpu_uuid"), + sample.get("used_gpu_memory_mib"), + ) + unique_samples[key] = sample + return { + "exit_code": process.returncode, + "timed_out": timed_out, + "stdout": stdout, + "stderr": stderr, + "gpu_process_samples": list(unique_samples.values()), + "root_pid": process.pid, + } + + +def detect_fatal_output(stdout: str, stderr: str) -> list[str]: + combined = stdout + "\n" + stderr + reasons = [] + for pattern in FATAL_OUTPUT_PATTERNS: + if pattern.search(combined): + reasons.append(f"FATAL_OUTPUT:{pattern.pattern}") + return reasons + + +def finite_number(value: Any) -> bool: + return isinstance(value, (int, float)) and math.isfinite(float(value)) + + +def identity_reasons( + checkpoint_path: Path, image_ref: str | None, image_digest: str | None +) -> tuple[dict[str, Any], list[str]]: + checkpoint = describe_file(checkpoint_path) + reasons: list[str] = [] + if checkpoint["sha256"] != EXPECTED_MODEL_SHA256: + reasons.append("CHECKPOINT_SHA256_MISMATCH") + image, image_reasons = normalize_image_identity(image_ref, image_digest) + reasons.extend(image_reasons) + return {"checkpoint_pt": checkpoint, "image": image}, reasons + + +def runtime_model_identity( + runtime_model: Path, device: str, root: Path | None = None +) -> tuple[dict[str, Any], list[str]]: + descriptor = describe_file(runtime_model, root) + descriptor["device"] = device + reasons: list[str] = [] + if runtime_model.suffix.lower() != ".pt2": + reasons.append("RUNTIME_MODEL_MUST_BE_PT2") + if descriptor["bytes"] <= 0: + reasons.append("RUNTIME_MODEL_EMPTY") + return descriptor, reasons + + +def resolve_lammps_plugin_path( + explicit: str | None, inherited: str | None +) -> tuple[str | None, list[dict[str, Any]], list[str]]: + """Resolve the b95 plugin directory contract without modifying the image.""" + + raw = (explicit if explicit is not None else inherited) or "" + raw = raw.strip() + if not raw: + return None, [], ["LAMMPS_PLUGIN_PATH_MISSING"] + directories: list[Path] = [] + reasons: list[str] = [] + for entry in raw.split(os.pathsep): + entry = entry.strip() + if not entry: + reasons.append("LAMMPS_PLUGIN_PATH_EMPTY_ENTRY") + continue + directory = Path(entry).expanduser().resolve() + if not directory.is_dir(): + reasons.append(f"LAMMPS_PLUGIN_DIRECTORY_MISSING:{directory}") + continue + directories.append(directory) + plugins = [] + for directory in directories: + candidate = directory / "libdeepmd_lmpplugin.so" + if candidate.is_file(): + plugins.append(describe_file(candidate)) + if not plugins: + reasons.append("LIBDEEPMD_LMPPLUGIN_SO_NOT_FOUND") + normalized = os.pathsep.join(str(directory) for directory in directories) + return normalized or None, plugins, reasons + + +def single_rank_command_reasons( + command: Sequence[str], allowed_programs: set[str] +) -> list[str]: + """Reject launchers unless they explicitly request exactly one rank.""" + + if not command: + return ["COMMAND_EMPTY"] + launcher = Path(command[0]).name.lower() + if launcher not in {"mpirun", "mpiexec", "srun"}: + program = launcher + return [] if program in allowed_programs else [f"PROGRAM_NOT_ALLOWED:{program}"] + rank_values: list[str] = [] + for index, token in enumerate(command): + if token in {"-n", "-np", "--np", "--ntasks"}: + if index + 1 >= len(command): + return ["MPI_RANK_ARGUMENT_MISSING"] + rank_values.append(command[index + 1]) + elif token.startswith("--ntasks=") or token.startswith("--np="): + rank_values.append(token.split("=", 1)[1]) + if not rank_values: + return ["SINGLE_RANK_NOT_EXPLICIT"] + if any(value != "1" for value in rank_values): + return ["MULTI_RANK_NOT_ALLOWED"] + program = Path(command[-1]).name.lower() + if program not in allowed_programs: + return [f"PROGRAM_NOT_ALLOWED:{program}"] + return [] + + +def t4_environment_reasons( + environment: dict[str, Any], gpu_index: str +) -> list[str]: + inventory = environment.get("gpu_inventory", []) + selected = [item for item in inventory if str(item.get("index")) == str(gpu_index)] + if len(selected) != 1: + return ["T4_GPU_NOT_UNIQUELY_IDENTIFIED"] + name = str(selected[0].get("name", "")).strip() + if not re.search(r"(?:^|\s)T4$", name, re.IGNORECASE): + return [f"GPU_NOT_T4:{name or 'unknown'}"] + reasons = [] + for field in ("uuid", "driver_version", "compute_capability"): + if not str(selected[0].get(field, "")).strip(): + reasons.append(f"T4_{field.upper()}_MISSING") + if selected[0].get("memory_mib") is None: + reasons.append("T4_MEMORY_MIB_MISSING") + if str(selected[0].get("compute_capability", "")).strip() != "7.5": + reasons.append("T4_COMPUTE_CAPABILITY_NOT_7_5") + return reasons + + +def t4_environment_reasons_for_device( + environment: dict[str, Any], gpu_index: str, device: str +) -> list[str]: + """Require an exact T4 only for GPU legs. + + CPU parity intentionally runs in the same candidate image on the same T4 + node with ``CUDA_VISIBLE_DEVICES`` empty. Its hardware inventory is useful + provenance, but the CPU command must not be rejected merely because the + host still exposes a T4 to ``nvidia-smi``. + """ + + if device == "cpu": + return [] + return t4_environment_reasons(environment, gpu_index) + + +def missing_package_reasons(environment: dict[str, Any]) -> list[str]: + packages = environment.get("packages", {}) + return [ + f"PACKAGE_VERSION_MISSING:{name}" + for name in REQUIRED_PACKAGES + if not packages.get(name) + ] + + +def canonical_hash(value: Any) -> str: + encoded = json.dumps(value, sort_keys=True, separators=(",", ":")).encode( + "utf-8" + ) + return hashlib.sha256(encoded).hexdigest() diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/build_manifest.py b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/build_manifest.py new file mode 100644 index 00000000..5d869482 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/build_manifest.py @@ -0,0 +1,375 @@ +#!/usr/bin/env python3 +"""Aggregate benchmark evidence into a fail-closed manifest.""" + +from __future__ import annotations + +import argparse +import subprocess +from pathlib import Path +from typing import Any, Iterable, Sequence + +from benchmark_lib import ( + BENCHMARK_ID, + EXPECTED_MODEL_SHA256, + SCHEMA_VERSION, + canonical_hash, + describe_file, + identity_reasons, + load_json, + utc_now, + write_json_new, +) +from compare_parity import DEFAULT_THRESHOLDS + + +EXPECTED_CASES = ("Ti_hcp", "V_bcc", "TiV_B2") +EXPECTED_MODES = ("run0", "md") +EXPECTED_DEVICES = ("cpu", "gpu") + + +def _git_provenance(script_dir: Path) -> dict[str, Any]: + def call(*args: str) -> tuple[int, str]: + completed = subprocess.run( + ["git", "-C", str(script_dir), *args], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + check=False, + ) + return completed.returncode, completed.stdout.strip() + + commit_rc, commit = call("rev-parse", "HEAD") + status_rc, status = call("status", "--porcelain") + return { + "apex_git_commit": commit if commit_rc == 0 else None, + "apex_git_dirty": bool(status) if status_rc == 0 else None, + } + + +def _load_records(paths: Iterable[Path]) -> list[tuple[Path, dict[str, Any]]]: + return [(path, load_json(path)) for path in sorted(paths)] + + +def _evidence_file(value: dict[str, Any], key: str, record_path: Path, workspace: Path) -> dict[str, Any] | None: + descriptor = value.get(key) + if not isinstance(descriptor, dict) or not descriptor.get("path"): + return None + path = Path(descriptor["path"]) + if not path.is_absolute(): + path = record_path.parent / path + if not path.is_file(): + return None + return describe_file(path, workspace) + + +def _artifact_hashes(workspace: Path, excluded: set[Path]) -> list[dict[str, Any]]: + records = [] + excluded_resolved = {path.resolve() for path in excluded} + for path in sorted(workspace.rglob("*")): + if not path.is_file() or path.resolve() in excluded_resolved: + continue + records.append(describe_file(path, workspace)) + return records + + +def build(args: argparse.Namespace) -> dict[str, Any]: + workspace = args.workspace.resolve() + output = args.output.resolve() + checkpoint = args.checkpoint.resolve() + if not workspace.is_dir(): + raise FileNotFoundError(workspace) + if not checkpoint.is_file(): + raise FileNotFoundError(checkpoint) + identity, identity_issues = identity_reasons( + checkpoint, args.image_ref, args.image_digest + ) + + cases_path = workspace / "cases" / "cases.json" + cases_record = load_json(cases_path) if cases_path.is_file() else None + cases = cases_record.get("cases", []) if cases_record else [] + + run_records = _load_records(workspace.glob("runs/*/*/*/result.json")) + parity_records = _load_records(workspace.glob("parity/*/*.json")) + phonon_records = _load_records(workspace.glob("phonolammps/*/*/result.json")) + all_records = run_records + parity_records + phonon_records + + reasons: list[str] = [] + hard_failures: list[str] = [] + if identity["checkpoint_pt"]["sha256"] != EXPECTED_MODEL_SHA256: + hard_failures.append("CHECKPOINT_SHA256_MISMATCH") + reasons.extend( + issue for issue in identity_issues if issue != "CHECKPOINT_SHA256_MISMATCH" + ) + if cases_record is None: + reasons.append("CASES_MANIFEST_MISSING") + else: + observed_cases = {case.get("name") for case in cases} + for name in EXPECTED_CASES: + if name not in observed_cases: + reasons.append(f"CASE_MISSING:{name}") + + run_by_key: dict[tuple[str, str, str], tuple[Path, dict[str, Any]]] = {} + for path, record in run_records: + key = (record.get("case"), record.get("device"), record.get("mode")) + if key in run_by_key: + hard_failures.append(f"DUPLICATE_RUN:{'/'.join(str(x) for x in key)}") + run_by_key[key] = (path, record) + if record.get("status") == "failed": + hard_failures.append(f"RUN_FAILED:{'/'.join(str(x) for x in key)}") + elif record.get("status") != "passed": + reasons.append(f"RUN_NOT_PASSED:{'/'.join(str(x) for x in key)}") + for case in EXPECTED_CASES: + for device in EXPECTED_DEVICES: + for mode in EXPECTED_MODES: + if (case, device, mode) not in run_by_key: + reasons.append(f"RUN_MISSING:{case}/{device}/{mode}") + + parity_by_key: dict[tuple[str, str], tuple[Path, dict[str, Any]]] = {} + for path, record in parity_records: + key = (record.get("case"), record.get("mode")) + if key in parity_by_key: + hard_failures.append(f"DUPLICATE_PARITY:{'/'.join(str(x) for x in key)}") + parity_by_key[key] = (path, record) + if record.get("status") == "failed": + hard_failures.append(f"PARITY_FAILED:{'/'.join(str(x) for x in key)}") + elif record.get("status") != "passed": + reasons.append(f"PARITY_NOT_PASSED:{'/'.join(str(x) for x in key)}") + for case in EXPECTED_CASES: + for mode in EXPECTED_MODES: + if (case, mode) not in parity_by_key: + reasons.append(f"PARITY_MISSING:{case}/{mode}") + + expected_phonon = ("Ti_hcp", "gpu") + phonon_by_key: dict[tuple[str, str], tuple[Path, dict[str, Any]]] = {} + for path, record in phonon_records: + key = (record.get("case"), record.get("device")) + if key in phonon_by_key: + hard_failures.append(f"DUPLICATE_PHONOLAMMPS:{'/'.join(str(x) for x in key)}") + phonon_by_key[key] = (path, record) + if record.get("status") == "failed": + hard_failures.append(f"PHONOLAMMPS_FAILED:{'/'.join(str(x) for x in key)}") + elif record.get("status") != "passed": + reasons.append(f"PHONOLAMMPS_NOT_PASSED:{'/'.join(str(x) for x in key)}") + validation = record.get("force_constants_validation") + if not isinstance(validation, dict) or validation.get("status") != "passed": + hard_failures.append( + f"PHONOLAMMPS_FORCE_CONSTANTS_INVALID:{'/'.join(str(x) for x in key)}" + ) + if expected_phonon not in phonon_by_key: + reasons.append("PHONOLAMMPS_MISSING:Ti_hcp/gpu") + + case_summaries = [] + for generated_case in cases: + case = dict(generated_case) + name = case.get("name") + related = [ + record + for key, (_, record) in run_by_key.items() + if key[0] == name + ] + related.extend( + record + for key, (_, record) in parity_by_key.items() + if key[0] == name + ) + if name == "Ti_hcp" and expected_phonon in phonon_by_key: + related.append(phonon_by_key[expected_phonon][1]) + expected_count = 7 if name == "Ti_hcp" else 6 + if not related: + case["status"] = "untested" + elif any(record.get("status") == "failed" for record in related): + case["status"] = "failed" + elif len(related) == expected_count and all( + record.get("status") == "passed" for record in related + ): + case["status"] = "passed" + else: + case["status"] = "inconclusive" + case_summaries.append(case) + + pt2_by_device: dict[str, dict[str, dict[str, Any]]] = { + "cpu": {}, + "gpu": {}, + } + environment_by_hash: dict[str, dict[str, Any]] = {} + expected_image = identity["image"] + expected_checkpoint_hash = identity["checkpoint_pt"]["sha256"] + for path, record in run_records + phonon_records: + record_identity = record.get("identity", {}) + if ( + record_identity.get("checkpoint_pt", {}).get("sha256") + != expected_checkpoint_hash + ): + hard_failures.append( + f"RECORD_CHECKPOINT_MISMATCH:{path.relative_to(workspace)}" + ) + if record_identity.get("image") != expected_image: + reasons.append(f"RECORD_IMAGE_MISMATCH:{path.relative_to(workspace)}") + runtime = record_identity.get("runtime_model_pt2", {}) + record_device = record.get("device") + runtime_device = runtime.get("device") + runtime_hash = runtime.get("sha256") + if runtime_device != record_device or runtime_device not in pt2_by_device: + hard_failures.append( + f"RECORD_RUNTIME_DEVICE_MISMATCH:{path.relative_to(workspace)}" + ) + elif not runtime_hash: + hard_failures.append( + f"RECORD_RUNTIME_SHA256_MISSING:{path.relative_to(workspace)}" + ) + else: + aggregate = pt2_by_device[runtime_device].setdefault( + runtime_hash, + { + "device": runtime_device, + "sha256": runtime_hash, + "bytes": runtime.get("bytes"), + "sources": [], + }, + ) + aggregate["sources"].append(str(path.relative_to(workspace))) + environment = record.get("environment") + if isinstance(environment, dict): + fingerprint = canonical_hash(environment) + environment_by_hash.setdefault( + fingerprint, + { + "fingerprint_sha256": fingerprint, + "source": str(path.relative_to(workspace)), + "gpu_inventory": environment.get("gpu_inventory", []), + "driver_version": environment.get("driver_version"), + "cuda_version": environment.get("cuda_version"), + "cuda_versions": environment.get("cuda_versions", {}), + "packages": environment.get("packages", {}), + "executables": environment.get("executables", {}), + "selected_environment": environment.get( + "selected_environment", {} + ), + "lammps_plugins": environment.get("lammps_plugins", []), + }, + ) + runtime_models = [] + for device in ("cpu", "gpu"): + hashes = pt2_by_device[device] + if len(hashes) > 1: + hard_failures.append(f"MULTIPLE_{device.upper()}_RUNTIME_MODEL_HASHES") + runtime_models.extend(hashes[key] for key in sorted(hashes)) + if runtime_models: + cpu_hashes = set(pt2_by_device["cpu"]) + gpu_hashes = set(pt2_by_device["gpu"]) + if not cpu_hashes: + reasons.append("CPU_RUNTIME_MODEL_PT2_MISSING") + if not gpu_hashes: + reasons.append("GPU_RUNTIME_MODEL_PT2_MISSING") + if cpu_hashes & gpu_hashes: + hard_failures.append("RUNTIME_MODEL_HASH_NOT_DEVICE_DISTINCT") + identity["runtime_models_pt2"] = runtime_models + + run_summaries = [] + for path, record in run_records: + run_summaries.append( + { + "case": record.get("case"), + "device": record.get("device"), + "mode": record.get("mode"), + "status": record.get("status"), + "exit_code": record.get("exit_code"), + "timed_out": record.get("timed_out"), + "result": describe_file(path, workspace), + "stdout": _evidence_file(record, "stdout", path, workspace), + "stderr": _evidence_file(record, "stderr", path, workspace), + "reason_codes": record.get("reason_codes", []), + } + ) + parity_summaries = [ + { + "case": record.get("case"), + "mode": record.get("mode"), + "status": record.get("status"), + "metrics": record.get("metrics"), + "thresholds": record.get("thresholds"), + "result": describe_file(path, workspace), + "reason_codes": record.get("reason_codes", []), + } + for path, record in parity_records + ] + phonon_summaries = [ + { + "case": record.get("case"), + "device": record.get("device"), + "status": record.get("status"), + "exit_code": record.get("exit_code"), + "timed_out": record.get("timed_out"), + "result": describe_file(path, workspace), + "stdout": _evidence_file(record, "stdout", path, workspace), + "stderr": _evidence_file(record, "stderr", path, workspace), + "force_constants_validation": record.get( + "force_constants_validation" + ), + "reason_codes": record.get("reason_codes", []), + } + for path, record in phonon_records + ] + + if not all_records: + status = "untested" + reasons.append("NO_RUNTIME_EXECUTION_EVIDENCE") + elif hard_failures: + status = "failed" + elif reasons: + status = "inconclusive" + else: + status = "passed" + reason_codes = sorted(set(hard_failures + reasons)) + excluded = {output} + artifacts = _artifact_hashes(workspace, excluded) + provenance = _git_provenance(Path(__file__).resolve().parent) + manifest = { + "$schema": str((Path(__file__).resolve().parent / "manifest.schema.json")), + "schema_version": SCHEMA_VERSION, + "benchmark_id": BENCHMARK_ID, + "created_at": utc_now(), + "status": status, + "reason_codes": reason_codes, + "provenance": provenance, + "identity": identity, + "environment": { + "observations": [ + environment_by_hash[key] for key in sorted(environment_by_hash) + ] + }, + "thresholds": DEFAULT_THRESHOLDS, + "cases": case_summaries, + "runs": run_summaries, + "parity": parity_summaries, + "phonolammps": phonon_summaries, + "artifact_hashes": artifacts, + "notes": [ + "The checkpoint .pt is identity/probe input only; LAMMPS receives a device-specific .pt2 runtime artifact.", + "CPU and T4 GPU runtime .pt2 hashes must be distinct and internally consistent across all cases.", + "phonoLAMMPS evidence must include all finite N^2 FORCE_CONSTANTS blocks for the requested supercell.", + "Only status=passed is positive image/model/GPU compatibility evidence.", + ], + } + write_json_new(output, manifest) + return manifest + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--workspace", type=Path, default=Path("workspace")) + parser.add_argument("--checkpoint", required=True, type=Path) + parser.add_argument("--image-ref", required=True) + parser.add_argument("--image-digest") + parser.add_argument("--output", type=Path, default=Path("workspace/manifest.json")) + args = parser.parse_args(argv) + manifest = build(args) + print(f"manifest: {args.output}") + print(f"status={manifest['status']}") + for reason in manifest["reason_codes"]: + print(f" - {reason}") + return 0 if manifest["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/compare_parity.py b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/compare_parity.py new file mode 100644 index 00000000..9acac649 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/compare_parity.py @@ -0,0 +1,186 @@ +#!/usr/bin/env python3 +"""Compare one CPU/GPU LAMMPS result pair with explicit numerical gates.""" + +from __future__ import annotations + +import argparse +import math +from pathlib import Path +from typing import Any, Sequence + +from benchmark_lib import ( + BENCHMARK_ID, + SCHEMA_VERSION, + describe_file, + load_json, + utc_now, + write_json_new, +) + + +DEFAULT_THRESHOLDS = { + "energy_per_atom_abs_eV": 1.0e-4, + "force_rms_eV_per_A": 1.0e-3, + "force_max_abs_eV_per_A": 5.0e-3, + "stress_max_abs_GPa": 1.0e-2, +} + + +def _force_map(record: dict[str, Any]) -> dict[int, tuple[float, float, float]]: + forces = record["observables"]["forces"] + mapping = {} + for atom in forces: + atom_id = int(atom["id"]) + if atom_id in mapping: + raise ValueError(f"duplicate atom id: {atom_id}") + vector = tuple(float(atom[key]) for key in ("fx", "fy", "fz")) + if not all(math.isfinite(value) for value in vector): + raise ValueError(f"non-finite force for atom {atom_id}") + mapping[atom_id] = vector + return mapping + + +def compare( + cpu: dict[str, Any], gpu: dict[str, Any], thresholds: dict[str, float] +) -> tuple[dict[str, float] | None, list[str], list[str]]: + failures: list[str] = [] + inconclusive: list[str] = [] + if cpu.get("status") != "passed": + inconclusive.append("CPU_RESULT_NOT_PASSED") + if gpu.get("status") != "passed": + inconclusive.append("GPU_RESULT_NOT_PASSED") + for key in ("case", "mode"): + if cpu.get(key) != gpu.get(key): + inconclusive.append(f"{key.upper()}_MISMATCH") + if cpu.get("device") != "cpu" or gpu.get("device") != "gpu": + inconclusive.append("DEVICE_LABEL_MISMATCH") + cpu_identity = cpu.get("identity", {}) + gpu_identity = gpu.get("identity", {}) + if cpu_identity.get("checkpoint_pt", {}).get("sha256") != gpu_identity.get( + "checkpoint_pt", {} + ).get("sha256"): + inconclusive.append("CHECKPOINT_SHA256_MISMATCH") + if cpu_identity.get("image") != gpu_identity.get("image"): + inconclusive.append("IMAGE_IDENTITY_MISMATCH") + cpu_runtime = cpu_identity.get("runtime_model_pt2", {}) + gpu_runtime = gpu_identity.get("runtime_model_pt2", {}) + if cpu_runtime.get("device") != "cpu" or gpu_runtime.get("device") != "gpu": + failures.append("RUNTIME_MODEL_DEVICE_LABEL_MISMATCH") + if not str(cpu_runtime.get("path", "")).endswith(".pt2") or not str( + gpu_runtime.get("path", "") + ).endswith(".pt2"): + failures.append("RUNTIME_MODEL_NOT_PT2") + cpu_runtime_hash = cpu_runtime.get("sha256") + gpu_runtime_hash = gpu_runtime.get("sha256") + if not cpu_runtime_hash or not gpu_runtime_hash: + inconclusive.append("RUNTIME_MODEL_SHA256_MISSING") + elif cpu_runtime_hash == gpu_runtime_hash: + failures.append("RUNTIME_MODEL_HASH_NOT_DEVICE_DISTINCT") + if inconclusive: + return None, failures, sorted(set(inconclusive)) + if failures: + return None, sorted(set(failures)), [] + + cpu_obs = cpu.get("observables") + gpu_obs = gpu.get("observables") + if not isinstance(cpu_obs, dict) or not isinstance(gpu_obs, dict): + return None, failures, ["OBSERVABLES_MISSING"] + if cpu_obs.get("natoms") != gpu_obs.get("natoms"): + return None, failures, ["ATOM_COUNT_MISMATCH"] + natoms = int(cpu_obs["natoms"]) + if natoms <= 0: + return None, failures, ["INVALID_ATOM_COUNT"] + try: + energy_delta = abs(float(cpu_obs["energy_eV"]) - float(gpu_obs["energy_eV"])) / natoms + cpu_forces = _force_map(cpu) + gpu_forces = _force_map(gpu) + if set(cpu_forces) != set(gpu_forces): + return None, failures, ["FORCE_ATOM_IDS_MISMATCH"] + components = [ + cpu_forces[atom_id][axis] - gpu_forces[atom_id][axis] + for atom_id in sorted(cpu_forces) + for axis in range(3) + ] + force_rms = math.sqrt(sum(value * value for value in components) / len(components)) + force_max = max(abs(value) for value in components) + stress_keys = ("pxx", "pyy", "pzz", "pxy", "pxz", "pyz") + # LAMMPS metal pressure is bar; 1 bar = 1e-4 GPa. + stress_max = max( + abs( + float(cpu_obs["stress_bar"][key]) + - float(gpu_obs["stress_bar"][key]) + ) + * 1.0e-4 + for key in stress_keys + ) + except (KeyError, TypeError, ValueError, ZeroDivisionError) as exc: + return None, failures, [f"COMPARISON_PARSE_FAILED:{type(exc).__name__}:{exc}"] + metrics = { + "energy_per_atom_abs_eV": energy_delta, + "force_rms_eV_per_A": force_rms, + "force_max_abs_eV_per_A": force_max, + "stress_max_abs_GPa": stress_max, + } + for name, value in metrics.items(): + if not math.isfinite(value): + failures.append(f"NONFINITE_METRIC:{name}") + elif value > thresholds[name]: + failures.append(f"THRESHOLD_EXCEEDED:{name}") + return metrics, sorted(set(failures)), [] + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--cpu", required=True, type=Path) + parser.add_argument("--gpu", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--energy-tol", type=float, default=DEFAULT_THRESHOLDS["energy_per_atom_abs_eV"]) + parser.add_argument("--force-rms-tol", type=float, default=DEFAULT_THRESHOLDS["force_rms_eV_per_A"]) + parser.add_argument("--force-max-tol", type=float, default=DEFAULT_THRESHOLDS["force_max_abs_eV_per_A"]) + parser.add_argument("--stress-tol", type=float, default=DEFAULT_THRESHOLDS["stress_max_abs_GPa"]) + args = parser.parse_args(argv) + thresholds = { + "energy_per_atom_abs_eV": args.energy_tol, + "force_rms_eV_per_A": args.force_rms_tol, + "force_max_abs_eV_per_A": args.force_max_tol, + "stress_max_abs_GPa": args.stress_tol, + } + if any((not math.isfinite(value) or value < 0) for value in thresholds.values()): + parser.error("all tolerances must be finite and non-negative") + cpu = load_json(args.cpu) + gpu = load_json(args.gpu) + metrics, failures, inconclusive = compare(cpu, gpu, thresholds) + if failures: + status = "failed" + reasons = failures + elif inconclusive: + status = "inconclusive" + reasons = inconclusive + else: + status = "passed" + reasons = [] + record = { + "schema_version": SCHEMA_VERSION, + "benchmark_id": BENCHMARK_ID, + "kind": "cpu_gpu_parity", + "created_at": utc_now(), + "status": status, + "reason_codes": reasons, + "case": cpu.get("case"), + "mode": cpu.get("mode"), + "thresholds": thresholds, + "metrics": metrics, + "inputs": { + "cpu": describe_file(args.cpu), + "gpu": describe_file(args.gpu), + }, + } + write_json_new(args.output, record) + print(f"{record['case']} {record['mode']} parity: {status}") + for reason in reasons: + print(f" - {reason}") + return 0 if status == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/generate_cases.py b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/generate_cases.py new file mode 100644 index 00000000..d6a6d476 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/generate_cases.py @@ -0,0 +1,290 @@ +#!/usr/bin/env python3 +"""Generate deterministic, dependency-free Ti, V, and B2 TiV benchmark cells.""" + +from __future__ import annotations + +import argparse +import math +from dataclasses import dataclass +from pathlib import Path +from typing import Sequence + +from benchmark_lib import ( + BENCHMARK_ID, + SCHEMA_VERSION, + describe_file, + utc_now, + write_json_new, + write_text_if_identical, +) + + +@dataclass(frozen=True) +class Case: + name: str + formula: str + elements: tuple[str, ...] + masses: tuple[float, ...] + lattice: tuple[tuple[float, float, float], ...] + atoms: tuple[tuple[int, float, float, float], ...] + fractional: tuple[tuple[float, float, float], ...] + poscar_counts: tuple[int, ...] + crystal: str + + +def _cases() -> tuple[Case, ...]: + ti_a = 2.95 + ti_c = 4.68 + ti_y = math.sqrt(3.0) * ti_a / 2.0 + return ( + Case( + name="Ti_hcp", + formula="Ti2", + elements=("Ti",), + masses=(47.867,), + lattice=((ti_a, 0.0, 0.0), (ti_a / 2.0, ti_y, 0.0), (0.0, 0.0, ti_c)), + atoms=((1, 0.0, 0.0, 0.0), (1, ti_a / 2.0, ti_y / 3.0, ti_c / 2.0)), + fractional=((0.0, 0.0, 0.0), (1.0 / 3.0, 1.0 / 3.0, 0.5)), + poscar_counts=(2,), + crystal="hcp primitive", + ), + Case( + name="V_bcc", + formula="V2", + elements=("V",), + masses=(50.9415,), + lattice=((3.03, 0.0, 0.0), (0.0, 3.03, 0.0), (0.0, 0.0, 3.03)), + atoms=((1, 0.0, 0.0, 0.0), (1, 1.515, 1.515, 1.515)), + fractional=((0.0, 0.0, 0.0), (0.5, 0.5, 0.5)), + poscar_counts=(2,), + crystal="bcc conventional", + ), + Case( + name="TiV_B2", + formula="TiV", + elements=("Ti", "V"), + masses=(47.867, 50.9415), + lattice=((3.18, 0.0, 0.0), (0.0, 3.18, 0.0), (0.0, 0.0, 3.18)), + atoms=((1, 0.0, 0.0, 0.0), (2, 1.59, 1.59, 1.59)), + fractional=((0.0, 0.0, 0.0), (0.5, 0.5, 0.5)), + poscar_counts=(1, 1), + crystal="B2 (CsCl prototype)", + ), + ) + + +def _poscar(case: Case) -> str: + lines = [ + f"{case.name} deterministic DPA4 benchmark cell", + "1.0", + ] + lines.extend(" %.12f %.12f %.12f" % vector for vector in case.lattice) + lines.append(" " + " ".join(case.elements)) + lines.append(" " + " ".join(str(value) for value in case.poscar_counts)) + lines.append("Direct") + lines.extend(" %.12f %.12f %.12f" % position for position in case.fractional) + return "\n".join(lines) + "\n" + + +def _lammps_data(case: Case) -> str: + a, b, c = case.lattice + if any(abs(value) > 1.0e-12 for value in (a[1], a[2], b[2], c[0], c[1])): + raise ValueError(f"{case.name}: lattice is not LAMMPS restricted triclinic") + lines = [ + f"LAMMPS data: {case.name} deterministic DPA4 benchmark cell", + "", + f"{len(case.atoms)} atoms", + f"{len(case.elements)} atom types", + "", + f"0.0 {a[0]:.12f} xlo xhi", + f"0.0 {b[1]:.12f} ylo yhi", + f"0.0 {c[2]:.12f} zlo zhi", + f"{b[0]:.12f} {c[0]:.12f} {c[1]:.12f} xy xz yz", + "", + "Masses", + "", + ] + lines.extend( + f"{index} {mass:.8f} # {element}" + for index, (mass, element) in enumerate(zip(case.masses, case.elements), start=1) + ) + lines.extend(["", "Atoms # atomic", ""]) + lines.extend( + f"{index} {atom_type} {x:.12f} {y:.12f} {z:.12f}" + for index, (atom_type, x, y, z) in enumerate(case.atoms, start=1) + ) + return "\n".join(lines) + "\n" + + +def _common_input(case: Case) -> list[str]: + element_map = " ".join(case.elements) + lines = [ + "clear", + "units metal", + "dimension 3", + "boundary p p p", + "atom_style atomic", + "atom_modify map yes", + "box tilt large", + "read_data structure.data", + ] + lines.extend( + f"mass {index} {mass:.8f} # {element}" + for index, (mass, element) in enumerate( + zip(case.masses, case.elements), start=1 + ) + ) + lines.extend( + [ + "neigh_modify every 1 delay 0 check no", + "pair_style deepmd ${runtime_model}", + f"pair_coeff * * {element_map}", + "thermo 1", + "thermo_style custom step atoms pe ke etotal temp press pxx pyy pzz pxy pxz pyz", + "thermo_modify flush yes", + ] + ) + return lines + + +def _result_tail() -> list[str]: + return [ + "variable bench_n equal count(all)", + "variable bench_pe equal pe", + "variable bench_pxx equal pxx", + "variable bench_pyy equal pyy", + "variable bench_pzz equal pzz", + "variable bench_pxy equal pxy", + "variable bench_pxz equal pxz", + "variable bench_pyz equal pyz", + 'print "BENCH_RESULT natoms=$(v_bench_n:%.0f) pe=$(v_bench_pe:%.17g) pxx=$(v_bench_pxx:%.17g) pyy=$(v_bench_pyy:%.17g) pzz=$(v_bench_pzz:%.17g) pxy=$(v_bench_pxy:%.17g) pxz=$(v_bench_pxz:%.17g) pyz=$(v_bench_pyz:%.17g)"', + "write_dump all custom forces.dump id type x y z fx fy fz modify sort id", + ] + + +def _run0_input(case: Case) -> str: + lines = _common_input(case) + lines.extend(["run 0", *_result_tail()]) + return "\n".join(lines) + "\n" + + +def _md_input(case: Case) -> str: + lines = _common_input(case) + lines.extend( + [ + "timestep 0.001", + "velocity all create 300.0 4928459 mom yes rot no dist gaussian", + "fix bench_nve all nve", + "run 20", + "unfix bench_nve", + "run 0 post no", + *_result_tail(), + ] + ) + return "\n".join(lines) + "\n" + + +def _phonon_input(case: Case) -> str: + lines = _common_input(case) + # APEX truncates its ordinary input immediately after pair_coeff before + # invoking phonoLAMMPS. Match that contract exactly. + pair_index = next(i for i, line in enumerate(lines) if line.startswith("pair_coeff")) + return "\n".join(lines[: pair_index + 1]) + "\n" + + +def _validate_dpa4_input(text: str, case_name: str, mode: str) -> None: + lines = [line.strip() for line in text.splitlines() if line.strip()] + for required in ( + "atom_style atomic", + "atom_modify map yes", + "read_data structure.data", + ): + if lines.count(required) != 1: + raise ValueError( + f"{case_name}/{mode}: expected exactly one {required!r} line" + ) + atom_style = lines.index("atom_style atomic") + atom_map = lines.index("atom_modify map yes") + read_data = lines.index("read_data structure.data") + if not atom_style < atom_map < read_data: + raise ValueError( + f"{case_name}/{mode}: atom_modify map yes must precede read_data" + ) + if any(line.startswith("plugin load ") for line in lines): + raise ValueError( + f"{case_name}/{mode}: explicit plugin load is forbidden; " + "use LAMMPS_PLUGIN_PATH auto-loading" + ) + pair_styles = [line for line in lines if line.startswith("pair_style deepmd ")] + if pair_styles != ["pair_style deepmd ${runtime_model}"]: + raise ValueError( + f"{case_name}/{mode}: pair_style must use the runtime-model placeholder" + ) + + +def generate(output: Path) -> dict: + output.mkdir(parents=True, exist_ok=True) + descriptors = [] + for case in _cases(): + case_dir = output / case.name + rendered_inputs = { + "in.run0.lammps": _run0_input(case), + "in.md.lammps": _md_input(case), + "in.phonon.lammps": _phonon_input(case), + } + for name, text in rendered_inputs.items(): + _validate_dpa4_input(text, case.name, name) + write_text_if_identical(case_dir / "POSCAR", _poscar(case)) + write_text_if_identical(case_dir / "structure.data", _lammps_data(case)) + for name, text in rendered_inputs.items(): + write_text_if_identical(case_dir / name, text) + artifacts = [ + describe_file(case_dir / name, output) + for name in ( + "POSCAR", + "structure.data", + "in.run0.lammps", + "in.md.lammps", + "in.phonon.lammps", + ) + ] + descriptors.append( + { + "name": case.name, + "formula": case.formula, + "elements": list(case.elements), + "atoms": len(case.atoms), + "crystal": case.crystal, + "status": "untested", + "artifacts": artifacts, + } + ) + manifest = { + "schema_version": SCHEMA_VERSION, + "benchmark_id": BENCHMARK_ID, + "generated_at": utc_now(), + "status": "untested", + "reason_codes": ["NO_RUNTIME_EXECUTION_EVIDENCE"], + "cases": descriptors, + } + write_json_new(output / "cases.json", manifest) + return manifest + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--output", + type=Path, + default=Path("workspace/cases"), + help="case directory to create (default: workspace/cases)", + ) + args = parser.parse_args(argv) + manifest = generate(args.output) + print(f"generated {len(manifest['cases'])} deterministic cases in {args.output}") + print("status=untested (generation is not runtime compatibility evidence)") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/manifest.example.json b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/manifest.example.json new file mode 100644 index 00000000..e17e3338 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/manifest.example.json @@ -0,0 +1,73 @@ +{ + "$schema": "./manifest.schema.json", + "schema_version": "2.0.0", + "benchmark_id": "dpa4-alloytongqi-small-cell-gpu-compat-v2", + "created_at": null, + "status": "untested", + "reason_codes": [ + "NO_RUNTIME_EXECUTION_EVIDENCE" + ], + "provenance": { + "apex_git_commit": null, + "apex_git_dirty": null + }, + "identity": { + "checkpoint_pt": { + "path": "../../models/DPA4-alloytongqi/model.pt", + "sha256": "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad", + "bytes": 30403297 + }, + "runtime_models_pt2": [], + "image": { + "reference": null, + "digest": null + } + }, + "environment": { + "observations": [] + }, + "thresholds": { + "energy_per_atom_abs_eV": 0.0001, + "force_rms_eV_per_A": 0.001, + "force_max_abs_eV_per_A": 0.005, + "stress_max_abs_GPa": 0.01 + }, + "cases": [ + { + "name": "Ti_hcp", + "formula": "Ti2", + "elements": ["Ti"], + "atoms": 2, + "crystal": "hcp primitive", + "status": "untested", + "artifacts": [] + }, + { + "name": "V_bcc", + "formula": "V2", + "elements": ["V"], + "atoms": 2, + "crystal": "bcc conventional", + "status": "untested", + "artifacts": [] + }, + { + "name": "TiV_B2", + "formula": "TiV", + "elements": ["Ti", "V"], + "atoms": 2, + "crystal": "B2 (CsCl prototype)", + "status": "untested", + "artifacts": [] + } + ], + "runs": [], + "parity": [], + "phonolammps": [], + "artifact_hashes": [], + "notes": [ + "This example intentionally contains no image, GPU, package, exit-code, stderr, or numerical result claims.", + "The checkpoint .pt is identity/probe input only; LAMMPS requires device-specific .pt2 runtime artifacts.", + "An empty runtime_models_pt2 list means no CPU/T4 runtime execution artifact has been recorded." + ] +} diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/manifest.schema.json b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/manifest.schema.json new file mode 100644 index 00000000..a9506d90 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/manifest.schema.json @@ -0,0 +1,381 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://apex-flow.local/schemas/dpa4-alloytongqi-benchmark-v2.json", + "title": "DPA4 alloytongqi small-cell GPU compatibility manifest", + "type": "object", + "additionalProperties": false, + "required": [ + "$schema", + "schema_version", + "benchmark_id", + "created_at", + "status", + "reason_codes", + "provenance", + "identity", + "environment", + "thresholds", + "cases", + "runs", + "parity", + "phonolammps", + "artifact_hashes", + "notes" + ], + "properties": { + "$schema": {"type": "string"}, + "schema_version": {"const": "2.0.0"}, + "benchmark_id": {"const": "dpa4-alloytongqi-small-cell-gpu-compat-v2"}, + "created_at": {"type": ["string", "null"]}, + "status": {"$ref": "#/$defs/status"}, + "reason_codes": { + "type": "array", + "items": {"type": "string", "minLength": 1}, + "uniqueItems": true + }, + "provenance": { + "type": "object", + "additionalProperties": false, + "required": ["apex_git_commit", "apex_git_dirty"], + "properties": { + "apex_git_commit": { + "type": ["string", "null"], + "pattern": "^[0-9a-f]{40}$" + }, + "apex_git_dirty": {"type": ["boolean", "null"]} + } + }, + "identity": { + "type": "object", + "additionalProperties": false, + "required": ["checkpoint_pt", "runtime_models_pt2", "image"], + "properties": { + "checkpoint_pt": {"$ref": "#/$defs/file"}, + "runtime_models_pt2": { + "type": "array", + "items": {"$ref": "#/$defs/runtimeModel"} + }, + "image": {"$ref": "#/$defs/image"} + } + }, + "environment": { + "type": "object", + "additionalProperties": false, + "required": ["observations"], + "properties": { + "observations": { + "type": "array", + "items": {"$ref": "#/$defs/environmentObservation"} + } + } + }, + "thresholds": {"$ref": "#/$defs/thresholds"}, + "cases": { + "type": "array", + "items": {"$ref": "#/$defs/case"} + }, + "runs": { + "type": "array", + "items": {"$ref": "#/$defs/runSummary"} + }, + "parity": { + "type": "array", + "items": {"$ref": "#/$defs/paritySummary"} + }, + "phonolammps": { + "type": "array", + "items": {"$ref": "#/$defs/phononSummary"} + }, + "artifact_hashes": { + "type": "array", + "items": {"$ref": "#/$defs/file"} + }, + "notes": { + "type": "array", + "items": {"type": "string"} + } + }, + "allOf": [ + { + "if": {"properties": {"status": {"const": "passed"}}}, + "then": { + "properties": { + "reason_codes": {"maxItems": 0}, + "identity": { + "properties": { + "checkpoint_pt": { + "properties": { + "sha256": { + "const": "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad" + } + } + }, + "runtime_models_pt2": {"minItems": 2, "maxItems": 2}, + "image": { + "properties": { + "reference": {"type": "string", "minLength": 1}, + "digest": { + "type": "string", + "pattern": "^sha256:[0-9a-f]{64}$" + } + } + } + } + }, + "environment": { + "properties": {"observations": {"minItems": 1}} + }, + "runs": {"minItems": 12, "maxItems": 12}, + "parity": {"minItems": 6, "maxItems": 6}, + "phonolammps": {"minItems": 1}, + "artifact_hashes": {"minItems": 1} + } + } + } + ], + "$defs": { + "status": { + "enum": ["untested", "inconclusive", "failed", "passed"] + }, + "hexSha256": { + "type": "string", + "pattern": "^[0-9a-f]{64}$" + }, + "file": { + "type": "object", + "additionalProperties": false, + "required": ["path", "sha256", "bytes"], + "properties": { + "path": {"type": "string", "minLength": 1}, + "sha256": {"$ref": "#/$defs/hexSha256"}, + "bytes": {"type": "integer", "minimum": 0} + } + }, + "runtimeModel": { + "type": "object", + "additionalProperties": false, + "required": ["device", "sha256", "bytes", "sources"], + "properties": { + "device": {"enum": ["cpu", "gpu"]}, + "sha256": {"$ref": "#/$defs/hexSha256"}, + "bytes": {"type": "integer", "minimum": 1}, + "sources": { + "type": "array", + "items": {"type": "string", "minLength": 1}, + "minItems": 1, + "uniqueItems": true + } + } + }, + "image": { + "type": "object", + "additionalProperties": false, + "required": ["reference", "digest"], + "properties": { + "reference": {"type": ["string", "null"]}, + "digest": { + "type": ["string", "null"], + "pattern": "^sha256:[0-9a-f]{64}$" + } + } + }, + "thresholds": { + "type": "object", + "additionalProperties": false, + "required": [ + "energy_per_atom_abs_eV", + "force_rms_eV_per_A", + "force_max_abs_eV_per_A", + "stress_max_abs_GPa" + ], + "properties": { + "energy_per_atom_abs_eV": {"type": "number", "minimum": 0}, + "force_rms_eV_per_A": {"type": "number", "minimum": 0}, + "force_max_abs_eV_per_A": {"type": "number", "minimum": 0}, + "stress_max_abs_GPa": {"type": "number", "minimum": 0} + } + }, + "case": { + "type": "object", + "additionalProperties": false, + "required": ["name", "formula", "elements", "atoms", "crystal", "status", "artifacts"], + "properties": { + "name": {"enum": ["Ti_hcp", "V_bcc", "TiV_B2"]}, + "formula": {"type": "string"}, + "elements": { + "type": "array", + "items": {"type": "string"}, + "minItems": 1 + }, + "atoms": {"type": "integer", "minimum": 1}, + "crystal": {"type": "string"}, + "status": {"$ref": "#/$defs/status"}, + "artifacts": { + "type": "array", + "items": {"$ref": "#/$defs/file"} + } + } + }, + "environmentObservation": { + "type": "object", + "additionalProperties": false, + "required": [ + "fingerprint_sha256", + "source", + "gpu_inventory", + "driver_version", + "cuda_version", + "cuda_versions", + "packages", + "executables", + "selected_environment", + "lammps_plugins" + ], + "properties": { + "fingerprint_sha256": {"$ref": "#/$defs/hexSha256"}, + "source": {"type": "string"}, + "gpu_inventory": { + "type": "array", + "items": {"$ref": "#/$defs/gpu"} + }, + "driver_version": {"type": ["string", "null"]}, + "cuda_version": {"type": ["string", "null"]}, + "cuda_versions": { + "type": "object", + "additionalProperties": {"type": ["string", "null"]} + }, + "packages": { + "type": "object", + "additionalProperties": {"type": ["string", "null"]} + }, + "executables": { + "type": "object", + "additionalProperties": {"type": ["string", "null"]} + }, + "selected_environment": { + "type": "object", + "additionalProperties": {"type": ["string", "null"]} + }, + "lammps_plugins": { + "type": "array", + "items": {"$ref": "#/$defs/file"} + } + } + }, + "gpu": { + "type": "object", + "additionalProperties": false, + "required": [ + "index", + "name", + "uuid", + "driver_version", + "memory_mib", + "compute_capability" + ], + "properties": { + "index": {"type": "string"}, + "name": {"type": "string", "minLength": 1}, + "uuid": {"type": "string", "minLength": 1}, + "driver_version": {"type": "string", "minLength": 1}, + "memory_mib": {"type": ["integer", "null"], "minimum": 1}, + "compute_capability": {"type": "string", "minLength": 1} + } + }, + "runSummary": { + "type": "object", + "additionalProperties": false, + "required": [ + "case", + "device", + "mode", + "status", + "exit_code", + "timed_out", + "result", + "stdout", + "stderr", + "reason_codes" + ], + "properties": { + "case": {"type": "string"}, + "device": {"enum": ["cpu", "gpu"]}, + "mode": {"enum": ["run0", "md"]}, + "status": {"$ref": "#/$defs/status"}, + "exit_code": {"type": ["integer", "null"]}, + "timed_out": {"type": "boolean"}, + "result": {"$ref": "#/$defs/file"}, + "stdout": {"anyOf": [{"$ref": "#/$defs/file"}, {"type": "null"}]}, + "stderr": {"anyOf": [{"$ref": "#/$defs/file"}, {"type": "null"}]}, + "reason_codes": {"type": "array", "items": {"type": "string"}} + } + }, + "paritySummary": { + "type": "object", + "additionalProperties": false, + "required": ["case", "mode", "status", "metrics", "thresholds", "result", "reason_codes"], + "properties": { + "case": {"type": ["string", "null"]}, + "mode": {"type": ["string", "null"]}, + "status": {"$ref": "#/$defs/status"}, + "metrics": {"anyOf": [{"$ref": "#/$defs/thresholds"}, {"type": "null"}]}, + "thresholds": {"$ref": "#/$defs/thresholds"}, + "result": {"$ref": "#/$defs/file"}, + "reason_codes": {"type": "array", "items": {"type": "string"}} + } + }, + "phononSummary": { + "type": "object", + "additionalProperties": false, + "required": [ + "case", + "device", + "status", + "exit_code", + "timed_out", + "result", + "stdout", + "stderr", + "force_constants_validation", + "reason_codes" + ], + "properties": { + "case": {"type": "string"}, + "device": {"enum": ["cpu", "gpu"]}, + "status": {"$ref": "#/$defs/status"}, + "exit_code": {"type": ["integer", "null"]}, + "timed_out": {"type": "boolean"}, + "result": {"$ref": "#/$defs/file"}, + "stdout": {"anyOf": [{"$ref": "#/$defs/file"}, {"type": "null"}]}, + "stderr": {"anyOf": [{"$ref": "#/$defs/file"}, {"type": "null"}]}, + "force_constants_validation": { + "$ref": "#/$defs/forceConstantsValidation" + }, + "reason_codes": {"type": "array", "items": {"type": "string"}} + } + }, + "forceConstantsValidation": { + "type": "object", + "additionalProperties": false, + "required": [ + "status", + "expected_atom_count", + "atom_count", + "matrix_blocks", + "finite_values", + "error" + ], + "properties": { + "status": {"enum": ["failed", "passed"]}, + "expected_atom_count": { + "type": ["integer", "null"], + "minimum": 1 + }, + "atom_count": {"type": ["integer", "null"], "minimum": 1}, + "matrix_blocks": {"type": "integer", "minimum": 0}, + "finite_values": {"type": "integer", "minimum": 0}, + "error": {"type": ["string", "null"]} + } + } + } +} diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/run_lammps_case.py b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/run_lammps_case.py new file mode 100644 index 00000000..c99e4627 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/run_lammps_case.py @@ -0,0 +1,465 @@ +#!/usr/bin/env python3 +"""Run one generated LAMMPS case and capture fail-closed evidence.""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import re +import shutil +import subprocess +import time +from pathlib import Path +from typing import Any, Sequence + +from benchmark_lib import ( + BENCHMARK_ID, + SCHEMA_VERSION, + capture_runtime_environment, + describe_file, + detect_fatal_output, + identity_reasons, + missing_package_reasons, + resolve_lammps_plugin_path, + run_monitored, + runtime_model_identity, + single_rank_command_reasons, + t4_environment_reasons_for_device, + utc_now, + write_json_new, +) + + +RESULT_RE = re.compile(r"^BENCH_RESULT\s+(?P.+)$", re.MULTILINE) + + +def _parse_result_line(stdout: str) -> dict[str, Any]: + matches = list(RESULT_RE.finditer(stdout)) + if len(matches) != 1: + raise ValueError(f"expected exactly one BENCH_RESULT line, found {len(matches)}") + values: dict[str, float] = {} + for token in matches[0].group("fields").split(): + if "=" not in token: + raise ValueError(f"malformed BENCH_RESULT token: {token!r}") + key, raw = token.split("=", 1) + value = float(raw) + if not math.isfinite(value): + raise ValueError(f"non-finite BENCH_RESULT value: {key}={raw}") + values[key] = value + required = {"natoms", "pe", "pxx", "pyy", "pzz", "pxy", "pxz", "pyz"} + missing = sorted(required - set(values)) + if missing: + raise ValueError(f"BENCH_RESULT missing fields: {', '.join(missing)}") + natoms = int(values.pop("natoms")) + if natoms <= 0: + raise ValueError(f"invalid atom count: {natoms}") + return { + "natoms": natoms, + "energy_eV": values.pop("pe"), + "stress_bar": values, + } + + +def _parse_force_dump(path: Path, expected_atoms: int) -> list[dict[str, Any]]: + lines = path.read_text(encoding="utf-8", errors="strict").splitlines() + headers = [i for i, line in enumerate(lines) if line.startswith("ITEM: ATOMS ")] + if not headers: + raise ValueError("forces.dump has no ITEM: ATOMS block") + start = headers[-1] + columns = lines[start].split()[2:] + required = ("id", "type", "x", "y", "z", "fx", "fy", "fz") + if any(name not in columns for name in required): + raise ValueError(f"forces.dump columns are incomplete: {columns}") + indices = {name: columns.index(name) for name in required} + records = [] + for line in lines[start + 1 :]: + if line.startswith("ITEM:"): + break + tokens = line.split() + if not tokens: + continue + try: + record = { + "id": int(tokens[indices["id"]]), + "type": int(tokens[indices["type"]]), + } + for name in ("x", "y", "z", "fx", "fy", "fz"): + value = float(tokens[indices[name]]) + if not math.isfinite(value): + raise ValueError(f"non-finite {name} for atom {record['id']}") + record[name] = value + except (IndexError, ValueError) as exc: + raise ValueError(f"invalid forces.dump row: {line!r}: {exc}") from exc + records.append(record) + records.sort(key=lambda item: item["id"]) + if len(records) != expected_atoms: + raise ValueError( + f"forces.dump atom count {len(records)} != BENCH_RESULT {expected_atoms}" + ) + if [item["id"] for item in records] != list(range(1, expected_atoms + 1)): + raise ValueError("forces.dump atom ids are not contiguous and one-based") + return records + + +def _prepare_output( + case_dir: Path, runtime_model: Path, output: Path, mode: str, device: str +) -> tuple[Path, Path, Path]: + if output.exists() and any(output.iterdir()): + raise FileExistsError(f"refusing to overwrite non-empty evidence directory: {output}") + output.mkdir(parents=True, exist_ok=True) + input_source = case_dir / f"in.{mode}.lammps" + data_source = case_dir / "structure.data" + for source in (input_source, data_source): + if not source.is_file(): + raise FileNotFoundError(source) + input_target = output / input_source.name + data_target = output / data_source.name + runtime_target = output / f"runtime.{device}.pt2" + shutil.copy2(input_source, input_target) + shutil.copy2(data_source, data_target) + shutil.copy2(runtime_model, runtime_target) + return input_target, data_target, runtime_target + + +def _load_command(raw: str) -> list[str]: + value = json.loads(raw) + if not isinstance(value, list) or not value or not all( + isinstance(item, str) and item for item in value + ): + raise ValueError("--command-json must be a non-empty JSON string array") + return value + + +def _checkpoint_probe( + checkpoint: Path, env: dict[str, str], timeout: float, cwd: Path +) -> dict[str, Any]: + command = [ + "dp", + "--pt", + "show", + str(checkpoint), + "type-map", + "descriptor", + "fitting-net", + "size", + ] + try: + completed = subprocess.run( + command, + cwd=str(cwd), + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + timeout=timeout, + check=False, + ) + return { + "command": command, + "exit_code": completed.returncode, + "timed_out": False, + "stdout": completed.stdout, + "stderr": completed.stderr, + } + except FileNotFoundError as exc: + return { + "command": command, + "exit_code": None, + "timed_out": False, + "stdout": "", + "stderr": f"FileNotFoundError: {exc}", + } + except subprocess.TimeoutExpired as exc: + stdout = exc.stdout or "" + stderr = exc.stderr or "" + if isinstance(stdout, bytes): + stdout = stdout.decode("utf-8", errors="replace") + if isinstance(stderr, bytes): + stderr = stderr.decode("utf-8", errors="replace") + return { + "command": command, + "exit_code": None, + "timed_out": True, + "stdout": stdout, + "stderr": stderr, + } + + +def run(args: argparse.Namespace) -> dict[str, Any]: + case_dir = args.case_dir.resolve() + checkpoint = args.checkpoint.resolve() + runtime_model = args.runtime_model.resolve() + if not checkpoint.is_file(): + raise FileNotFoundError(checkpoint) + if not runtime_model.is_file(): + raise FileNotFoundError(runtime_model) + if runtime_model.suffix.lower() != ".pt2": + raise ValueError("--runtime-model must be a device-specific .pt2 file") + input_path, data_path, runtime_target = _prepare_output( + case_dir, runtime_model, args.output, args.mode, args.device + ) + base_command = _load_command(args.command_json) + command = [ + *base_command, + "-in", + input_path.name, + "-var", + "runtime_model", + runtime_target.name, + ] + child_env = os.environ.copy() + child_env["OMP_NUM_THREADS"] = "1" + child_env["DP_INTRA_OP_PARALLELISM_THREADS"] = "1" + child_env["DP_INTER_OP_PARALLELISM_THREADS"] = "1" + if args.device == "cpu": + child_env["CUDA_VISIBLE_DEVICES"] = "" + else: + child_env["CUDA_VISIBLE_DEVICES"] = args.gpu_index + plugin_path, plugins, plugin_issues = resolve_lammps_plugin_path( + args.lammps_plugin_path, os.environ.get("LAMMPS_PLUGIN_PATH") + ) + if plugin_path: + child_env["LAMMPS_PLUGIN_PATH"] = plugin_path + + identity, identity_issues = identity_reasons( + checkpoint, args.image_ref, args.image_digest + ) + runtime_identity, runtime_issues = runtime_model_identity( + runtime_target, args.device, args.output + ) + identity["runtime_model_pt2"] = runtime_identity + environment = capture_runtime_environment() + environment["selected_environment"] = { + key: child_env.get(key) + for key in ( + "CONDA_DEFAULT_ENV", + "CUDA_VISIBLE_DEVICES", + "LAMMPS_PLUGIN_PATH", + "OMP_NUM_THREADS", + "DP_INTRA_OP_PARALLELISM_THREADS", + "DP_INTER_OP_PARALLELISM_THREADS", + ) + } + environment["lammps_plugins"] = plugins + setup_failures = single_rank_command_reasons( + base_command, {"lmp", "lmp_mpi", "lmp_serial", "lammps", "dpa4-lmp"} + ) + setup_failures.extend(plugin_issues) + setup_failures.extend( + t4_environment_reasons_for_device( + environment, args.gpu_index, args.device + ) + ) + started_at = utc_now() + start = time.monotonic() + checkpoint_probe = _checkpoint_probe( + checkpoint, child_env, min(args.timeout, 120.0), args.output + ) + checkpoint_probe_stdout = args.output / "dp_show.stdout.log" + checkpoint_probe_stderr = args.output / "dp_show.stderr.log" + checkpoint_probe_stdout.write_text(checkpoint_probe["stdout"], encoding="utf-8") + checkpoint_probe_stderr.write_text(checkpoint_probe["stderr"], encoding="utf-8") + execution_error = None + if ( + setup_failures + or checkpoint_probe["exit_code"] != 0 + or checkpoint_probe["timed_out"] + ): + execution_error = "LAMMPS skipped because setup or checkpoint probe did not pass" + execution = { + "exit_code": None, + "timed_out": False, + "stdout": "", + "stderr": execution_error, + "gpu_process_samples": [], + "root_pid": None, + } + else: + try: + execution = run_monitored(command, args.output, child_env, args.timeout) + except (FileNotFoundError, OSError) as exc: + execution_error = f"{type(exc).__name__}: {exc}" + execution = { + "exit_code": None, + "timed_out": False, + "stdout": "", + "stderr": execution_error, + "gpu_process_samples": [], + "root_pid": None, + } + # Successful DeepMD .pt2 GPU loading is explicit runtime evidence emitted + # by the backend. Keep the high-frequency nvidia-smi samples as the + # primary process proof, but retain this independent fallback for tiny + # run-0 jobs that can finish between process-table samples. + gpu_backend_loaded = bool( + re.search( + r"load model .*\.pt2 to gpu\s+\d+", + execution.get("stdout", "") + "\n" + execution.get("stderr", ""), + re.IGNORECASE, + ) + ) + duration = time.monotonic() - start + finished_at = utc_now() + stdout_path = args.output / "stdout.log" + stderr_path = args.output / "stderr.log" + stdout_path.write_text(execution["stdout"], encoding="utf-8") + stderr_path.write_text(execution["stderr"], encoding="utf-8") + + failure_reasons: list[str] = list(setup_failures) + inconclusive_reasons: list[str] = [] + if checkpoint_probe["timed_out"]: + failure_reasons.append("DP_SHOW_TIMEOUT") + if checkpoint_probe["exit_code"] != 0: + failure_reasons.append("DP_SHOW_NONZERO_EXIT_CODE") + failure_reasons.extend( + f"DP_SHOW:{reason}" + for reason in detect_fatal_output( + checkpoint_probe["stdout"], checkpoint_probe["stderr"] + ) + ) + if execution_error: + failure_reasons.append("COMMAND_NOT_STARTED") + if execution["timed_out"]: + failure_reasons.append("COMMAND_TIMEOUT") + if execution["exit_code"] != 0: + failure_reasons.append("NONZERO_EXIT_CODE") + failure_reasons.extend(detect_fatal_output(execution["stdout"], execution["stderr"])) + for reason in identity_issues: + if reason == "CHECKPOINT_SHA256_MISMATCH": + failure_reasons.append(reason) + else: + inconclusive_reasons.append(reason) + failure_reasons.extend(runtime_issues) + inconclusive_reasons.extend(missing_package_reasons(environment)) + + samples = execution["gpu_process_samples"] + device_evidence = { + "requested_device": args.device, + "cuda_visible_devices": child_env["CUDA_VISIBLE_DEVICES"], + "observed_benchmark_gpu_process": bool(samples), + "deepmd_backend_reported_gpu_load": gpu_backend_loaded, + "gpu_process_samples": samples, + "criterion": ( + "GPU requires a benchmark process or descendant in nvidia-smi; " + "CPU requires no such process while CUDA_VISIBLE_DEVICES is empty" + ), + } + if args.device == "gpu": + if not environment["gpu_inventory"]: + inconclusive_reasons.append("GPU_INVENTORY_NOT_OBSERVED") + if not environment.get("driver_version"): + inconclusive_reasons.append("NVIDIA_DRIVER_VERSION_NOT_OBSERVED") + if not environment.get("cuda_version"): + inconclusive_reasons.append("CUDA_VERSION_NOT_OBSERVED") + if not samples and not gpu_backend_loaded: + inconclusive_reasons.append("GPU_PROCESS_USAGE_NOT_OBSERVED") + elif samples: + failure_reasons.append("CPU_RUN_OBSERVED_ON_GPU") + + observables = None + try: + observables = _parse_result_line(execution["stdout"]) + observables["forces"] = _parse_force_dump( + args.output / "forces.dump", observables["natoms"] + ) + except (FileNotFoundError, ValueError) as exc: + failure_reasons.append(f"OBSERVABLE_PARSE_FAILED:{type(exc).__name__}:{exc}") + + if failure_reasons: + status = "failed" + reason_codes = sorted(set(failure_reasons + inconclusive_reasons)) + elif inconclusive_reasons: + status = "inconclusive" + reason_codes = sorted(set(inconclusive_reasons)) + else: + status = "passed" + reason_codes = [] + + artifacts = [ + input_path, + data_path, + runtime_target, + checkpoint_probe_stdout, + checkpoint_probe_stderr, + stdout_path, + stderr_path, + ] + force_dump = args.output / "forces.dump" + if force_dump.is_file(): + artifacts.append(force_dump) + record = { + "schema_version": SCHEMA_VERSION, + "benchmark_id": BENCHMARK_ID, + "kind": "lammps_case", + "status": status, + "reason_codes": reason_codes, + "case": case_dir.name, + "mode": args.mode, + "device": args.device, + "started_at": started_at, + "finished_at": finished_at, + "duration_seconds": duration, + "command": command, + "working_directory": str(args.output.resolve()), + "exit_code": execution["exit_code"], + "timed_out": execution["timed_out"], + "stdout": describe_file(stdout_path, args.output), + "stderr": describe_file(stderr_path, args.output), + "checkpoint_probe": { + "command": checkpoint_probe["command"], + "exit_code": checkpoint_probe["exit_code"], + "timed_out": checkpoint_probe["timed_out"], + "stdout": describe_file(checkpoint_probe_stdout, args.output), + "stderr": describe_file(checkpoint_probe_stderr, args.output), + }, + "identity": identity, + "environment": environment, + "device_evidence": device_evidence, + "observables": observables, + "artifacts": [describe_file(path, args.output) for path in artifacts], + } + write_json_new(args.output / "result.json", record) + return record + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--case-dir", required=True, type=Path) + parser.add_argument("--checkpoint", required=True, type=Path) + parser.add_argument( + "--runtime-model", + required=True, + type=Path, + help="device-specific DeepMD AOTI artifact; must end in .pt2", + ) + parser.add_argument("--mode", required=True, choices=("run0", "md")) + parser.add_argument("--device", required=True, choices=("cpu", "gpu")) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--image-ref", required=True) + parser.add_argument("--image-digest") + parser.add_argument( + "--command-json", + default='["lmp"]', + help='shell-free command prefix, e.g. ["mpirun","-n","1","lmp"]', + ) + parser.add_argument("--gpu-index", default="0") + parser.add_argument( + "--lammps-plugin-path", + help=( + "colon-separated plugin directories; defaults to LAMMPS_PLUGIN_PATH " + "and must contain libdeepmd_lmpplugin.so" + ), + ) + parser.add_argument("--timeout", type=float, default=300.0) + args = parser.parse_args(argv) + record = run(args) + print(f"{record['case']} {record['device']} {record['mode']}: {record['status']}") + for reason in record["reason_codes"]: + print(f" - {reason}") + return 0 if record["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/run_phonolammps_smoke.py b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/run_phonolammps_smoke.py new file mode 100644 index 00000000..a558f1ea --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/run_phonolammps_smoke.py @@ -0,0 +1,519 @@ +#!/usr/bin/env python3 +"""Run an APEX-shaped phonoLAMMPS smoke test and capture evidence.""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import re +import shutil +import subprocess +import time +from pathlib import Path +from typing import Any, Sequence + +from benchmark_lib import ( + BENCHMARK_ID, + SCHEMA_VERSION, + capture_runtime_environment, + describe_file, + detect_fatal_output, + identity_reasons, + missing_package_reasons, + resolve_lammps_plugin_path, + run_monitored, + runtime_model_identity, + single_rank_command_reasons, + t4_environment_reasons, + utc_now, + write_json_new, +) + + +def _command(raw: str) -> list[str]: + value = json.loads(raw) + if not isinstance(value, list) or not value or not all( + isinstance(item, str) and item for item in value + ): + raise ValueError("--command-json must be a non-empty JSON string array") + return value + + +def _poscar_atom_count(path: Path) -> int: + """Return the atom count from a VASP 4/5 POSCAR without external packages.""" + lines = path.read_text(encoding="utf-8").splitlines() + for index in (6, 5): + if index >= len(lines): + continue + tokens = lines[index].split() + try: + counts = [int(token) for token in tokens] + except ValueError: + continue + if counts and all(count > 0 for count in counts): + return sum(counts) + raise ValueError(f"could not read positive atom counts from POSCAR: {path}") + + +def validate_force_constants(path: Path, expected_atoms: int) -> dict[str, Any]: + """Parse a phonopy FORCE_CONSTANTS file and verify every finite 3x3 block.""" + if expected_atoms <= 0: + raise ValueError("expected atom count must be positive") + lines = [ + line.strip() + for line in path.read_text(encoding="utf-8").splitlines() + if line.strip() + ] + if not lines: + raise ValueError("FORCE_CONSTANTS is empty") + header = lines[0].split() + if len(header) != 1: + raise ValueError("FORCE_CONSTANTS header must contain exactly one atom count") + try: + atom_count = int(header[0]) + except ValueError as exc: + raise ValueError("FORCE_CONSTANTS atom count is not an integer") from exc + if atom_count <= 0: + raise ValueError("FORCE_CONSTANTS atom count must be positive") + if atom_count != expected_atoms: + raise ValueError( + f"FORCE_CONSTANTS atom count {atom_count} != expected {expected_atoms}" + ) + + cursor = 1 + finite_values = 0 + for first in range(1, atom_count + 1): + for second in range(1, atom_count + 1): + if cursor >= len(lines): + raise ValueError( + f"FORCE_CONSTANTS is truncated before block {first} {second}" + ) + pair_tokens = lines[cursor].split() + cursor += 1 + if len(pair_tokens) != 2: + raise ValueError( + f"FORCE_CONSTANTS block header {first} {second} is malformed" + ) + try: + pair = tuple(int(token) for token in pair_tokens) + except ValueError as exc: + raise ValueError( + f"FORCE_CONSTANTS block header {first} {second} is not integral" + ) from exc + if pair != (first, second): + raise ValueError( + "FORCE_CONSTANTS block order mismatch: " + f"expected {first} {second}, found {pair[0]} {pair[1]}" + ) + for row in range(3): + if cursor >= len(lines): + raise ValueError( + "FORCE_CONSTANTS is truncated in matrix block " + f"{first} {second}, row {row + 1}" + ) + values_raw = lines[cursor].split() + cursor += 1 + if len(values_raw) != 3: + raise ValueError( + "FORCE_CONSTANTS matrix row must contain three values: " + f"block {first} {second}, row {row + 1}" + ) + try: + values = [float(value) for value in values_raw] + except ValueError as exc: + raise ValueError( + "FORCE_CONSTANTS matrix row contains a non-number: " + f"block {first} {second}, row {row + 1}" + ) from exc + if not all(math.isfinite(value) for value in values): + raise ValueError( + "FORCE_CONSTANTS matrix row contains a non-finite value: " + f"block {first} {second}, row {row + 1}" + ) + finite_values += len(values) + if cursor != len(lines): + raise ValueError("FORCE_CONSTANTS contains trailing non-empty data") + return { + "status": "passed", + "expected_atom_count": expected_atoms, + "atom_count": atom_count, + "matrix_blocks": atom_count * atom_count, + "finite_values": finite_values, + "error": None, + } + + +def _prepare( + case_dir: Path, runtime_model: Path, output: Path +) -> tuple[Path, Path]: + if output.exists() and any(output.iterdir()): + raise FileExistsError(f"refusing to overwrite non-empty evidence directory: {output}") + output.mkdir(parents=True, exist_ok=True) + for name in ("POSCAR", "structure.data", "in.phonon.lammps"): + source = case_dir / name + if not source.is_file(): + raise FileNotFoundError(source) + shutil.copy2(source, output / name) + copied_model = output / "runtime.gpu.pt2" + shutil.copy2(runtime_model, copied_model) + input_path = output / "in.phonon.lammps" + text = input_path.read_text(encoding="utf-8") + if text.count("${runtime_model}") != 1: + raise ValueError( + "in.phonon.lammps must contain exactly one ${runtime_model} placeholder" + ) + input_path.write_text( + text.replace("${runtime_model}", copied_model.name), encoding="utf-8" + ) + return input_path, copied_model + + +def _checkpoint_probe( + checkpoint: Path, env: dict[str, str], timeout: float, cwd: Path +) -> dict[str, Any]: + command = [ + "dp", + "--pt", + "show", + str(checkpoint), + "type-map", + "descriptor", + "fitting-net", + "size", + ] + try: + completed = subprocess.run( + command, + cwd=str(cwd), + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + timeout=timeout, + check=False, + ) + return { + "command": command, + "exit_code": completed.returncode, + "timed_out": False, + "stdout": completed.stdout, + "stderr": completed.stderr, + } + except FileNotFoundError as exc: + return { + "command": command, + "exit_code": None, + "timed_out": False, + "stdout": "", + "stderr": f"FileNotFoundError: {exc}", + } + except subprocess.TimeoutExpired as exc: + stdout = exc.stdout or "" + stderr = exc.stderr or "" + if isinstance(stdout, bytes): + stdout = stdout.decode("utf-8", errors="replace") + if isinstance(stderr, bytes): + stderr = stderr.decode("utf-8", errors="replace") + return { + "command": command, + "exit_code": None, + "timed_out": True, + "stdout": stdout, + "stderr": stderr, + } + + +def run(args: argparse.Namespace) -> dict[str, Any]: + case_dir = args.case_dir.resolve() + checkpoint = args.checkpoint.resolve() + runtime_model = args.runtime_model.resolve() + if not checkpoint.is_file(): + raise FileNotFoundError(checkpoint) + if not runtime_model.is_file(): + raise FileNotFoundError(runtime_model) + if runtime_model.suffix.lower() != ".pt2": + raise ValueError("--runtime-model must be a GPU-specific .pt2 file") + if args.device != "gpu": + raise ValueError("this qualification smoke is T4 GPU-only") + input_path, copied_model = _prepare(case_dir, runtime_model, args.output) + base_command = _command(args.command_json) + command = [ + *base_command, + input_path.name, + "-c", + "POSCAR", + "--dim", + *(str(value) for value in args.dim), + "-pa", + *(str(value) for value in args.primitive_axes), + ] + child_env = os.environ.copy() + child_env["OMP_NUM_THREADS"] = "1" + child_env["DP_INTRA_OP_PARALLELISM_THREADS"] = "1" + child_env["DP_INTER_OP_PARALLELISM_THREADS"] = "1" + child_env["CUDA_VISIBLE_DEVICES"] = ( + "" if args.device == "cpu" else args.gpu_index + ) + plugin_path, plugins, plugin_issues = resolve_lammps_plugin_path( + args.lammps_plugin_path, os.environ.get("LAMMPS_PLUGIN_PATH") + ) + if plugin_path: + child_env["LAMMPS_PLUGIN_PATH"] = plugin_path + identity, identity_issues = identity_reasons( + checkpoint, args.image_ref, args.image_digest + ) + runtime_identity, runtime_issues = runtime_model_identity( + copied_model, "gpu", args.output + ) + identity["runtime_model_pt2"] = runtime_identity + + environment = capture_runtime_environment() + environment["selected_environment"] = { + key: child_env.get(key) + for key in ( + "CONDA_DEFAULT_ENV", + "CUDA_VISIBLE_DEVICES", + "LAMMPS_PLUGIN_PATH", + "OMP_NUM_THREADS", + "DP_INTRA_OP_PARALLELISM_THREADS", + "DP_INTER_OP_PARALLELISM_THREADS", + ) + } + environment["lammps_plugins"] = plugins + setup_failures = single_rank_command_reasons( + base_command, {"phonolammps", "dpa4-phonolammps"} + ) + setup_failures.extend(plugin_issues) + setup_failures.extend(t4_environment_reasons(environment, args.gpu_index)) + + started_at = utc_now() + start = time.monotonic() + checkpoint_probe = _checkpoint_probe( + checkpoint, child_env, min(args.timeout, 120.0), args.output + ) + checkpoint_stdout = args.output / "dp_show.stdout.log" + checkpoint_stderr = args.output / "dp_show.stderr.log" + checkpoint_stdout.write_text(checkpoint_probe["stdout"], encoding="utf-8") + checkpoint_stderr.write_text(checkpoint_probe["stderr"], encoding="utf-8") + execution_error = None + if ( + setup_failures + or checkpoint_probe["exit_code"] != 0 + or checkpoint_probe["timed_out"] + ): + execution_error = ( + "phonoLAMMPS skipped because setup or checkpoint probe did not pass" + ) + execution = { + "exit_code": None, + "timed_out": False, + "stdout": "", + "stderr": execution_error, + "gpu_process_samples": [], + "root_pid": None, + } + else: + try: + execution = run_monitored(command, args.output, child_env, args.timeout) + except (FileNotFoundError, OSError) as exc: + execution_error = f"{type(exc).__name__}: {exc}" + execution = { + "exit_code": None, + "timed_out": False, + "stdout": "", + "stderr": execution_error, + "gpu_process_samples": [], + "root_pid": None, + } + gpu_backend_loaded = bool( + re.search( + r"load model .*\.pt2 to gpu\s+\d+", + execution.get("stdout", "") + "\n" + execution.get("stderr", ""), + re.IGNORECASE, + ) + ) + duration = time.monotonic() - start + stdout_path = args.output / "stdout.log" + stderr_path = args.output / "stderr.log" + stdout_path.write_text(execution["stdout"], encoding="utf-8") + stderr_path.write_text(execution["stderr"], encoding="utf-8") + + failure_reasons: list[str] = list(setup_failures) + inconclusive_reasons: list[str] = [] + if checkpoint_probe["timed_out"]: + failure_reasons.append("DP_SHOW_TIMEOUT") + if checkpoint_probe["exit_code"] != 0: + failure_reasons.append("DP_SHOW_NONZERO_EXIT_CODE") + failure_reasons.extend( + f"DP_SHOW:{reason}" + for reason in detect_fatal_output( + checkpoint_probe["stdout"], checkpoint_probe["stderr"] + ) + ) + if execution_error: + failure_reasons.append("COMMAND_NOT_STARTED") + if execution["timed_out"]: + failure_reasons.append("COMMAND_TIMEOUT") + if execution["exit_code"] != 0: + failure_reasons.append("NONZERO_EXIT_CODE") + failure_reasons.extend(detect_fatal_output(execution["stdout"], execution["stderr"])) + for reason in identity_issues: + if reason == "CHECKPOINT_SHA256_MISMATCH": + failure_reasons.append(reason) + else: + inconclusive_reasons.append(reason) + failure_reasons.extend(runtime_issues) + inconclusive_reasons.extend(missing_package_reasons(environment)) + + samples = execution["gpu_process_samples"] + if args.device == "gpu": + if not environment["gpu_inventory"]: + inconclusive_reasons.append("GPU_INVENTORY_NOT_OBSERVED") + if not environment.get("driver_version"): + inconclusive_reasons.append("NVIDIA_DRIVER_VERSION_NOT_OBSERVED") + if not environment.get("cuda_version"): + inconclusive_reasons.append("CUDA_VERSION_NOT_OBSERVED") + if not samples and not gpu_backend_loaded: + inconclusive_reasons.append("GPU_PROCESS_USAGE_NOT_OBSERVED") + elif samples: + failure_reasons.append("CPU_RUN_OBSERVED_ON_GPU") + + force_constants = args.output / "FORCE_CONSTANTS" + force_constants_validation: dict[str, Any] = { + "status": "failed", + "expected_atom_count": None, + "atom_count": None, + "matrix_blocks": 0, + "finite_values": 0, + "error": "FORCE_CONSTANTS was not validated", + } + if not force_constants.is_file() or force_constants.stat().st_size == 0: + failure_reasons.append("FORCE_CONSTANTS_MISSING_OR_EMPTY") + force_constants_validation["error"] = "FORCE_CONSTANTS is missing or empty" + else: + try: + expected_atoms = _poscar_atom_count(args.output / "POSCAR") * math.prod( + args.dim + ) + force_constants_validation = validate_force_constants( + force_constants, expected_atoms + ) + except (OSError, ValueError) as exc: + failure_reasons.append("FORCE_CONSTANTS_INVALID") + force_constants_validation["error"] = f"{type(exc).__name__}: {exc}" + + if failure_reasons: + status = "failed" + reasons = sorted(set(failure_reasons + inconclusive_reasons)) + elif inconclusive_reasons: + status = "inconclusive" + reasons = sorted(set(inconclusive_reasons)) + else: + status = "passed" + reasons = [] + + artifact_names = [ + "POSCAR", + "structure.data", + "in.phonon.lammps", + "runtime.gpu.pt2", + "dp_show.stdout.log", + "dp_show.stderr.log", + "stdout.log", + "stderr.log", + "FORCE_CONSTANTS", + "phonopy_disp.yaml", + "phonopy.yaml", + ] + artifacts = [ + describe_file(args.output / name, args.output) + for name in artifact_names + if (args.output / name).is_file() + ] + record = { + "schema_version": SCHEMA_VERSION, + "benchmark_id": BENCHMARK_ID, + "kind": "phonolammps_smoke", + "status": status, + "reason_codes": reasons, + "case": case_dir.name, + "device": args.device, + "supercell_dimension": args.dim, + "primitive_axes": args.primitive_axes, + "started_at": started_at, + "finished_at": utc_now(), + "duration_seconds": duration, + "command": command, + "working_directory": str(args.output.resolve()), + "exit_code": execution["exit_code"], + "timed_out": execution["timed_out"], + "stdout": describe_file(stdout_path, args.output), + "stderr": describe_file(stderr_path, args.output), + "checkpoint_probe": { + "command": checkpoint_probe["command"], + "exit_code": checkpoint_probe["exit_code"], + "timed_out": checkpoint_probe["timed_out"], + "stdout": describe_file(checkpoint_stdout, args.output), + "stderr": describe_file(checkpoint_stderr, args.output), + }, + "identity": identity, + "environment": environment, + "device_evidence": { + "requested_device": args.device, + "cuda_visible_devices": child_env["CUDA_VISIBLE_DEVICES"], + "observed_benchmark_gpu_process": bool(samples), + "deepmd_backend_reported_gpu_load": gpu_backend_loaded, + "gpu_process_samples": samples, + }, + "force_constants_validation": force_constants_validation, + "artifacts": artifacts, + } + write_json_new(args.output / "result.json", record) + return record + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--case-dir", required=True, type=Path) + parser.add_argument("--checkpoint", required=True, type=Path) + parser.add_argument( + "--runtime-model", + required=True, + type=Path, + help="T4 GPU-specific DeepMD AOTI artifact; must end in .pt2", + ) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--image-ref", required=True) + parser.add_argument("--image-digest") + parser.add_argument("--device", choices=("gpu",), default="gpu") + parser.add_argument("--gpu-index", default="0") + parser.add_argument( + "--lammps-plugin-path", + help=( + "colon-separated plugin directories; defaults to LAMMPS_PLUGIN_PATH " + "and must contain libdeepmd_lmpplugin.so" + ), + ) + parser.add_argument("--dim", nargs=3, type=int, default=[2, 2, 2]) + parser.add_argument( + "--primitive-axes", + nargs=9, + type=float, + default=[1.0, 0.0, 0.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0], + ) + parser.add_argument("--command-json", default='["phonolammps"]') + parser.add_argument("--timeout", type=float, default=600.0) + args = parser.parse_args(argv) + if any(value <= 0 for value in args.dim): + parser.error("all --dim values must be positive") + record = run(args) + print(f"{record['case']} phonoLAMMPS {record['device']}: {record['status']}") + for reason in record["reason_codes"]: + print(f" - {reason}") + return 0 if record["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/verify_manifest.py b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/verify_manifest.py new file mode 100644 index 00000000..001d6031 --- /dev/null +++ b/apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/verify_manifest.py @@ -0,0 +1,304 @@ +#!/usr/bin/env python3 +"""Verify manifest invariants and every declared artifact hash without extras.""" + +from __future__ import annotations + +import argparse +import math +from pathlib import Path +from typing import Sequence + +from benchmark_lib import ( + BENCHMARK_ID, + EXPECTED_MODEL_SHA256, + HEX_SHA256_RE, + REQUIRED_PACKAGES, + SHA256_RE, + load_json, + sha256_file, +) + + +def _safe_artifact_path(root: Path, raw: str) -> Path: + path = Path(raw) + resolved = path.resolve() if path.is_absolute() else (root / path).resolve() + try: + resolved.relative_to(root.resolve()) + except ValueError as exc: + raise ValueError(f"artifact path escapes manifest root: {raw}") from exc + return resolved + + +def verify(manifest_path: Path, root: Path) -> list[str]: + manifest = load_json(manifest_path) + errors: list[str] = [] + required_top = { + "$schema", + "schema_version", + "benchmark_id", + "created_at", + "status", + "reason_codes", + "provenance", + "identity", + "environment", + "thresholds", + "cases", + "runs", + "parity", + "phonolammps", + "artifact_hashes", + "notes", + } + missing_top = sorted(required_top - set(manifest)) + if missing_top: + errors.append("missing top-level fields: " + ", ".join(missing_top)) + if manifest.get("schema_version") != "2.0.0": + errors.append("schema_version mismatch") + if manifest.get("benchmark_id") != BENCHMARK_ID: + errors.append("benchmark_id mismatch") + status = manifest.get("status") + if status not in {"untested", "inconclusive", "failed", "passed"}: + errors.append(f"invalid status: {status!r}") + checkpoint = manifest.get("identity", {}).get("checkpoint_pt", {}) + if checkpoint.get("sha256") != EXPECTED_MODEL_SHA256: + errors.append("checkpoint model.pt sha256 mismatch") + if checkpoint.get("bytes") != 30403297: + errors.append("checkpoint model.pt size mismatch") + pt2 = manifest.get("identity", {}).get("runtime_models_pt2") + if not isinstance(pt2, list): + errors.append("identity.runtime_models_pt2 must be an array") + else: + for index, descriptor in enumerate(pt2): + if not isinstance(descriptor, dict): + errors.append(f"runtime_models_pt2[{index}] is not an object") + continue + if not isinstance(descriptor.get("sha256"), str) or not HEX_SHA256_RE.fullmatch( + descriptor["sha256"] + ): + errors.append(f"runtime_models_pt2[{index}] has invalid sha256") + if not isinstance(descriptor.get("bytes"), int) or descriptor["bytes"] <= 0: + errors.append(f"runtime_models_pt2[{index}] has invalid size") + if descriptor.get("device") not in {"cpu", "gpu"}: + errors.append(f"runtime_models_pt2[{index}] has invalid device") + sources = descriptor.get("sources") + if not isinstance(sources, list) or not sources or not all( + isinstance(source, str) and source for source in sources + ): + errors.append(f"runtime_models_pt2[{index}] has invalid sources") + digest = manifest.get("identity", {}).get("image", {}).get("digest") + if digest is not None and not ( + isinstance(digest, str) and SHA256_RE.fullmatch(digest) + ): + errors.append("image digest is present but invalid") + if status == "passed" and not (isinstance(digest, str) and SHA256_RE.fullmatch(digest)): + errors.append("passed manifest lacks full image digest") + reasons = manifest.get("reason_codes") + if not isinstance(reasons, list) or not all( + isinstance(reason, str) and reason for reason in reasons + ): + errors.append("reason_codes must be non-empty strings") + elif len(reasons) != len(set(reasons)): + errors.append("reason_codes contains duplicates") + + expected_thresholds = { + "energy_per_atom_abs_eV", + "force_rms_eV_per_A", + "force_max_abs_eV_per_A", + "stress_max_abs_GPa", + } + thresholds = manifest.get("thresholds") + if not isinstance(thresholds, dict) or set(thresholds) != expected_thresholds: + errors.append("threshold fields mismatch") + elif any( + not isinstance(value, (int, float)) + or not math.isfinite(float(value)) + or value < 0 + for value in thresholds.values() + ): + errors.append("threshold values must be finite and non-negative") + + cases = manifest.get("cases") + if not isinstance(cases, list): + errors.append("cases must be an array") + else: + case_names = [case.get("name") for case in cases if isinstance(case, dict)] + if sorted(case_names) != sorted(("Ti_hcp", "V_bcc", "TiV_B2")): + errors.append("cases must contain Ti_hcp, V_bcc, and TiV_B2 exactly once") + for key in ("runs", "parity", "phonolammps", "artifact_hashes", "notes"): + if not isinstance(manifest.get(key), list): + errors.append(f"{key} must be an array") + if not isinstance(manifest.get("environment", {}).get("observations"), list): + errors.append("environment.observations must be an array") + + seen_paths: set[str] = set() + for descriptor in manifest.get("artifact_hashes", []): + raw = descriptor.get("path") + expected = descriptor.get("sha256") + size = descriptor.get("bytes") + if not isinstance(raw, str) or not raw: + errors.append("artifact has invalid path") + continue + if raw in seen_paths: + errors.append(f"duplicate artifact path: {raw}") + seen_paths.add(raw) + if not isinstance(expected, str) or not HEX_SHA256_RE.fullmatch(expected): + errors.append(f"artifact has invalid sha256: {raw}") + continue + try: + path = _safe_artifact_path(root, raw) + except ValueError as exc: + errors.append(str(exc)) + continue + if not path.is_file(): + errors.append(f"artifact missing: {raw}") + continue + if path.stat().st_size != size: + errors.append(f"artifact size mismatch: {raw}") + if sha256_file(path) != expected: + errors.append(f"artifact sha256 mismatch: {raw}") + + for collection in ("runs", "parity", "phonolammps"): + for index, summary in enumerate(manifest.get(collection, [])): + if not isinstance(summary, dict): + errors.append(f"{collection}[{index}] is not an object") + continue + for field in ("result", "stdout", "stderr"): + if field not in summary or summary[field] is None: + if field in ("stdout", "stderr") and collection == "parity": + continue + errors.append(f"{collection}[{index}] lacks {field} evidence") + continue + raw = summary[field].get("path") + if raw not in seen_paths: + errors.append( + f"{collection}[{index}] {field} is absent from artifact_hashes" + ) + + if status == "passed": + runs = manifest.get("runs", []) + parity = manifest.get("parity", []) + phonon = manifest.get("phonolammps", []) + if len(runs) != 12 or any(item.get("status") != "passed" for item in runs): + errors.append("passed manifest must contain exactly 12 passed LAMMPS runs") + if len(parity) != 6 or any(item.get("status") != "passed" for item in parity): + errors.append("passed manifest must contain exactly 6 passed parity records") + passed_phonon = False + for item in phonon: + if not isinstance(item, dict): + continue + validation = item.get("force_constants_validation") + if not isinstance(validation, dict): + continue + atom_count = validation.get("atom_count") + expected_count = validation.get("expected_atom_count") + complete_validation = ( + validation.get("status") == "passed" + and isinstance(atom_count, int) + and atom_count > 0 + and atom_count == expected_count + and validation.get("matrix_blocks") == atom_count * atom_count + and validation.get("finite_values") == 9 * atom_count * atom_count + and validation.get("error") is None + ) + if ( + item.get("case") == "Ti_hcp" + and item.get("device") == "gpu" + and item.get("status") == "passed" + and complete_validation + ): + passed_phonon = True + break + if not passed_phonon: + errors.append( + "passed manifest lacks Ti_hcp GPU phonoLAMMPS smoke with " + "complete finite FORCE_CONSTANTS validation" + ) + if any(case.get("status") != "passed" for case in manifest.get("cases", [])): + errors.append("passed manifest contains a non-passed generated case") + observations = manifest.get("environment", {}).get("observations", []) + if not observations: + errors.append("passed manifest lacks runtime environment observations") + if observations and not any( + observation.get("gpu_inventory") + and observation.get("driver_version") + and observation.get("cuda_version") + for observation in observations + ): + errors.append("passed manifest lacks a complete GPU/driver/CUDA observation") + for observation in observations: + for gpu in observation.get("gpu_inventory", []): + if not str(gpu.get("name", "")).strip().lower().endswith("t4"): + errors.append("passed manifest contains a non-T4 GPU observation") + if str(gpu.get("compute_capability", "")).strip() != "7.5": + errors.append("passed manifest T4 compute capability is not 7.5") + for observation in observations: + packages = observation.get("packages", {}) + missing = [name for name in REQUIRED_PACKAGES if not packages.get(name)] + if missing: + errors.append( + "environment observation lacks package versions: " + ", ".join(missing) + ) + plugin_path = observation.get("selected_environment", {}).get( + "LAMMPS_PLUGIN_PATH" + ) + if not plugin_path: + errors.append("passed manifest lacks LAMMPS_PLUGIN_PATH") + plugins = observation.get("lammps_plugins", []) + if not plugins: + errors.append("passed manifest lacks hashed b95 LAMMPS plugin evidence") + elif not any( + Path(plugin.get("path", "")).name == "libdeepmd_lmpplugin.so" + and isinstance(plugin.get("sha256"), str) + and HEX_SHA256_RE.fullmatch(plugin["sha256"]) + for plugin in plugins + ): + errors.append( + "passed manifest lacks libdeepmd_lmpplugin.so hash evidence" + ) + runtime_models = manifest.get("identity", {}).get("runtime_models_pt2", []) + if len(runtime_models) != 2: + errors.append("passed manifest must contain exactly CPU and GPU .pt2 identities") + else: + devices = {item.get("device") for item in runtime_models} + hashes = {item.get("sha256") for item in runtime_models} + if devices != {"cpu", "gpu"}: + errors.append("passed manifest .pt2 identities must cover CPU and GPU") + if len(hashes) != 2: + errors.append("passed manifest CPU/GPU .pt2 hashes must be distinct") + if manifest.get("reason_codes"): + errors.append("passed manifest must have no reason_codes") + elif status == "untested": + if not manifest.get("reason_codes"): + errors.append("untested manifest must explain missing execution evidence") + if manifest.get("runs") or manifest.get("parity") or manifest.get("phonolammps"): + errors.append("untested manifest must not contain runtime result summaries") + elif status in {"failed", "inconclusive"}: + if not (manifest.get("runs") or manifest.get("parity") or manifest.get("phonolammps")): + errors.append(f"{status} manifest lacks runtime result summaries") + return errors + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("manifest", type=Path) + parser.add_argument( + "--root", + type=Path, + help="artifact root (default: manifest parent)", + ) + args = parser.parse_args(argv) + root = (args.root or args.manifest.parent).resolve() + errors = verify(args.manifest, root) + if errors: + print("manifest verification FAILED") + for error in errors: + print(f" - {error}") + return 1 + print("manifest verification PASSED") + print("This verifies structure/hashes only; compatibility requires status=passed.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/apex/skills/apex-flow/data/default_templates.json b/apex/skills/apex-flow/data/default_templates.json index 556d53f0..8ea2e665 100644 --- a/apex/skills/apex-flow/data/default_templates.json +++ b/apex/skills/apex-flow/data/default_templates.json @@ -1,28 +1,37 @@ { "apex_image": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", + "dispatcher_image_sandbox": "registry.dp.tech/dptech/polycalibur:dpdispatcher-storehost-plan-a-20260811", "backends": { "lammps_gpu": { - "image": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", - "machine": "c8_m31_1 * NVIDIA T4", + "image": "registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2", + "machine": "c16_m120_1 * NVIDIA L20", + "machine_sandbox": "c16_m120_1 * NVIDIA L20", "potentials": ["deepmd", "mace", "nep"] }, "lammps_cpu": { "image": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", "machine": "c16_m32_cpu", + "machine_sandbox": "c8_m32_cpu", "potentials": ["gap", "snap", "rann", "eam_alloy", "eam_fs", "meam", "meam_spline"] }, "abacus": { "image": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", "machine": "c16_m32_cpu", + "machine_sandbox": "c8_m32_cpu", "run_command": "mpirun -n 8 abacus" }, "vasp": { "image": null, "machine": "c32_m128_cpu", + "machine_sandbox": "c32_m128_cpu", "run_command": "bash -c \"source /opt/intel/oneapi/setvars.sh && ulimit -s unlimited && mpirun -n 32 /opt/vasp.5.4.4/bin/vasp_std\"", "note": "Commercial software. Resolve image via Bohrium list_images(keyword=vasp) or a user-known authorized address; if neither exists, stop. Never invent a default image. Never use bare mpirun ... vasp_std." } }, + "sandbox_machines": { + "cpu": ["c2_m4_cpu", "c2_m8_cpu", "c8_m32_cpu", "c32_m128_cpu", "c64_m256_cpu"], + "gpu": ["c16_m120_1 * NVIDIA L20", "c8_m32_1 * NVIDIA 4090", "c16_m64_1 * NVIDIA 4090", "c16_m64_1 * NVIDIA 5090"] + }, "properties": { "eos": {"lammps": true, "abacus": true, "vasp": true}, "cohesive": {"lammps": true, "abacus": true, "vasp": true}, @@ -34,9 +43,10 @@ "gamma": {"lammps": true, "abacus": true, "vasp": true}, "gamma_surface": {"lammps": true, "abacus": true, "vasp": true}, "decohesive": {"lammps": true, "abacus": true, "vasp": true}, - "finite_t_latt": {"lammps": true, "abacus": false, "vasp": false}, + "finite_t_latt": {"lammps": true, "abacus": false, "vasp": true}, "finite_t_elastic": {"lammps": true, "abacus": false, "vasp": false}, "gruneisen": {"lammps": true, "abacus": true, "vasp": true}, - "annealing": {"lammps": true, "abacus": false, "vasp": false} + "annealing": {"lammps": true, "abacus": false, "vasp": true}, + "melting_point": {"lammps": true, "abacus": false, "vasp": false} } } diff --git a/apex/skills/apex-flow/data/dpa4_alloytongqi_t4_profile.json b/apex/skills/apex-flow/data/dpa4_alloytongqi_t4_profile.json new file mode 100644 index 00000000..3a622fd1 --- /dev/null +++ b/apex/skills/apex-flow/data/dpa4_alloytongqi_t4_profile.json @@ -0,0 +1,52 @@ +{ + "schema_version": 1, + "profile_id": "dpa4-alloytongqi-t4", + "description": "Candidate DPA4 alloytongqi runtime compiled for one NVIDIA T4 (SM 7.5); fail closed until an immutable image identity and post-snapshot evidence are recorded.", + "qualification_status": "pre_snapshot_only", + "image": { + "ref": "__DPA4_IMAGE_REF__", + "digest": "__DPA4_IMAGE_DIGEST__" + }, + "calculator": { + "backend": "lammps", + "potential": "deepmd", + "run_command": "/usr/local/bin/dpa4-lmp -in in.lammps", + "phonolammps_command": "/usr/local/bin/dpa4-phonolammps {input_file} -c {poscar} --dim {dim} {primitive_axes}" + }, + "runtime": { + "kind": "dpa4_pt2", + "model_in_image": true, + "model_path": "/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.t4-sm75.pt2", + "model_sha256": "2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf", + "source_checkpoint_path": "/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt", + "source_checkpoint_sha256": "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad", + "type_map": "auto" + }, + "machine_compatibility": { + "recommended": [ + { + "scass_type": "c4_m15_1 * NVIDIA T4", + "mpi_ranks": 1, + "gpu_count": 1, + "status": "candidate_after_publish", + "evidence": "Pre-snapshot only: 12 LAMMPS runs, 6 CPU/GPU parity checks, and one phonoLAMMPS smoke completed on c4_m15_1 * NVIDIA T4. This is not exact-image or registry-digest acceptance." + } + ], + "prohibited_exact": { + "c4_m16_cpu": "CPU execution cannot load the production T4 PT2 artifact and this SKU is also unavailable on Bohrium.", + "c12_m46_1 * NVIDIA T4": "This Bohrium machine type does not exist." + }, + "prohibited_classes": { + "cpu": "The qualified artifact is compiled for NVIDIA T4 SM 7.5, not CPU.", + "multi_gpu": "Only one rank on one T4 was tested; multi-GPU or multi-rank execution is prohibited for this frozen PT2 profile.", + "cross_arch_pt2": "The T4 SM 7.5 PT2 artifact must not be reused on another GPU architecture.", + "sm70_or_older": "V100 (SM 7.0) and older GPUs are unsupported by this CUDA 13 runtime contract.", + "driver_lt_580_65_06": "CUDA 13 requires an NVIDIA Linux driver at least 580.65.06 for this runtime contract." + }, + "unverified": [ + "Any other single-GPU T4 SKU, including c8_m31_1 and c16_m62_1", + "Any newer non-T4 GPU with its own architecture-specific PT2, including A10, A100/A800, L4, H100/H20, and RTX 4090" + ], + "unknown_policy": "fail_closed" + } +} diff --git a/apex/skills/apex-flow/data/global_bohrium_direct.json b/apex/skills/apex-flow/data/global_bohrium_direct.json new file mode 100644 index 00000000..e4b20cd1 --- /dev/null +++ b/apex/skills/apex-flow/data/global_bohrium_direct.json @@ -0,0 +1,12 @@ +{ + "dflow_host": "https://workflows.deepmodeling.com", + "k8s_api_server": "https://workflows.deepmodeling.com", + "batch_type": "Bohrium", + "context_type": "Bohrium", + "apex_image_name": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", + "lammps_image_name": "registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2", + "lammps_run_command": "lmp -in in.lammps", + "scass_type": "c16_m120_1 * NVIDIA L20", + "group_size": 1, + "pool_size": 1 +} diff --git a/apex/skills/apex-flow/data/global_local_cluster_slurm.json b/apex/skills/apex-flow/data/global_local_cluster_slurm.json new file mode 100644 index 00000000..42c3c79a --- /dev/null +++ b/apex/skills/apex-flow/data/global_local_cluster_slurm.json @@ -0,0 +1,22 @@ +{ + "context_type": "Local", + "run_command": "", + "machine": { + "batch_type": "Slurm", + "context_type": "Local", + "local_root": "./", + "remote_root": "", + "clean_asynchronously": true + }, + "resources": { + "number_node": 1, + "cpu_per_node": 1, + "gpu_per_node": 0, + "group_size": 1, + "module_list": [], + "custom_flags": [ + "#SBATCH --partition=", + "#SBATCH --time=" + ] + } +} diff --git a/apex/skills/apex-flow/data/global_local_debug.json b/apex/skills/apex-flow/data/global_local_debug.json new file mode 100644 index 00000000..2e2e5187 --- /dev/null +++ b/apex/skills/apex-flow/data/global_local_debug.json @@ -0,0 +1,7 @@ +{ + "context_type": "Local", + "batch_type": "Shell", + "local_root": "./", + "remote_root": "./", + "run_command": "lmp -in in.lammps" +} diff --git a/apex/skills/apex-flow/models/DPA-3.2-5M/README.md b/apex/skills/apex-flow/models/DPA-3.2-5M/README.md deleted file mode 100644 index 8b23a5d0..00000000 --- a/apex/skills/apex-flow/models/DPA-3.2-5M/README.md +++ /dev/null @@ -1,14 +0,0 @@ -# DPA-3.2-5M - -- `DPA-3.2-5M-OMat24.pth` — frozen single-task PyTorch model, ready for - `interaction.model` in APEX. -- Source: official DPA-3.2-5M multi-task checkpoint. -- Frozen head: `OMat24`. -- DeePMD-kit version: 3.1.3. -- Observed-element coverage: 89 elements, including O. -- SHA-256: - `055fbbcb83c9063f7809a74803c600eb34cf0b3f7caf15e0e0d2b86834f30e8e`. - -Use `"type_map": "auto"` and copy the `.pth` file into the submitted job -directory. The original multi-task `.pt` checkpoint is intentionally excluded -from the skill zip. diff --git a/apex/skills/apex-flow/models/DPA4-alloytongqi/README.md b/apex/skills/apex-flow/models/DPA4-alloytongqi/README.md new file mode 100644 index 00000000..2ceb9261 --- /dev/null +++ b/apex/skills/apex-flow/models/DPA4-alloytongqi/README.md @@ -0,0 +1,24 @@ +# DPA4 alloytongqi + +- File: `model.pt` +- Model type: DPA4 (`dpa4_ener` fitting) +- Task form: single/default task (`model_branch_alias=[]`) +- Branch provenance: `alloytongqi`, supplied by the user; not embedded in the checkpoint +- Size: 30,403,297 bytes +- SHA-256: `c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad` +- APEX default/phonon/Grüneisen image: `registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3` +- Compatibility result: the old tag's Bohrium registry mirror, repo digest + `sha256:43a27ca4a7bba7f774bbd56104d205a6a80cd9d65928f249f6109e9ef37b8402`, + failed with `RuntimeError: Unknown model type: dpa4`; LAMMPS aborted while + initializing `pair_style deepmd`. + +Keep `model.pt` as the source identity; never pass it directly to LAMMPS under +the DPA4 T4 profile. That profile is currently locked by image placeholders and +`pre_snapshot_only` qualification. Pre-snapshot checks passed on one rank/one +GPU at `c4_m15_1 * NVIDIA T4`, but this is not exact-image acceptance; c8/c16 +T4 and non-T4 GPUs remain unverified. After an immutable digest rerun, generate +the hashed image-resident `.pt2` interaction with `--runtime-profile +dpa4-alloytongqi-t4`; its generated commands use absolute +`/usr/local/bin/dpa4-lmp` and `/usr/local/bin/dpa4-phonolammps` wrappers. The configured `registry.dp.tech` endpoint was +pull-denied; the legacy digest above identifies the tested artifact from +`registry.bohrium.dp.tech`. diff --git a/apex/skills/apex-flow/models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth b/apex/skills/apex-flow/models/DPA4-alloytongqi/model.pt similarity index 56% rename from apex/skills/apex-flow/models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth rename to apex/skills/apex-flow/models/DPA4-alloytongqi/model.pt index 71432055..6d825449 100644 Binary files a/apex/skills/apex-flow/models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth and b/apex/skills/apex-flow/models/DPA4-alloytongqi/model.pt differ diff --git a/apex/skills/apex-flow/models/README.md b/apex/skills/apex-flow/models/README.md index 863074a5..64a35e1e 100644 --- a/apex/skills/apex-flow/models/README.md +++ b/apex/skills/apex-flow/models/README.md @@ -1,44 +1,69 @@ -# Bundled DPA model +# Bundled DPA4 model > **Naming:** APEX **backend** = `lammps` / `abacus` / `vasp`. -> DPA-3.2-5M is a DeePMD model used by the LAMMPS backend. +> DPA4 alloytongqi is a DeePMD model used by the LAMMPS backend. -The skill ships one ready-to-run frozen model: +The skill ships exactly one model: -| Path | Size | Branch | Ready for APEX? | -|------|------|--------|-----------------| -| `DPA-3.2-5M/DPA-3.2-5M-OMat24.pth` | ~23MB | `OMat24` | Yes | +| Path | Size | Task/branch provenance | +|------|------|------------------------| +| `DPA4-alloytongqi/model.pt` | 30,403,297 bytes | Single/default task; `alloytongqi` supplied by the user | -The bundled model is a single-task PyTorch model frozen from the official -DPA-3.2-5M checkpoint with DeePMD-kit 3.1.3. Its `OMat24` branch has 89 -observed elements, including oxygen. The model `type_map` spans all 118 element -symbols, but an element should be considered supported only when it appears in -the branch's observed-element list. +SHA-256: +`c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad`. + +Safe static inspection confirms `model_params.type=dpa4`, fitting type +`dpa4_ener`, and `model_branch_alias=[]`. The file does not embed the literal +branch name or a Git commit, so `alloytongqi` is recorded as user-provided +provenance. Do not infer observed-element or training-domain support from the +type map alone. + +## Runtime compatibility boundary + +APEX keeps this old image as both its default LAMMPS image and its forced +LAMMPS phonon/Grüneisen image: + +`registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3` + +On 2026-08-10, the configured `registry.dp.tech` endpoint was pull-denied, but +the same repository path/tag was available from the Bohrium registry mirror. +The tested mirror artifact was linux/amd64 with repo digest +`sha256:43a27ca4a7bba7f774bbd56104d205a6a80cd9d65928f249f6109e9ef37b8402`, +DeepMD-kit 3.1.3, phonoLAMMPS 0.10.1, and LAMMPS 29 Aug 2024. Against the +bundled file matching the SHA-256 above: + +- `dp show` failed with `RuntimeError: Unknown model type: dpa4`. +- LAMMPS 29 Aug 2024 aborted while initializing `pair_style deepmd`; it never + reached `run 0`. + +This proves that digest's DeepMD-kit 3.1.3 runtime cannot execute the model. +Retain the mirror registry domain and digest when reporting provenance; the +configured registry endpoint itself was not readable. Do not submit this model +with the default old image. Keep that image as the general default and legacy +phonon/Grüneisen forced runtime. + +The new `dpa4-alloytongqi-t4` profile is still unpublished: its image ref and +digest are placeholders and its qualification is `pre_snapshot_only`. +Pre-snapshot checks completed 12 LAMMPS runs, 6 CPU/GPU parity comparisons, +and one phonoLAMMPS smoke on one rank/one GPU at +`c4_m15_1 * NVIDIA T4`. This is candidate evidence, not exact-image or registry +acceptance. c8/c16 T4, every non-T4 GPU, CPU, multi-rank, multi-GPU, and +cross-architecture PT2 reuse remain unverified or prohibited. +V100/SM 7.0 and older devices and NVIDIA Linux drivers below 580.65.06 are +prohibited by this CUDA 13 runtime contract. ## Agent workflow -1. For LAMMPS + DeePMD/DPA, copy - `models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth` into the job directory. -2. Set `"type": "deepmd"`, `"model": "DPA-3.2-5M-OMat24.pth"`, and - `"type_map": "auto"`. APEX infers a zero-based, contiguous map from the - structure during submission. Do not use atomic numbers or model-internal - indices. -3. If the user explicitly needs another DPA-3.2 task head, download the source - checkpoint outside the skill zip: - ```bash - python scripts/fetch_models.py --source-checkpoint - # or: - dp --pt pretrained download DPA-3.2-5M - ``` -4. Freeze the selected head before LAMMPS/APEX use: - ```bash - dp --pt freeze -c DPA-3.2-5M.pt -o DPA-3.2-5M-Alloy_APEX.pth --head Alloy_APEX - ``` - -Never pass the multi-task `.pt` checkpoint directly to LAMMPS. Do not invent a -model filename. If the OMat24 observed-element coverage or training domain is -unsuitable, explain the limitation and obtain the user's task-head choice. - -## Source - -- Official checkpoint: [DPA-3.2-5M](https://huggingface.co/deepmodelingcommunity/DPA-3.2-5M) +1. Retain `models/DPA4-alloytongqi/model.pt` as the source identity; do not pass + it to LAMMPS. +2. Inspect the candidate with `validate_apex_combo.py list-combos + --runtime-profile dpa4-alloytongqi-t4`. +3. Stop while `recommend` and `generate_config.py create --runtime-profile + dpa4-alloytongqi-t4` fail closed. After exact-image qualification, use the + generated image-resident `.pt2` contract and audited wrapper commands only. + The exact wrappers are `/usr/local/bin/dpa4-lmp` and + `/usr/local/bin/dpa4-phonolammps`. + +The skill contains no alternate DPA model or old-model downloader. APEX infers +a zero-based, contiguous element map from the structure; do not substitute +atomic numbers or model-internal indices. diff --git a/apex/skills/apex-flow/reference/calculators.md b/apex/skills/apex-flow/reference/calculators.md index 11c887e2..55b712c6 100644 --- a/apex/skills/apex-flow/reference/calculators.md +++ b/apex/skills/apex-flow/reference/calculators.md @@ -6,7 +6,7 @@ APEX supports three calculator **backends**: LAMMPS, ABACUS, and VASP. Each requires specific configuration in `param.json` under the `"interaction"` key. > APEX **backend** = calculator (`lammps` / `abacus` / `vasp`). -> DPA-3.2-5M is a DeePMD **model** under LAMMPS (`interaction.type: deepmd`). +> DPA4 alloytongqi is a DeePMD **model** under LAMMPS (`interaction.type: deepmd`). > Ask calculator backend first; if LAMMPS+DeePMD, then ask which model file. --- @@ -29,7 +29,7 @@ Each requires specific configuration in `param.json` under the `"interaction"` k | type | pair_style | Model File | Notes | |------|-----------|-----------|-------| -| `deepmd` | `deepmd` | `.pb` or `.pth` | DeePMD-kit model | +| `deepmd` | `deepmd` | `.pb`, `.pth`, or compatible single-task `.pt` | DeePMD-kit model | | `mace` | `mace no_domain_decomposition` | `.model` | MACE model | | `nep` | `nep` | `nep.txt` | NEP potential | | `gap` | `quip` | `.xml` + `.xml.sparseX.TERM` | GAP/QUIP potential | @@ -46,7 +46,7 @@ Each requires specific configuration in `param.json` under the `"interaction"` k { "interaction": { "type": "deepmd", - "model": "frozen_model.pb", + "model": "model.pt", "type_map": "auto" } } @@ -100,86 +100,85 @@ Each requires specific configuration in `param.json` under the `"interaction"` k } ``` -### ⚠️ LAMMPS Image Version +### Default LAMMPS image and DPA4 compatibility -Default image: `registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3` +GPU image (`deepmd`, `mace`, `nep`): +`registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2` -> ⚠️ **Do NOT use `deepmd-kit:3.1.1` with GPU/T4** — known startup / triclinic issues. Prefer `3.1.3` or later. +CPU image (`gap`, `snap`, `rann`, `eam_alloy`, `eam_fs`, `meam`, +`meam_spline`): +`registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` -> **LAMMPS phonon and Grüneisen**: `apex submit` forces the validated default image above. It supports the tested NVIDIA T4 configuration and includes phonoLAMMPS. +This image uses integrated USER-DEEPMD and CUDA 12.8. Do not add +`plugin load libdeepmd_lmp.so`; the plugin command is unsupported. Its unified +`lmp` entry dispatches RTX 4090/L20-class GPUs to the sm89 build and RTX 5090 +to sm120. The L20 and RTX 4090 sm89 paths are validated; L20 is the default +because it has greater resource availability. Test sm120 on a real 5090 before +relying on that path. -### DPA-3.2-5M Multi-Head Model Preparation +For provenance, an older phonoLAMMPS 3.1.3 tag was +available from the Bohrium registry mirror at repo digest +`sha256:43a27ca4a7bba7f774bbd56104d205a6a80cd9d65928f249f6109e9ef37b8402`. +That image reports DeepMD-kit 3.1.3 and phonoLAMMPS 0.10.1. `dp show` failed +with `RuntimeError: Unknown model type: dpa4`, and LAMMPS aborted while +initializing `pair_style deepmd`. Therefore the tested old runtime cannot +execute the bundled DPA4 model. Do not submit the bundled model with that old +image or select another image without an explicit qualified profile. -The source DPA-3.2-5M `.pt` checkpoint is **multi-head** and cannot be used -directly as `interaction.model` in APEX/LAMMPS. Freeze a specific task head -first. +> **LAMMPS phonon and Grüneisen**: for GPU potentials, `apex submit` forces +> the DPA4 image above. It is validated on NVIDIA L20 and RTX 4090 and includes +> phonoLAMMPS. L20 is the default resource. +> CPU potentials keep the CPU image. Do not use the DPA4 image on a CPU +> machine; sequential CPU jobs stalled during container preparation. -**Freeze command:** -```bash -dp --pt freeze -c DPA-3.2-5M.pt -o DPA-3.2-5M-OMat24.pth --head OMat24 -``` - -- `-c`: Input multi-head `.pt` checkpoint -- `-o`: Output frozen model (`.pth` for PyTorch, `.pb` for TensorFlow) -- `--head`: Which task head to extract - -The **frozen** `.pth` or `.pb` file is what goes into `interaction.model`: -```json -{ - "interaction": { - "type": "deepmd", - "model": "DPA-3.2-5M-OMat24.pth", - "type_map": "auto" - } -} -``` +### Bundled DPA4 model (`models/`) -Useful heads include: -- `OMat24` — broad materials coverage; bundled default, including oxygen -- `Alloy_APEX` — APEX alloy/defect data; 53 metallic observed elements and no oxygen -- `Domains_Alloy` — general alloy energetics; 53 metallic observed elements -- `OC22` — oxide electrocatalyst structures - -Inspect the checkpoint before choosing: -```bash -dp --pt show DPA-3.2-5M.pt model-branch observed-type -``` - -> 💡 If you pass an unfrozen multi-head `.pt` file directly to LAMMPS, it will error with a message about missing head selection. - -### Bundled DPA model (`models/`) - -**Priority:** use the bundled **frozen** model first. The skill zip includes -the ready-to-run OMat24 `.pth`, not the multi-head source `.pt`. Details: -`models/README.md`. +The skill ships one DeePMD model and no alternate checkpoint downloader: | Path | Format | Use | |------|--------|-----| -| `models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth` | Frozen PyTorch (~23MB) | Default DPA-3.2 model for APEX | +| `models/DPA4-alloytongqi/model.pt` | Single-task DPA4 PyTorch checkpoint (30,403,297 bytes) | Source identity/provenance only; never the image-profile LAMMPS input | -Copy the bundled model into the task directory and set: +Safe static inspection confirms a valid PyTorch ZIP checkpoint with +`model_params.type=dpa4`, fitting type `dpa4_ener`, and an empty +`model_branch_alias` list, consistent with a default/single task. The +`alloytongqi` branch identity is user-provided provenance; no Git branch or +commit string is embedded in the file. A type map does not by itself prove +training-domain coverage. + +After publication, generate (do not hand-write) this image-resident contract: ```json { "interaction": { "type": "deepmd", - "model": "DPA-3.2-5M-OMat24.pth", + "deepmd_runtime": "dpa4_pt2", + "model_in_image": true, + "model": "/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.t4-sm75.pt2", + "runtime_model_sha256": "2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf", + "source_checkpoint": "/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt", + "source_checkpoint_sha256": "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad", "type_map": "auto" } } ``` -The source checkpoint is optional and downloaded only when another task head -is explicitly requested: - -```bash -python scripts/fetch_models.py --source-checkpoint -``` - -> Use `"type_map": "auto"` by default. APEX infers the local contiguous type -> mapping from the structure; do not copy atomic-number or model-internal indices. -> Multi-head `.pt` files **must** be frozen (`dp --pt freeze ... --head `) -> before use as `interaction.model`. +> Use `generate_config.py create --runtime-profile dpa4-alloytongqi-t4`. +> It is currently locked by placeholder image identity and +> `pre_snapshot_only` qualification. Pre-snapshot checks completed 12 LAMMPS +> runs, 6 CPU/GPU parity checks, and one phonoLAMMPS smoke on one rank/one GPU +> `c4_m15_1 * NVIDIA T4`; this is not exact-image acceptance. After a digest +> rerun, only that exact SKU is eligible; c8/c16 T4 and non-T4 GPUs remain +> unverified. The generated global config uses +> `/usr/local/bin/dpa4-lmp -in in.lammps` and +> `/usr/local/bin/dpa4-phonolammps {input_file} -c {poscar} --dim {dim} +> {primitive_axes}` exactly. V100/SM 7.0 and older devices and NVIDIA Linux +> drivers below 580.65.06 are prohibited by the CUDA 13 contract. Use +> `"type_map": "auto"` by default. APEX infers the local contiguous type +> mapping from the structure; do not copy atomic-number or model-internal +> indices. A `.pt` checkpoint is directly usable only when it is a compatible +> single-task model supported by the selected runtime; do not generalize this +> bundled model's handling to arbitrary multi-task training checkpoints. ### Relaxation cal_setting (LAMMPS) @@ -499,7 +498,8 @@ KGAMMA = True ### VASP global.json Settings (Bohrium / dflow) -> ⚠️ **Never** use bare `mpirun -n 16 vasp_std`. The Bohrium VASP image needs Intel oneAPI env + absolute binary path. +The Bohrium VASP image needs the Intel oneAPI environment and an absolute +VASP binary path. Use symbolic values when planning resources: ```json { @@ -508,8 +508,8 @@ KGAMMA = True "batch_type": "Bohrium", "context_type": "Bohrium", "vasp_image_name": "", - "vasp_run_command": "bash -c \"source /opt/intel/oneapi/setvars.sh && ulimit -s unlimited && mpirun -n 32 /opt/vasp.5.4.4/bin/vasp_std\"", - "scass_type": "c32_m128_cpu", + "vasp_run_command": "bash -c \"source && ulimit -s unlimited && mpirun -n \"", + "scass_type": "", "group_size": 1, "pool_size": 1 } @@ -536,11 +536,35 @@ KGAMMA = True |-------|-----| | `source /opt/intel/oneapi/setvars.sh` | Loads Intel MPI / MKL | | `ulimit -s unlimited` | Avoids stack overflow on large cells | -| Absolute `vasp_std` path | PATH `vasp_std` is unreliable | -| `mpirun -n ` | `` must match `scass_type` CPUs (`c32_*`→32, `c16_*`→16) | +| Absolute `vasp_std` or `vasp_gam` path | PATH lookup is unreliable | +| `mpirun -n ` | `RANKS` must equal the CPU count encoded by `scass_type` | Local/debug Shell jobs may use a simpler command only when the host already has VASP + MPI on PATH; for Bohrium use the template above. `generate_config.py` emits the run_command template for `--backend vasp` but leaves `vasp_image_name` unset. +### VASP executable and parallel guards + +The validator reads the run-command template, `scass_type`, representative +generated KPOINTS, and the INCAR parallel tags together. At runtime APEX reads +the actual `KPOINTS` in every task and selects the executable independently: + +- Gamma-centered `1x1x1` always runs with `vasp_gam`; every other grid runs + with `vasp_std`. The rule applies to relaxation and every property, not only + Gamma/GammaSurface. +- `KGAMMA=True` only selects centering; it does not prove that the grid has one + point. The generated task `KPOINTS` is authoritative. +- `vasp_run_command` may name either `vasp_std` or `vasp_gam`; APEX derives + both sibling command variants and chooses after task generation. +- Any representative task that resolves to `vasp_gam` requires `KPAR=1`. +- Let `R` be MPI ranks and `K` be `KPAR` (default `1`). `K` must divide `R`. +- If `NCORE=C` is set, `C` must be a positive integer and divide `R/K`. +- `NCORE` and `NPAR` must not both be set. Every explicit `NPAR` and `KPAR` + value must also be a positive integer. +- Missing `NCORE` is a warning, because the best value depends on the licensed + executable, CPU profile, and system size. + +The automatic selector uses `vasp_std` whenever the generated grid has more +than one k-point or is Monkhorst-Pack centered. + ### VASP POTCAR Handling APEX concatenates files at: @@ -611,9 +635,9 @@ Before submit: | Workload | Bohrium Machine | Rationale | |----------|----------------|-----------| -| LAMMPS + GPU potential (DeePMD/MACE/NEP) | `c8_m31_1 * NVIDIA T4` | GPU acceleration | +| LAMMPS + GPU potential (DeePMD/MACE/NEP) | `c16_m120_1 * NVIDIA L20` | Validated sm89 runtime; greater resource availability | | LAMMPS + CPU potential (EAM/MEAM/SNAP) | `c16_m32_cpu` | CPU sufficient | | ABACUS DFT (small cell <50 atoms) | `c16_m32_cpu` | 8 MPI ranks | | ABACUS DFT (large cell 50-200 atoms) | `c32_m128_cpu` | 16-32 MPI ranks | | VASP DFT | User choice | Depends on system size | -| Finite-T MD (long runs) | `c8_m31_1 * NVIDIA T4` | Long MD = GPU beneficial | +| Finite-T MD (long runs) | `c16_m120_1 * NVIDIA L20` | Long MD = GPU beneficial; validated sm89 runtime | diff --git a/apex/skills/apex-flow/reference/lammps_potentials.md b/apex/skills/apex-flow/reference/lammps_potentials.md index d19f6786..2b710e72 100644 --- a/apex/skills/apex-flow/reference/lammps_potentials.md +++ b/apex/skills/apex-flow/reference/lammps_potentials.md @@ -15,12 +15,13 @@ APEX supports 10 LAMMPS potential types through the `interaction.type` field. Ea ```json { "type": "deepmd", - "model": "frozen_model.pb", + "model": "model.pt", "type_map": "auto" } ``` -**Model files**: `.pb` (frozen graph) or `.pth` (PyTorch) +**Model files**: `.pb` (frozen graph), `.pth` (PyTorch), or a compatible +single-task `.pt` checkpoint supported by the selected runtime **Notes**: - Most widely used MLIP in APEX workflows - GPU strongly recommended for large systems @@ -218,9 +219,10 @@ manual map. | Potential Type | Recommended Image | GPU? | |---------------|-------------------|------| -| deepmd | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | Yes | -| mace | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | Yes | -| nep | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | Yes | +| deepmd | `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2` | Yes (NVIDIA L20 default; RTX 4090 compatible) | +| mace | `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2` | Yes (NVIDIA L20 default; RTX 4090 compatible) | +| nep | `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2` | Yes (NVIDIA L20 default; RTX 4090 compatible) | +| DPA4 alloytongqi candidate | Locked `dpa4-alloytongqi-t4` profile; no published image yet | One T4 only after exact-image qualification | | gap | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | No (CPU) | | snap | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | No | | rann | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | No | @@ -229,7 +231,23 @@ manual map. | meam | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | No | | meam_spline | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | No | -All potential types are supported by the unified APEX image. The APEX 1.3.0 image ships LAMMPS compiled with DeePMD, MACE, NEP, and standard LAMMPS potentials. +GPU and CPU potentials use separate images. The DPA4 image is not a CPU +fallback: sequential `c8_m32_cpu` validation stalled during container +preparation. Standard CPU potentials use the APEX 1.3.0 image. +The table records APEX defaults, not compatibility with every model format. +The old tag's +Bohrium registry mirror at repo digest +`sha256:43a27ca4a7bba7f774bbd56104d205a6a80cd9d65928f249f6109e9ef37b8402` +rejects the bundled DPA4 `model.pt` with `Unknown model type: dpa4`. Do not +submit that model under the default image or silently choose another image. +The DPA4 profile fails closed while image placeholders or +`pre_snapshot_only` status remain. The only pre-snapshot candidate hardware is +one rank/one GPU on `c4_m15_1 * NVIDIA T4`; c8/c16 T4, non-T4 GPUs, CPU, +multi-rank, multi-GPU, and cross-architecture PT2 reuse are not recommended. +V100/SM 7.0 and older devices and NVIDIA Linux drivers below 580.65.06 are +prohibited by the CUDA 13 runtime. A published profile must use absolute +`/usr/local/bin/dpa4-lmp` and `/usr/local/bin/dpa4-phonolammps` wrappers. +Other potential types also require their own compatible model files. --- @@ -237,7 +255,7 @@ All potential types are supported by the unified APEX image. The APEX 1.3.0 imag | Error | Cause | Fix | |-------|-------|-----| -| `pair_style not found` | LAMMPS not compiled with required package | Use APEX official image | +| `pair_style not found` | LAMMPS not compiled with required package | Use the potential-specific configured image | | `type_map` inference failure | Structure file is missing or matched structures use incompatible element sets | Fix `structures` paths or provide one verified manual map | | `model file not found` | File not in job directory | Ensure model file is copied to submission dir | | `GPU not available` | Running GPU potential on CPU node | Switch to GPU machine type | diff --git a/apex/skills/apex-flow/reference/properties.md b/apex/skills/apex-flow/reference/properties.md index 29d9ed05..d8826a1c 100644 --- a/apex/skills/apex-flow/reference/properties.md +++ b/apex/skills/apex-flow/reference/properties.md @@ -263,35 +263,57 @@ These parameters appear in most property configurations: |-----------|-----------|------|---------|-------------| | `plane_miller` | **REQUIRED** | list[int] | `[1,1,1]` | Slip plane Miller indices | | `slip_direction` | **REQUIRED** | list[int] | `[-1,1,0]` | Slip direction (**must lie ON the plane**) | +| `parent_lattice` | optional | str/null | `null` | Parent lattice for RSS/SQS. Gamma line automatically infers the integer parent-supercell mapping and treats the Miller/direction inputs as parent indices without symmetrizing the supplied geometry | | `slip_length` | optional | float | `null` | Total slip distance (Å); auto if null | | `plane_shift` | optional | int/float | `0` | Shift of slip plane position | -| `supercell_size` | optional | list[int] | `[1,1,5]` | Supercell for slab | -| `vacuum_size` | optional | float | `0` | Vacuum above slab (Å) | +| `supercell_size` | optional | list[int] | `[1,1,5]` | In-plane replication and target Miller-plane spacings | +| `min_slab_height` | optional | float/null | `null` | Auto-add oriented-cell repeats until this material thickness (Å) is reached | +| `max_atoms` | optional | int/null | `null` | Stop if the generated slab exceeds this atom count | +| `min_distance` | optional | float | `0.2` | Stop if a periodic atom-pair distance is below this value (Å) | +| `vacuum_size` | optional | float | `20` | Vacuum above slab (Å) | +| `require_orthogonal_cell` | optional | bool | `false` | Fail unless the generated slab is an orthogonal Cartesian-z zero-tilt cell; never changes periodic boundaries | | `n_steps` | optional | int | `10` | Number of slip increments | +| `displacement_points` | optional | list[float]/null | `null` | Explicit unique fractions in `[0,1]`; must include `0` and, when set, replaces the uniform `n_steps` grid | | `add_fix` | optional | list[str] | `["true","true","false"]` | Selective dynamics per axis | **cal_setting defaults**: `relax_pos=true`, `relax_shape=false`, `relax_vol=false` -**Complete working default**: +**Backward-compatible default** (the safety limits remain unset unless the +user supplies them): ```json { "type": "gamma", "plane_miller": [1, 1, 1], "slip_direction": [-1, 1, 0], "supercell_size": [1, 1, 5], + "vacuum_size": 20, "n_steps": 10 } ``` -**Output**: Stacking fault energy (mJ/m²) vs displacement fraction. +For endpoint-only work, for example, set +`"displacement_points": [0.0, 0.5]`. APEX sorts the fractions and fails if +they are duplicated, outside `[0,1]`, non-finite, or omit the required zero +reference. The generated task count is exactly the number of explicit points. + +**Output**: Stacking fault energy (J/m²) vs displacement fraction. Multiply by +1000 only when a plot or table explicitly uses mJ/m². ⚠️ **CRITICAL CONSTRAINT**: The `slip_direction` **must be a vector ON the slip plane** (dot product with `plane_miller` must equal zero). -**Canonical slip systems (FCC / BCC / HCP):** use the predefined table in the -APEX repository **README §4.10 Gamma line/surface** — do not invent planes or -directions outside that table without explicit user confirmation. Nested +**Physically recommended slip systems (FCC / BCC / HCP):** use the table in +the APEX repository **README §4.10 Gamma line/surface**. Systems outside that +registry are allowed, but APEX warns and falls back to checking only that the +direction lies on the plane; inspect the generated slab manually. Nested `fcc` / `bcc` / `hcp` blocks in `param.json` override top-level `plane_miller` / `slip_direction` for the matching lattice type (see README). +When chemical disorder or a large supercell makes symmetry detection return +`other`, set `parent_lattice` explicitly. For Gamma line APEX automatically +maps parent indices into the actual relaxed supercell, uses the true reciprocal +normal and elementary parent Burgers translation, freezes the fault split before +vacuum is added, and writes `gamma_geometry.json`. Mapping, layer-gap split, +minimum distance, and parent-translation topology are fail-closed validations; +RSS relaxed-coordinate and chemical `u=1` mismatches are diagnostic metadata. Quick primary picks when the user has not specified a system (still confirm): @@ -315,38 +337,111 @@ Quick primary picks when the user has not specified a system (still confirm): |-----------|-----------|------|---------|-------------| | `plane_miller` | **REQUIRED** | list[int] | `[1,1,1]` | Slip plane | | `slip_direction` | **REQUIRED** | list[int] | `[-1,1,0]` | x-direction of 2D grid (**must lie ON the plane**) | +| `parent_lattice` | optional | str/null | `null` | Explicit `bcc`, `fcc`, or `hcp` parent hint for RSS/disordered supercells; infers the integer parent-supercell mapping and treats plane/direction as parent indices without changing the supplied geometry | | `slip_length` | optional | float | `null` | Slip distance in x | | `slip_length_y` | optional | float | `null` | Slip distance in y | | `plane_shift` | optional | int/float | `0` | Slip plane shift | -| `supercell_size` | optional | list[int] | `[1,1,5]` | Supercell | -| `vacuum_size` | optional | float | `0` | Vacuum (Å) | +| `supercell_size` | optional | list[int] | `[1,1,5]` | In-plane replication and target Miller-plane spacings | +| `min_slab_height` | optional | float/null | `null` | Auto-add oriented-cell repeats until this material thickness (Å) is reached | +| `max_atoms` | optional | int/null | `null` | Stop if the generated slab exceeds this atom count | +| `min_distance` | optional | float | `0.2` | Stop if a periodic atom-pair distance is below this value (Å) | +| `vacuum_size` | optional | float | `20` | Vacuum (Å); set `0` explicitly only for a bulk-like periodic fault model | +| `require_orthogonal_cell` | optional | bool | `false` | Fail unless the generated slab is an orthogonal Cartesian-z zero-tilt cell; never Gram-Schmidts the lattice | | `closed_loop` | optional | bool | `false` | Derive a periodic, possibly oblique in-plane basis | -| `n_steps_x` | optional | int | `10` | Grid points in x | -| `n_steps_y` | optional | int | `n_steps_x` | Grid points in y | +| `n_steps_x` | optional | int | `10` | Grid increments in x; produces `n_steps_x+1` fractions | +| `n_steps_y` | optional | int | `n_steps_x` | Grid increments in y; produces `n_steps_y+1` fractions | | `add_fix` | optional | list[str] | `["true","true","false"]` | Selective dynamics | **cal_setting defaults**: `relax_pos=true`, `relax_shape=false`, `relax_vol=false` -**Complete working default**: +**Backward-compatible core default** (the safety limits remain unset unless +the user supplies them): ```json { "type": "gamma_surface", "plane_miller": [1, 1, 1], "slip_direction": [-1, 1, 0], "supercell_size": [1, 1, 5], + "vacuum_size": 20, "closed_loop": false, "n_steps_x": 10, "n_steps_y": 10 } ``` +The shipped generator and GUI profile templates set `closed_loop=true` as the +recommended 2D default. Omitting the field still preserves the legacy core +behavior above. + **Output**: 2D grid of SFE values (J/m²), (n_steps_x+1) × (n_steps_y+1) points. With `closed_loop=true`, `slip_length` and `slip_length_y` must be omitted. APEX records the periodic basis vectors and the true Cartesian displacement of every grid point; use this mode for oblique or disordered supercells. - -⚠️ **Same constraint as gamma**: `slip_direction` must have zero dot product with -`plane_miller`. Canonical FCC/BCC/HCP systems: **README §4.10** (same table as `gamma`). +Both properties also write `slab_generation.json`. The third +`supercell_size` value is passed to Pymatgen as a plane count +(`in_unit_planes=true`), preventing an intended two-layer slab from being +promoted by floating-point round-off. +Both also write `gamma_geometry.json`, use the same frozen material-internal +fault split, and divide by its recorded `interface_count`: one with vacuum, +two for a fully periodic zero-vacuum cell. Task areas must match the reference. +`orthogonalize_cell=true` is accepted as a backward-compatible alias for the +strict `require_orthogonal_cell` gate; it never modifies the lattice. + +### Generator options and pre-submit validation + +`generate_config.py create` keeps the legacy `--properties` interface and +defaults unchanged. The following optional flags only apply when `gamma` +and/or `gamma_surface` is requested: + +| CLI option | JSON field | Applies to | +|---|---|---| +| `--gamma-parent-lattice {bcc,fcc,hcp}` | `parent_lattice` | both | +| `--gamma-plane-miller <...>` | `plane_miller` | both | +| `--gamma-slip-direction <...>` | `slip_direction` | both | +| `--gamma-supercell-size ` | `supercell_size` | both | +| `--gamma-vacuum-size ` | `vacuum_size` | both | +| `--gamma-require-orthogonal-cell` | `require_orthogonal_cell=true` | both | +| `--gamma-min-slab-height ` | `min_slab_height` | both | +| `--gamma-max-atoms ` | `max_atoms` | both | +| `--gamma-min-distance ` | `min_distance` | both | +| `--gamma-n-steps ` | `n_steps` | `gamma` | +| `--gamma-displacement-points ` | `displacement_points` | `gamma` | +| `--gamma-n-steps-x ` | `n_steps_x` | `gamma_surface` | +| `--gamma-n-steps-y ` | `n_steps_y` | `gamma_surface` | +| `--gamma-closed-loop` | `closed_loop=true` | `gamma_surface` | + +The generator prints the final Gamma JSON and expected task count. The input +validator then uses the APEX Gamma core in a temporary directory to generate a +representative slab for every local input structure. It reports parent/final +atom count, material thickness, oriented-cell repeats, effective plane count, +minimum periodic pair distance, expected task count, and generated VASP +KPOINTS when applicable. + +Explicit safety limits are enforced before submission. Missing +`min_slab_height` or `max_atoms` produces a compatibility warning rather than +changing legacy behavior. This representative check does not replace the full +displacement overlap check performed by `apex preview`. + +⚠️ **Same constraint and fallback as gamma**: `slip_direction` must have zero +dot product with `plane_miller`. Physically recommended FCC/BCC/HCP systems are +listed in **README §4.10**. Other systems warn and use geometry-only checking. + +⚠️ **Mandatory pre-submit check**: run `apex preview ` before submitting +any `gamma_surface` job. Preview builds the displacement slabs and, if any atom +pair is closer than `0.2` Å, prints to stderr: + +`Generated Gamma surface contains overlapping atoms.` + +If that warning appears, do not submit until geometry/parameters are fixed. +Agents must check this stderr message only — **do not open or read the GIF** +(the GIF is optional for humans). + +The default `--gif-view auto` writes slip-plane and parent-`bc` projections for +both Gamma lines and Gamma surfaces. Humans can select a single projection with +`--gif-view default`, `slip-plane`, or `parent-bc`, or request the pair +explicitly with `--gif-view both`. Projected unit-cell boundaries are retained +so the vacuum region is not cropped from side-facing projections. These views +are diagnostic renderings and do not replace the fail-closed geometry checks. **Notes**: - The y-direction is automatically computed as `plane_miller × slip_direction`. @@ -420,17 +515,17 @@ overrides — detect the lattice, pick from this table, and confirm with the use | Key | Required? | Type | Default | Description | |-----|-----------|------|---------|-------------| -| `temperature` | **REQUIRED** | list[float] | `[200,400,600,800]` | Target temperatures (K) | -| `equi_step` | optional | int | `80000` | Equilibration steps | -| `ave_step` | optional | int | `40000` | Averaging steps | +| `temperature` | optional | list[float] | `[200,400,600,800]` LAMMPS; `[300,500,700,900,1100,1300,1500]` VASP | Target temperatures (K) | +| `equi_step` | optional | int | `80000` LAMMPS; `5000` VASP | Equilibration steps | +| `ave_step` | optional | int | `40000` LAMMPS; `10000` VASP | Production/statistics steps | | `timestep` | optional | float | `0.001` | Timestep (ps) | | `tdamp` | optional | float | `0.1` | Thermostat damping | | `pdamp` | optional | float | `1.0` | Barostat damping | | `N_every` | optional | int | `100` | Sample frequency | | `N_repeat` | optional | int | `10` | Repeat count | | `N_freq` | optional | int | `2000` | Output frequency | -| `timestep_fs` | DFT optional | float | `1.0` | VASP/ABACUS timestep (fs) | -| `pressure_kbar` | DFT optional | float | `0.0` | VASP/ABACUS target pressure (kbar) | +| `timestep_fs` | VASP optional | float | `1.0` | VASP timestep (fs) | +| `pressure_kbar` | VASP optional | float | `0.0` | VASP target pressure (kbar) | **Complete working default**: ```json @@ -443,9 +538,9 @@ overrides — detect the lattice, pick from this table, and confirm with the use } ``` -**Output**: Lattice parameter a (and c/a for non-cubic) vs temperature. +**Output**: The legacy temperature-to-`[a,b,c,T]` mapping plus rich `_metadata` statistics (mean, standard deviation, block standard error, and sample count) for cell tensors, lengths, angles, and volume. -VASP uses Langevin–Parrinello–Rahman NpT; ABACUS uses Nose–Hoover-style NpT. These integrations share the thermodynamic target but not thermostat/barostat parameters. +Supported backends are LAMMPS and VASP; ABACUS is rejected. VASP uses Langevin–Parrinello–Rahman NpT (`MDALGO=3`, `ISIF=3`) and requires a VASP binary compiled with `-Dtbdyn`. **Notes**: - Each temperature is a separate NPT MD run. @@ -548,6 +643,7 @@ VASP uses Langevin–Parrinello–Rahman NpT; ABACUS uses Nose–Hoover-style Np |-----------|-----------|------|---------|-------------| | `supercell_size` | optional | list[int] | `[3,3,3]` | MD supercell | | `supercell_length` | optional | float | `null` | Auto-size from target edge length | +| `protocol` | optional | str | `"ramp_cool"` | Legacy heat/cool schedule, or VASP-only `"coexistence"` | **cal_setting keys (temperature cycle)**: @@ -560,12 +656,13 @@ VASP uses Langevin–Parrinello–Rahman NpT; ABACUS uses Nose–Hoover-style Np | `cool_rate` | optional | float | = ramp_rate | Cooling rate (K/ps) | | `equi_step` | optional | int | `20000` | Initial equilibration steps | | `hold_step` | optional | int | `20000` | Hold at target_temp steps | +| `production_step` | optional | int | `10000` | Fixed-T production steps for `protocol="coexistence"` | | `timestep` | optional | float | `0.001` | Timestep (ps) | | `thermostat` | optional | str | `"nose_hoover"` | Thermostat type | | `ensemble` | optional | str | `"npt"` | Ensemble type | | `velocity_seed` | optional | int | `123457` | Random seed | -| `timestep_fs` | DFT optional | float | `1.0` | VASP/ABACUS timestep (fs) | -| `pressure_kbar` | DFT optional | float | `0.0` | VASP/ABACUS target pressure (kbar) | +| `timestep_fs` | VASP optional | float | `1.0` | VASP timestep (fs) | +| `pressure_kbar` | VASP optional | float | `0.0` | VASP target pressure (kbar) | **cal_setting keys (analysis)**: @@ -590,9 +687,9 @@ VASP uses Langevin–Parrinello–Rahman NpT; ABACUS uses Nose–Hoover-style Np } ``` -**Output**: RDF at each stage, MSD, volume-temperature curves, final quenched structure. +**Output**: RDF and MSD by stage, plus volume, pressure, temperature, potential energy, and total energy series. Ramp/cool also produces a final quenched structure. -VASP and ABACUS run the same temperature schedule with their native NpT integrators and post-process compact trajectories into the same result schema. +Supported backends are LAMMPS and VASP; ABACUS is rejected. `protocol="ramp_cool"` retains the existing heating/cooling semantics. VASP-only `protocol="coexistence"` holds `target_temp` for an equilibration stage and then a production stage, both at fixed target temperature. VASP uses `MDALGO=3` and requires `-Dtbdyn`. **Notes**: - [3,3,3] supercell recommended for statistical sampling (108+ atoms). @@ -601,6 +698,91 @@ VASP and ABACUS run the same temperature schedule with their native NpT integrat --- +## 15. Two-Phase Coexistence Melting Point (LAMMPS only) + +**type**: `"melting_point"` **[LAMMPS only]** + +This property constructs a solid/liquid interface, premelts the upper region +with the lower crystal pinned, conditions the liquid at each target +temperature, and releases the complete cell under NPT dynamics. The bracket +uses the sign and 95% interval of the q6-derived interface velocity. + +```json +{ + "type": "melting_point", + "method": "two_phase", + "supercell_size": [1, 1, 2], + "cal_setting": { + "temperature": [1600, 1650, 1700], + "premelt_temperature": 4500, + "premelt_steps": 5000, + "conditioning_steps": 5000, + "production_steps": 100000, + "timestep": 0.001, + "tdamp": 0.1, + "pdamp": 1.0, + "pressure": 0.0, + "barostat": "iso", + "interface_axis": "z", + "liquid_fraction": 0.5, + "dump_step": 100, + "thermo_step": 100, + "restart_interval": 10000, + "q6_cutoff": 3.5, + "q6_neighbors": 12, + "replicas": 3, + "velocity_seeds": { + "premelt": 324159, + "condition": 271828, + "release": 161803 + } + } +} +``` + +Optional continuation inputs are temperature-indexed: + +```json +"cal_setting": { + "temperature": [1600, 1650, 1700], + "restart_files": [ + "restart.1600", + "restart.1650", + "restart.1700" + ] +} +``` + +`restart_files` must contain exactly one existing file per temperature. The +matching file is copied into every replica task for that temperature and +forwarded as `restart.coexistence.start`. This is a transport contract only: +the generated LAMMPS input is not automatically rewritten to `read_restart`. +`finite_t_latt` does not accept `restart_files` and never forwards +`restart.coexistence.start`. + +The generator exposes `--melting-temperatures`, `--melting-replicas`, and +`--melting-restart-files`; restart inputs are staged into the generated job +directory before `param.json` is written. + +Each temperature/replica pair is one GPU/CPU LAMMPS task. Provide independent +alloy chemical realizations as separate `structures`; the property never +silently randomizes atom types. Before paid submission, report the relaxed +base-cell atom count, `supercell_size`, final atom count, temperatures, +replicas, total task count, runtime, and accelerator resources. + +Aggregation is fail-closed against the configured matrix. Every requested +temperature must have exactly the configured number of distinct replicas; a +missing or duplicate replica makes that temperature's consensus +`inconclusive`, so it cannot establish a melting bracket. + +**Outputs**: `result.json`, `result.out`, `melting_point_tidy.csv`, solid +fraction versus time, interface velocity versus temperature, q6 interface +snapshots, and the raw `dump.melting`/`log.lammps` evidence in every task. +Each task also retrieves alternating `restart.melting.1/2` checkpoints and the +normal-completion `restart.melting.final`; `restart_interval` is in timesteps. + +--- + ## RSS Structure Generation Use `apex rss ` to generate input structures. @@ -631,6 +813,7 @@ See `reference/rss_workflow.md` for full details. | `finite_t_elastic` | cal_setting.temperature | [300] | LAMMPS-only | | `gruneisen` | volume_strains, temperatures | [-0.02..0.02], [100..500] | ⚠️ Missing temperatures → KeyError | | `annealing` | (none critical) | target_temp=300 | DFT MD is expensive | +| `melting_point` | temperature, supercell_size | three-point bracket, 100 ps release | LAMMPS-only; do not classify from energy alone | --- @@ -648,7 +831,8 @@ Before submitting any APEX property calculation, verify: - Finite-T properties: ≥ [3,3,3] 4. ✅ **LAMMPS-only properties** not sent to DFT backend: - `finite_t_elastic` + - `melting_point` 5. ✅ **Physical reasonableness**: - - Temperatures below melting point + - Thermal-expansion temperatures below melting; melting-point temperatures bracket both sides - Volume strains ≤ ±5% (avoid unphysical compression) - Slab thickness large enough for bulk-like interior diff --git a/apex/skills/apex-flow/reference/rss_workflow.md b/apex/skills/apex-flow/reference/rss_workflow.md index 14352d6e..eca280a7 100644 --- a/apex/skills/apex-flow/reference/rss_workflow.md +++ b/apex/skills/apex-flow/reference/rss_workflow.md @@ -5,7 +5,7 @@ APEX's RSS workflow generates occupationally disordered structures for random solid solutions, solid solutions, high-entropy alloys (HEA), high-entropy oxides/ceramics (HEO), and other high-entropy materials. RSS generates -structures locally; it is not one of the 14 APEX properties and does not require +structures locally; it is not one of the 15 APEX properties and does not require choosing a calculator backend unless the user also asks to calculate properties. When the user's request contains any of the material classes above, explicitly diff --git a/apex/skills/apex-flow/reference/submission.md b/apex/skills/apex-flow/reference/submission.md index 36cf3fe8..bce42ba5 100644 --- a/apex/skills/apex-flow/reference/submission.md +++ b/apex/skills/apex-flow/reference/submission.md @@ -27,7 +27,8 @@ APEX uses a **two-layer submission architecture**: ▼ ┌─────────────────────────────────────────────────────────────────┐ │ Inner containers (managed by dflow on Bohrium) │ -│ - LAMMPS image: deepmd-kit-phonolammps:3.1.3 │ +│ - GPU LAMMPS: dpa4-phonolammps:0.0.2 │ +│ - CPU LAMMPS: apex-flow:1.3.0.post │ │ - ABACUS image: registry.dp.tech/dptech/abacus:3.2.3 │ │ - VASP image: (user-provided) │ │ - Machine type per task: scass_type in global.json │ @@ -118,17 +119,43 @@ dflow validates workflow names against RFC 1123 subdomain regex. Names like `"Cu | Role | Image | Notes | |------|-------|-------| | **Outer job (submission client)** | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | Lightweight; just runs `apex submit` | -| **LAMMPS calculator** | `registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3` | Default; includes phonoLAMMPS | +| **LAMMPS calculator (GPU potentials)** | `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2` | NVIDIA L20 default; RTX 4090 compatible; includes phonoLAMMPS | +| **LAMMPS calculator (CPU potentials)** | `registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post` | EAM/MEAM/SNAP/GAP/RANN CPU backend | | **ABACUS calculator** | (same APEX image has ABACUS) | Or user-specified | | **VASP calculator** | User must provide after confirming license | Commercial; **never invent a default image** | -> ⚠️ **Do NOT combine `deepmd-kit:3.1.1` with any NVIDIA T4 machine**. It also has a known segfault bug when handling triclinic cells (non-orthogonal boxes), including on CPU. Use `3.1.3` or later. - -> LAMMPS phonon and Grüneisen tasks are forced to use `registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3`, which includes the required phonoLAMMPS executable. +> Legacy `deepmd-kit:3.1.1` remains blocked with NVIDIA T4 and for triclinic +> cells. + +> LAMMPS phonon and Grüneisen tasks using GPU potentials are forced to use +> `registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2`. +> CPU potentials retain the CPU image. Do not use 0.0.2 with a `*_cpu` +> machine: sequential CPU validation stalled before the container command +> started. Do not add the legacy `plugin load libdeepmd_lmp.so` command to +> 0.0.2. +> An older phonoLAMMPS 3.1.3 image at Bohrium registry mirror digest +> `sha256:43a27ca4a7bba7f774bbd56104d205a6a80cd9d65928f249f6109e9ef37b8402` +> rejects the bundled DPA4 model with `Unknown model type: dpa4`; do not submit +> that model under the old image or silently route it to another image. + +The DPA4 exception is explicit and currently locked. Inspect it with +`validate_apex_combo.py list-combos --runtime-profile +dpa4-alloytongqi-t4`. Do not use `recommend` or `generate_config.py create +--runtime-profile ...` until an immutable image ref/digest has passed the +packaged benchmark and the profile says `post_snapshot_passed`. +Pre-snapshot evidence exists only for one rank/one GPU on +`c4_m15_1 * NVIDIA T4`; it is not a published-image recommendation. Generic +APEX and legacy phonon/Grüneisen continue to use the old image above. +When published, `generate_config.py` writes both +`lammps_run_command=/usr/local/bin/dpa4-lmp -in in.lammps` and the complete +`phonolammps_run_command=/usr/local/bin/dpa4-phonolammps {input_file} -c +{poscar} --dim {dim} {primitive_axes}` template. Do not replace them with bare +binary names. These fields are DPA4-only; legacy configuration remains +unchanged. | Backend | scass_type (inner containers) | Notes | |---------|-------------------------------|-------| -| LAMMPS (DeePMD/MACE/NEP) | `c8_m31_1 * NVIDIA T4` | GPU beneficial | +| LAMMPS (DeePMD/MACE/NEP) | `c16_m120_1 * NVIDIA L20` | Validated sm89 default; RTX 4090 remains compatible | | LAMMPS (EAM/MEAM/SNAP) | `c16_m32_cpu` | CPU sufficient | | ABACUS | `c16_m32_cpu` | CPU | | VASP | `c32_m128_cpu` (default) | Align `mpirun -n ` with CPU count | @@ -188,7 +215,7 @@ The following is a type-annotated shape, not valid JSON: "project_id": }, "apex_image_name": "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post", - "lammps_image_name": "registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3", + "lammps_image_name": "registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2", "lammps_run_command": "lmp -in in.lammps", "scass_type": "c16_m32_cpu", "group_size": 1, @@ -209,7 +236,7 @@ Do not upload or submit unless validation reports `Validation PASSED` and both project ID lines report `type=int`. Submit the newly validated directory as a new outer Bohrium job; retrying an old outer job reuses its old input snapshot. -> For GPU potentials (DeePMD, MACE, NEP), change `scass_type` to `"c8_m31_1 * NVIDIA T4"`. +> For GPU potentials (DeePMD, MACE, NEP), change `scass_type` to `"c16_m120_1 * NVIDIA L20"` by default; RTX 4090 remains compatible. > Before submitting, run `scripts/validate_apex_combo.py check` on the chosen image × scass_type. ## Agent-Managed Submission Workflow (Complete Lifecycle) diff --git a/apex/skills/apex-flow/reference/workflow-control.md b/apex/skills/apex-flow/reference/workflow-control.md index 22258999..687e738d 100644 --- a/apex/skills/apex-flow/reference/workflow-control.md +++ b/apex/skills/apex-flow/reference/workflow-control.md @@ -55,7 +55,7 @@ as successful. Verify: property_success + property_failed + property_unfinished = property_total ``` -For one structure requesting all properties, `property_total` is 14. For +For one structure requesting all properties, `property_total` is 15. For multiple structures, these are property-task counts, normally `number of structures × number of requested properties`, less explicitly skipped tasks. @@ -121,7 +121,7 @@ Example for the single-structure SrTiO3 all-properties run: 作业名称:apex-srtio3-all-14props 材料:SrTiO₃(立方钙钛矿,a=3.9316 Å,5 atoms;已核对提交的 POSCAR) 当前阶段:relax 0, props 1(剩余弛豫 0;剩余性质任务 1) -性质进度:成功 <以 propertycal-* 的 Succeeded 数为准>,失败 <以 Failed/Error 数为准>,未完成 1,总计 14 +性质进度:成功 <以 propertycal-* 的 Succeeded 数为准>,失败 <以 Failed/Error 数为准>,未完成 1,总计 15 Argo workflow:https://workflows.deepmodeling.com/workflows/argo/ Workflow ID: Workflow UID: diff --git a/apex/skills/apex-flow/scripts/dpa4_profile.py b/apex/skills/apex-flow/scripts/dpa4_profile.py new file mode 100644 index 00000000..8bfba6b9 --- /dev/null +++ b/apex/skills/apex-flow/scripts/dpa4_profile.py @@ -0,0 +1,188 @@ +"""Load and validate the skill's immutable DPA4 alloytongqi T4 profile.""" + +from __future__ import annotations + +import json +import re +from pathlib import Path + + +DPA4_PROFILE_ID = "dpa4-alloytongqi-t4" +DPA4_PROFILE_PATH = ( + Path(__file__).resolve().parent.parent + / "data" + / "dpa4_alloytongqi_t4_profile.json" +) +DPA4_IMAGE_REF_PLACEHOLDER = "__DPA4_IMAGE_REF__" +DPA4_IMAGE_DIGEST_PLACEHOLDER = "__DPA4_IMAGE_DIGEST__" +DPA4_LAMMPS_RUN_COMMAND = "/usr/local/bin/dpa4-lmp -in in.lammps" +DPA4_PHONOLAMMPS_RUN_COMMAND = ( + "/usr/local/bin/dpa4-phonolammps {input_file} -c {poscar} " + "--dim {dim} {primitive_axes}" +) +DPA4_RUNTIME_MODEL_PATH = ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/" + "alloytongqi.t4-sm75.pt2" +) +DPA4_RUNTIME_MODEL_SHA256 = ( + "2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf" +) +DPA4_SOURCE_CHECKPOINT_PATH = ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt" +) +DPA4_SOURCE_CHECKPOINT_SHA256 = ( + "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad" +) +_IMAGE_DIGEST_RE = re.compile(r"sha256:[0-9a-f]{64}") + + +def load_dpa4_profile( + *, + require_published: bool = True, + path: Path | str | None = None, +) -> dict: + """Return the audited profile; reject missing or unpublished identity.""" + profile_path = Path(path) if path is not None else DPA4_PROFILE_PATH + try: + profile = json.loads(profile_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + raise RuntimeError(f"Cannot load DPA4 runtime profile: {profile_path}") from exc + + return validate_dpa4_profile( + profile, + require_published=require_published, + ) + + +def validate_dpa4_profile( + profile: dict, + *, + require_published: bool = True, +) -> dict: + """Validate an in-memory profile and return a shallow annotated copy.""" + if not isinstance(profile, dict): + raise RuntimeError("DPA4 runtime profile must contain a JSON object") + profile = dict(profile) + + if profile.get("profile_id") != DPA4_PROFILE_ID: + raise RuntimeError( + f"DPA4 runtime profile_id must be {DPA4_PROFILE_ID!r}" + ) + image = profile.get("image") + if not isinstance(image, dict): + raise RuntimeError("DPA4 runtime profile must contain an image object") + image_ref = str(image.get("ref", "")).strip() + image_digest = str(image.get("digest", "")).strip().lower() + identity_finalized = ( + bool(image_ref) + and image_ref != DPA4_IMAGE_REF_PLACEHOLDER + and "@" not in image_ref + and image_digest != DPA4_IMAGE_DIGEST_PLACEHOLDER.lower() + and _IMAGE_DIGEST_RE.fullmatch(image_digest) is not None + ) + qualification_status = str(profile.get("qualification_status", "")).strip() + qualified = qualification_status == "post_snapshot_passed" + published = identity_finalized and qualified + profile["identity_finalized"] = identity_finalized + profile["qualified"] = qualified + profile["published"] = published + if require_published and not published: + missing = [] + if not identity_finalized: + missing.append( + f"replace {DPA4_IMAGE_REF_PLACEHOLDER} and " + f"{DPA4_IMAGE_DIGEST_PLACEHOLDER}" + ) + if not qualified: + missing.append( + "set qualification_status='post_snapshot_passed' only after " + "the exact ref@digest passes the packaged benchmark" + ) + raise RuntimeError( + "DPA4 production profile is not published: " + + "; ".join(missing) + + ". Update data/dpa4_alloytongqi_t4_profile.json before " + "generating, recommending, or submitting this profile" + ) + + calculator = profile.get("calculator") + runtime = profile.get("runtime") + compatibility = profile.get("machine_compatibility") + if not all(isinstance(item, dict) for item in ( + calculator, runtime, compatibility + )): + raise RuntimeError( + "DPA4 runtime profile is missing calculator/runtime/machine fields" + ) + recommended = compatibility.get("recommended") + if not isinstance(recommended, list) or len(recommended) != 1: + raise RuntimeError("DPA4 profile must define exactly one recommended combo") + combo = recommended[0] + if ( + not isinstance(combo, dict) + or combo.get("scass_type") != "c4_m15_1 * NVIDIA T4" + or combo.get("mpi_ranks") != 1 + or combo.get("gpu_count") != 1 + ): + raise RuntimeError( + "DPA4 production profile must remain one rank on one " + "c4_m15_1 NVIDIA T4" + ) + if calculator.get("backend") != "lammps" or calculator.get("potential") != "deepmd": + raise RuntimeError("DPA4 profile must use the LAMMPS + DeePMD calculator") + if calculator.get("run_command") != DPA4_LAMMPS_RUN_COMMAND: + raise RuntimeError("DPA4 profile must use the audited dpa4-lmp wrapper") + if ( + calculator.get("phonolammps_command") + != DPA4_PHONOLAMMPS_RUN_COMMAND + ): + raise RuntimeError( + "DPA4 profile must use the audited dpa4-phonolammps wrapper template" + ) + expected_runtime = { + "kind": "dpa4_pt2", + "model_in_image": True, + "model_path": DPA4_RUNTIME_MODEL_PATH, + "model_sha256": DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint_path": DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": DPA4_SOURCE_CHECKPOINT_SHA256, + "type_map": "auto", + } + for key, expected in expected_runtime.items(): + if runtime.get(key) != expected: + raise RuntimeError( + f"DPA4 profile runtime.{key} must equal {expected!r}" + ) + return profile + + +def dpa4_image_name(profile: dict) -> str: + """Return ``ref@sha256:digest`` from a published validated profile.""" + if not profile.get("published"): + raise RuntimeError("DPA4 runtime profile image is unpublished") + image = profile["image"] + return f"{image['ref']}@{str(image['digest']).lower()}" + + +def dpa4_interaction(profile: dict) -> dict: + """Build the exact image-resident interaction contract.""" + if not profile.get("published"): + raise RuntimeError("DPA4 runtime profile is unpublished") + runtime = profile["runtime"] + return { + "type": "deepmd", + "deepmd_runtime": runtime["kind"], + "model_in_image": runtime["model_in_image"], + "model": runtime["model_path"], + "runtime_model_sha256": runtime["model_sha256"], + "source_checkpoint": runtime["source_checkpoint_path"], + "source_checkpoint_sha256": runtime["source_checkpoint_sha256"], + "type_map": runtime["type_map"], + } + + +def dpa4_recommended_combo(profile: dict) -> dict: + """Return a copy of the one audited machine contract.""" + if not profile.get("published"): + raise RuntimeError("DPA4 runtime profile is unpublished") + return dict(profile["machine_compatibility"]["recommended"][0]) diff --git a/apex/skills/apex-flow/scripts/fetch_models.py b/apex/skills/apex-flow/scripts/fetch_models.py deleted file mode 100644 index 7b06b295..00000000 --- a/apex/skills/apex-flow/scripts/fetch_models.py +++ /dev/null @@ -1,79 +0,0 @@ -#!/usr/bin/env python3 -""" -Optional helper to download the DPA-3.2-5M source checkpoint. - -The skill itself ships the ready-to-run frozen model: - models/DPA-3.2-5M/DPA-3.2-5M-OMat24.pth - -The multi-head source ``.pt`` is NOT bundled and is NOT downloaded by -``apex skill --zip``. Fetch it only when the user explicitly needs another -task head. - -Usage: - python fetch_models.py # show model location - python fetch_models.py --source-checkpoint # optional ~62MB source .pt - python fetch_models.py --source-checkpoint --force -""" - -from __future__ import annotations - -import argparse -import sys -import urllib.request -from pathlib import Path - -MODELS_ROOT = Path(__file__).resolve().parents[1] / "models" - -SOURCE_CHECKPOINT = ( - "DPA-3.2-5M", - "DPA-3.2-5M.pt", - "https://huggingface.co/deepmodelingcommunity/DPA-3.2-5M/resolve/main/DPA-3.2-5M.pt", -) -FROZEN_MODEL = MODELS_ROOT / "DPA-3.2-5M" / "DPA-3.2-5M-OMat24.pth" - - -def _download(url: str, dest: Path, force: bool = False) -> None: - dest.parent.mkdir(parents=True, exist_ok=True) - if dest.is_file() and dest.stat().st_size > 1_000_000 and not force: - print(f"SKIP {dest} ({dest.stat().st_size // (1024 * 1024)} MB)") - return - partial = dest.with_suffix(dest.suffix + ".partial") - print(f"GET {url}") - print(f" -> {dest}") - urllib.request.urlretrieve(url, partial) - partial.replace(dest) - print(f"OK {dest} ({dest.stat().st_size // (1024 * 1024)} MB)") - - -def main(argv: list[str] | None = None) -> int: - # Important: default to [] so runpy / accidental sys.argv inheritance - # (e.g. from `apex skill --zip`) cannot break this script. - if argv is None: - argv = sys.argv[1:] - - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--source-checkpoint", - action="store_true", - help="Download DPA-3.2-5M.pt (~62MB) to freeze a different task head", - ) - parser.add_argument("--force", action="store_true", help="Re-download even if present") - args = parser.parse_args(argv) - - MODELS_ROOT.mkdir(parents=True, exist_ok=True) - if args.source_checkpoint: - _download( - SOURCE_CHECKPOINT[2], - MODELS_ROOT / SOURCE_CHECKPOINT[0] / SOURCE_CHECKPOINT[1], - force=args.force, - ) - else: - print("No source checkpoint requested; use --source-checkpoint if needed.") - - print(f"Bundled frozen model: {FROZEN_MODEL}") - print(f"Models root: {MODELS_ROOT}") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/apex/skills/apex-flow/scripts/generate_config.py b/apex/skills/apex-flow/scripts/generate_config.py index a63a42c4..910029f0 100644 --- a/apex/skills/apex-flow/scripts/generate_config.py +++ b/apex/skills/apex-flow/scripts/generate_config.py @@ -24,7 +24,7 @@ python generate_config.py create \ --structure pristine.vasp Ti_hcp.vasp V_bcc.vasp \ --structure-dir ./defects/ \ - --backend lammps --potential deepmd --model DPA.pth \ + --backend lammps --potential deepmd --model model.pt \ --properties elastic --flow-type relax \ --output-dir ./job @@ -32,7 +32,9 @@ """ import argparse +import copy import json +import math import os import re import shutil @@ -47,12 +49,23 @@ # ============================================================================= TICKET_API_URL = "https://openapi.dp.tech/openapi/v1/ticket/get" +TICKET_EXPIRE_HOURS = 168 # 7 days; API 'expiration' parameter unit is HOURS DFLOW_HOST = "https://workflows.deepmodeling.com" +SANDBOX_DFLOW_HOST = "https://lbg-workflow-dflow.dp.tech" +SANDBOX_DISPATCHER_IMAGE = ( + "registry.dp.tech/dptech/polycalibur:dpdispatcher-storehost-plan-a-20260811" +) APEX_IMAGE = "registry.dp.tech/dptech/dp/native/prod-397637/apex-flow:1.3.0.post" -LAMMPS_IMAGE = ( - "registry.dp.tech/dptech/dp/native/prod-397637/" - "deepmd-kit-phonolammps:3.1.3" +LAMMPS_GPU_IMAGE = ( + "registry.dp.tech/dptech/dp/native/prod-16664/" + "dpa4-phonolammps:0.0.2" ) +LAMMPS_CPU_IMAGE = APEX_IMAGE +# Backward-compatible alias used by external callers and older tests. New code +# must select LAMMPS_GPU_IMAGE or LAMMPS_CPU_IMAGE from the potential type. +LAMMPS_IMAGE = LAMMPS_GPU_IMAGE +ABACUS_IMAGE = "registry.dp.tech/dptech/abacus:3.8.2" +DPA4_RUNTIME_PROFILE = "dpa4-alloytongqi-t4" # Recommended Bohrium VASP run command pieces (Intel oneAPI + absolute vasp_std). # Do NOT auto-set vasp_image_name — VASP is commercial; only set an image after # the user confirms they have a license and provides/approves the image. @@ -106,6 +119,7 @@ "plane_miller": [1, 1, 1], "slip_direction": [-1, 1, 0], "supercell_size": [1, 1, 5], + "vacuum_size": 20, "n_steps": 10, }, "gamma_surface": { @@ -113,7 +127,8 @@ "plane_miller": [1, 1, 1], "slip_direction": [-1, 1, 0], "supercell_size": [1, 1, 5], - "closed_loop": False, + "vacuum_size": 20, + "closed_loop": True, "n_steps_x": 10, "n_steps_y": 10, }, @@ -139,6 +154,27 @@ "strain": 0.001, }, }, + "melting_point": { + "type": "melting_point", + "method": "two_phase", + "supercell_size": [1, 1, 2], + "cal_setting": { + "temperature": [1500, 1600, 1700], + "premelt_temperature": 4500, + "premelt_steps": 5000, + "conditioning_steps": 5000, + "production_steps": 100000, + "restart_interval": 10000, + "timestep": 0.001, + "tdamp": 0.1, + "pdamp": 1.0, + "pressure": 0.0, + "barostat": "iso", + "interface_axis": "z", + "liquid_fraction": 0.5, + "replicas": 1, + }, + }, "gruneisen": { "type": "gruneisen", "supercell_size": [2, 2, 2], @@ -157,22 +193,138 @@ "temp_ramp_rate": 1000, }, }, + "melting_point": { + "type": "melting_point", + "method": "two_phase", + "supercell_size": [1, 1, 2], + "cal_setting": { + "temperature": [1500, 1600, 1700], + "premelt_temperature": 4500, + "premelt_steps": 5000, + "conditioning_steps": 5000, + "production_steps": 100000, + "timestep": 0.001, + "tdamp": 0.1, + "pdamp": 1.0, + "pressure": 0.0, + "barostat": "iso", + "interface_axis": "z", + "liquid_fraction": 0.5, + "dump_step": 100, + "thermo_step": 100, + "restart_interval": 10000, + "q6_cutoff": 3.5, + "q6_neighbors": 12, + "replicas": 1, + "velocity_seeds": { + "premelt": 324159, + "condition": 271828, + "release": 161803, + }, + }, + }, } # LAMMPS-only properties -LAMMPS_ONLY = {"finite_t_elastic"} +LAMMPS_ONLY = {"finite_t_elastic", "melting_point"} # GPU potential types — benefit from GPU scass_type GPU_POTENTIALS = {"deepmd", "mace", "nep"} -# scass_type defaults for inner dflow containers + +def select_lammps_image(potential: str = None) -> str: + """Return the validated LAMMPS image for a potential's resource class.""" + if potential in GPU_POTENTIALS: + return LAMMPS_GPU_IMAGE + return LAMMPS_CPU_IMAGE + + +# scass_type defaults for inner dflow containers (legacy Bohrium) SCASS_TYPES = { - "lammps_gpu": "c8_m31_1 * NVIDIA T4", + "lammps_gpu": "c16_m120_1 * NVIDIA L20", "lammps_cpu": "c16_m32_cpu", "abacus": "c16_m32_cpu", "vasp": "c32_m128_cpu", } +# machine_type defaults for OpenAPI Sandbox +SANDBOX_MACHINE_TYPES = { + "lammps_gpu": "c16_m120_1 * NVIDIA L20", + "lammps_cpu": "c8_m32_cpu", + "abacus": "c8_m32_cpu", + "vasp": "c32_m128_cpu", +} + + +# ============================================================================= +# Adaptive KSPACING based on system type and atom count +# ============================================================================= +# Rules: +# - Bulk crystal: denser k-mesh for small unit cells, sparser for supercells +# - Surface/slab: moderate k-mesh (vacuum direction handled by APEX) +# - Amorphous/liquid: low k-point requirement (no long-range periodicity) +# Always use KGAMMA = .TRUE. +# +# Users can override by providing explicit kspacing in --incar or cal_setting. + +def classify_system(structure) -> str: + """Classify a pymatgen Structure as 'bulk', 'surface', or 'amorphous'. + + Heuristic: + - If any lattice vector > 15 Å and the cell is highly anisotropic + (max/min ratio > 2.5), treat as surface/slab. + - Otherwise treat as bulk crystal. + - 'amorphous' must be explicitly requested by the user (not auto-detected). + """ + lengths = sorted(structure.lattice.abc) + ratio = lengths[-1] / max(lengths[0], 0.1) + if lengths[-1] > 15.0 and ratio > 2.5: + return "surface" + return "bulk" + + +def adaptive_kspacing(structure, system_type: str = None) -> float: + """Determine KSPACING based on system type and atom count. + + Args: + structure: pymatgen Structure object + system_type: 'bulk', 'surface', or 'amorphous' (auto-detected if None) + + Returns: + Recommended KSPACING value (Å⁻¹ for VASP, 1/Bohr for ABACUS) + + Rules (unified for all DFT properties): + bulk_crystal: + N <= 10: 0.16 + 10 < N <= 50: 0.20 + 50 < N <= 150: 0.25 + N > 150: 0.30 + surface_slab: + N <= 100: 0.22 + N > 100: 0.28 + amorphous_liquid: + N <= 100: 0.30 + N > 100: 0.38 + """ + if system_type is None: + system_type = classify_system(structure) + + n_atoms = len(structure) + + if system_type == "amorphous": + return 0.30 if n_atoms <= 100 else 0.38 + elif system_type == "surface": + return 0.22 if n_atoms <= 100 else 0.28 + else: # bulk (default) + if n_atoms <= 10: + return 0.16 + elif n_atoms <= 50: + return 0.20 + elif n_atoms <= 150: + return 0.25 + else: + return 0.30 + def _nprocs_from_scass(scass_type: str, default: int = 16) -> int: """Extract CPU count from scass strings like ``c32_m128_cpu`` → 32.""" @@ -218,7 +370,7 @@ def default_vasp_run_command(nprocs: int = 32) -> str: "force_thr_ev": 0.02, "stress_thr": 1.0, "relax_nmax": 50, - "kspacing": 0.20, # FAST: ~6x6x6 for typical metals (vs 0.10 → 12x12x12) + "kspacing": None, # Auto-filled by adaptive_kspacing(structure) }, "phonon_scf": { "calculation": "scf", @@ -231,10 +383,27 @@ def default_vasp_run_command(nprocs: int = 32) -> str: "mixing_type": "broyden", "mixing_beta": 0.7, "cal_force": 1, - "kspacing": 0.15, # Slightly tighter for force accuracy + "kspacing": None, # Auto-filled by adaptive_kspacing(structure) }, } +# Default VASP INCAR template (KSPACING auto-filled by adaptive_kspacing) +VASP_INCAR_TEMPLATE = """\ +SYSTEM = APEX calculation +PREC = Accurate +ENCUT = 520 +EDIFF = 1E-6 +EDIFFG = -0.01 +IBRION = 2 +NSW = 200 +ISIF = 3 +ISMEAR = 1 +SIGMA = 0.1 +LREAL = Auto +KSPACING = {kspacing} +KGAMMA = .TRUE. +""" + # DFT-specific property defaults: smaller supercells, fewer points DFT_PROPERTY_OVERRIDES = { "phonon": { @@ -264,18 +433,25 @@ def default_vasp_run_command(nprocs: int = 32) -> str: # Ticket conversion # ============================================================================= -def get_bohrium_ticket(access_key: str) -> str: +def get_bohrium_ticket(access_key: str, expire_hours: int = TICKET_EXPIRE_HOURS) -> str: """ Convert a Bohrium access_key to a dflow ticket via the OpenAPI. - API: GET https://openapi.dp.tech/openapi/v1/ticket/get?accessKey= + API: GET https://openapi.dp.tech/openapi/v1/ticket/get?accessKey=&expiration= Header: x-app-key: (empty string) Response: {"code": 0, "data": {"ticket": "UUID-36-chars"}} + Args: + access_key: Bohrium access key from environment. + expire_hours: Ticket validity in HOURS. Default 168 (7 days). + Must be called from sandbox where BOHRIUM_ACCESS_KEY is available. + The generated ticket is embedded in global.json for use by + containers that lack the access key. + Returns the ticket string (UUID). Raises RuntimeError on failure. """ - url = f"{TICKET_API_URL}?accessKey={access_key}" + url = f"{TICKET_API_URL}?accessKey={access_key}&expiration={expire_hours}" req = Request(url, method="GET") req.add_header("x-app-key", "") @@ -360,14 +536,46 @@ def resolve_project_id(project_id: int = None) -> int: ) from exc -def _validate_image_scass(image: str, scass: str) -> None: +def _load_dpa4_runtime_profile(runtime_profile): + """Resolve the one immutable DPA4 profile without duplicating its contract.""" + if runtime_profile is None: + return None + if isinstance(runtime_profile, dict): + script_dir = Path(__file__).resolve().parent + if str(script_dir) not in sys.path: + sys.path.insert(0, str(script_dir)) + from dpa4_profile import validate_dpa4_profile # noqa: WPS433 + + return validate_dpa4_profile( + runtime_profile, + require_published=True, + ) + if runtime_profile != DPA4_RUNTIME_PROFILE: + raise ValueError(f"Unknown runtime profile: {runtime_profile!r}") + script_dir = Path(__file__).resolve().parent + if str(script_dir) not in sys.path: + sys.path.insert(0, str(script_dir)) + from dpa4_profile import load_dpa4_profile # noqa: WPS433 + + return load_dpa4_profile(require_published=True) + + +def _validate_image_scass( + image: str, + scass: str, + runtime_profile: str = None, +) -> None: """Reject known-bad image × scass_type combinations before write-out.""" script_dir = Path(__file__).resolve().parent if str(script_dir) not in sys.path: sys.path.insert(0, str(script_dir)) from validate_apex_combo import check_combo # noqa: WPS433 - ok, errors = check_combo(image, scass) + ok, errors = check_combo( + image, + scass, + runtime_profile=runtime_profile, + ) if not ok: raise RuntimeError( "Blocked image × scass_type combination:\n - " @@ -380,7 +588,9 @@ def build_global_json(backend: str, potential: str = None, access_key: str = None, project_id: int = None, scass_type: str = None, run_command: str = None, - vasp_image: str = None) -> dict: + vasp_image: str = None, + lammps_image: str = None, + runtime_profile=None) -> dict: """ Build global.json for APEX dflow submission. @@ -390,6 +600,32 @@ def build_global_json(backend: str, potential: str = None, ``program_id`` / ``bohrium_config.project_id`` always come from ``--project-id`` or ``BOHRIUM_PROJECT_ID`` (required). """ + dpa4_profile = _load_dpa4_runtime_profile(runtime_profile) + if dpa4_profile is not None: + if backend != "lammps" or potential != "deepmd": + raise ValueError( + f"--runtime-profile {DPA4_RUNTIME_PROFILE} requires " + "--backend lammps --potential deepmd" + ) + script_dir = Path(__file__).resolve().parent + if str(script_dir) not in sys.path: + sys.path.insert(0, str(script_dir)) + from dpa4_profile import ( # noqa: WPS433 + dpa4_image_name, + dpa4_recommended_combo, + ) + + dpa4_combo = dpa4_recommended_combo(dpa4_profile) + expected_scass = dpa4_combo["scass_type"] + expected_command = dpa4_profile["calculator"]["run_command"] + if scass_type is not None and scass_type != expected_scass: + raise ValueError( + f"DPA4 runtime profile requires scass_type={expected_scass!r}" + ) + if run_command is not None and run_command != expected_command: + raise ValueError( + f"DPA4 runtime profile requires run_command={expected_command!r}" + ) pid = resolve_project_id(project_id) # Resolve access key and convert to ticket @@ -402,7 +638,9 @@ def build_global_json(backend: str, potential: str = None, ticket = get_bohrium_ticket(key) # Determine scass_type for inner containers - if scass_type: + if dpa4_profile is not None: + inner_scass = dpa4_combo["scass_type"] + elif scass_type: inner_scass = scass_type elif backend == "lammps" and potential in GPU_POTENTIALS: inner_scass = SCASS_TYPES["lammps_gpu"] @@ -416,7 +654,9 @@ def build_global_json(backend: str, potential: str = None, inner_scass = SCASS_TYPES["lammps_cpu"] # Determine run command for calculator - if run_command: + if dpa4_profile is not None: + calc_run_command = dpa4_profile["calculator"]["run_command"] + elif run_command: calc_run_command = run_command elif backend == "lammps": calc_run_command = "lmp -in in.lammps" @@ -430,30 +670,47 @@ def build_global_json(backend: str, potential: str = None, calc_run_command = "lmp -in in.lammps" # Determine calculator image - if backend == "lammps": - lammps_image = LAMMPS_IMAGE + if dpa4_profile is not None: + selected_lammps_image = dpa4_image_name(dpa4_profile) + elif backend == "lammps": + selected_lammps_image = ( + (lammps_image or "").strip() + or select_lammps_image(potential) + ) else: - lammps_image = LAMMPS_IMAGE # Still needed as fallback in global.json + selected_lammps_image = LAMMPS_CPU_IMAGE # Fallback in global.json - _validate_image_scass(lammps_image, inner_scass) + _validate_image_scass( + selected_lammps_image, + inner_scass, + runtime_profile=( + DPA4_RUNTIME_PROFILE if dpa4_profile is not None else None + ), + ) config = { "dflow_host": DFLOW_HOST, "k8s_api_server": DFLOW_HOST, "batch_type": "Bohrium", "context_type": "Bohrium", + "job_type": "container", + "platform": "ali", "program_id": pid, "bohrium_config": { "ticket": ticket, "project_id": pid, }, "apex_image_name": APEX_IMAGE, - "lammps_image_name": lammps_image, + "lammps_image_name": selected_lammps_image, "lammps_run_command": calc_run_command, "scass_type": inner_scass, "group_size": 1, "pool_size": 1, } + if dpa4_profile is not None: + config["phonolammps_run_command"] = dpa4_profile["calculator"][ + "phonolammps_command" + ] # Add backend-specific image fields if backend == "abacus": @@ -470,8 +727,134 @@ def build_global_json(backend: str, potential: str = None, return config +def build_global_json_sandbox(backend: str, potential: str = None, + access_key: str = None, project_id: int = None, + machine_type: str = None, + run_command: str = None, + vasp_image: str = None, + lammps_image: str = None) -> dict: + """ + Build global.json for APEX dflow submission via OpenAPI Sandbox. + + Uses access_key directly (no ticket conversion needed). + Routes jobs through the Sandbox dflow host with the storeHost-patched + dispatcher sidecar image. + """ + pid = resolve_project_id(project_id) + + # Resolve access key + key = access_key or os.environ.get("BOHRIUM_ACCESS_KEY") + if not key: + raise RuntimeError( + "BOHRIUM_ACCESS_KEY environment variable not set and --access-key " + "not provided. Required for OpenAPI Sandbox mode." + ) + + # Determine machine_type for inner containers + if machine_type: + inner_machine = machine_type + elif backend == "lammps" and potential in GPU_POTENTIALS: + inner_machine = SANDBOX_MACHINE_TYPES["lammps_gpu"] + elif backend == "lammps": + inner_machine = SANDBOX_MACHINE_TYPES["lammps_cpu"] + elif backend == "abacus": + inner_machine = SANDBOX_MACHINE_TYPES["abacus"] + elif backend == "vasp": + inner_machine = SANDBOX_MACHINE_TYPES["vasp"] + else: + inner_machine = SANDBOX_MACHINE_TYPES["lammps_cpu"] + + # Determine run command for calculator + if run_command: + calc_run_command = run_command + elif backend == "lammps": + calc_run_command = "lmp -in in.lammps" + elif backend == "abacus": + calc_run_command = "mpirun -n 8 abacus" + elif backend == "vasp": + calc_run_command = default_vasp_run_command( + _nprocs_from_scass(inner_machine, default=32) + ) + else: + calc_run_command = "lmp -in in.lammps" + + # Determine calculator image + selected_lammps_image = ( + ((lammps_image or "").strip() or select_lammps_image(potential)) + if backend == "lammps" + else LAMMPS_CPU_IMAGE + ) + _validate_image_scass(selected_lammps_image, inner_machine) + + config = { + "dflow_host": SANDBOX_DFLOW_HOST, + "k8s_api_server": SANDBOX_DFLOW_HOST, + "dflow_config": { + "host": SANDBOX_DFLOW_HOST, + "k8s_api_server": SANDBOX_DFLOW_HOST, + "namespace": "dflow", + "token": "", + }, + "batch_type": "OpenAPI", + "context_type": "OpenAPI", + "access_key": key, + "project_id": pid, + "app_key": os.environ.get("BOHRIUM_APP_KEY", "agent"), + "platform": "ali", + "machine_type": inner_machine, + "image_address": selected_lammps_image if backend == "lammps" else ABACUS_IMAGE if backend == "abacus" else APEX_IMAGE, + "output_log": False, + "dispatcher_image": SANDBOX_DISPATCHER_IMAGE, + "bohrium_config": { + "access_key": key, + "project_id": pid, + "app_key": os.environ.get("BOHRIUM_APP_KEY", "agent"), + }, + "apex_image_name": APEX_IMAGE, + "lammps_image_name": selected_lammps_image, + "lammps_run_command": calc_run_command, + "group_size": 1, + "pool_size": 1, + } + + # Add backend-specific image fields + if backend == "abacus": + config["abacus_image_name"] = APEX_IMAGE + config["abacus_run_command"] = calc_run_command + elif backend == "vasp": + config["vasp_run_command"] = calc_run_command + if vasp_image: + config["vasp_image_name"] = vasp_image.strip() + + return config + + def validate_project_id_types(config: dict) -> int: """Hard-check every supported Bohrium project ID before submission.""" + # OpenAPI Sandbox mode uses project_id directly (no program_id) + context_type = config.get("context_type", "") + if context_type.lower() == "openapi": + project_id = config.get("project_id") + if type(project_id) is not int or project_id <= 0: + raise ValueError( + "global.json project_id must be a positive, unquoted JSON integer " + "(OpenAPI Sandbox mode)" + ) + bohrium_config = config.get("bohrium_config") + if isinstance(bohrium_config, dict): + bc_pid = bohrium_config.get("project_id") + if type(bc_pid) is not int or bc_pid <= 0: + raise ValueError( + "global.json bohrium_config.project_id must be a positive, " + "unquoted JSON integer" + ) + if bc_pid != project_id: + raise ValueError( + "global.json project_id and bohrium_config.project_id must match" + ) + return project_id + + # Legacy Bohrium mode uses program_id program_id = config.get("program_id") if type(program_id) is not int or program_id <= 0: raise ValueError( @@ -566,8 +949,27 @@ def refresh_global_json(global_path, access_key: str = None, def build_interaction(backend: str, potential: str = None, model: str = None, incar: str = None, potcar_prefix: str = None, - potcars: dict = None, orb_files: dict = None) -> dict: + potcars: dict = None, orb_files: dict = None, + runtime_profile=None) -> dict: """Build interaction configuration.""" + dpa4_profile = _load_dpa4_runtime_profile(runtime_profile) + if dpa4_profile is not None: + if backend != "lammps" or potential != "deepmd": + raise ValueError( + f"--runtime-profile {DPA4_RUNTIME_PROFILE} requires " + "--backend lammps --potential deepmd" + ) + if model: + raise ValueError( + "DPA4 runtime profile uses an image-resident PT2 artifact; " + "do not pass --model or stage model.pt as the LAMMPS runtime" + ) + script_dir = Path(__file__).resolve().parent + if str(script_dir) not in sys.path: + sys.path.insert(0, str(script_dir)) + from dpa4_profile import dpa4_interaction # noqa: WPS433 + + return dpa4_interaction(dpa4_profile) if backend == "lammps": if not potential: raise ValueError("--potential required for LAMMPS backend") @@ -602,6 +1004,19 @@ def build_interaction(backend: str, potential: str = None, raise ValueError(f"Unknown backend: {backend}") +def stage_lammps_model(model: str, output_dir: Path) -> str: + """Copy a LAMMPS model into the job root and return its relative name.""" + if not model: + raise ValueError("--model required for LAMMPS backend") + source = Path(model).expanduser() + if not source.is_file(): + raise FileNotFoundError(f"LAMMPS model not found: {source}") + destination = Path(output_dir) / source.name + shutil.copy2(source, destination) + print(f"Copied model to {destination}") + return destination.name + + def resolve_source_potcar_file(prefix: Path, entry: str) -> Path: """Locate a readable POTCAR under prefix for a potcars entry.""" direct = (prefix / entry).expanduser() @@ -616,6 +1031,153 @@ def resolve_source_potcar_file(prefix: Path, entry: str) -> Path: ) + +# ============================================================================= +# ABACUS pseudopotential and orbital file download/staging +# ============================================================================= + +# SG15 ONCV pseudopotential + DZP orbital download URLs (Gitee mirror) +ABACUS_PP_ORB_BASE = "https://gitee.com/deepmodeling/abacus-develop/raw/develop/tests/PP_ORB" + +# Known pp/orb filenames per element (SG15 ONCV PBE + numerical orbital) +# Only commonly used elements listed; extend as needed. +ABACUS_PP_MAP = { + "H": "H_ONCV_PBE-1.0.upf", + "Li": "Li_ONCV_PBE-1.0.upf", + "Be": "Be_ONCV_PBE-1.0.upf", + "B": "B_ONCV_PBE-1.0.upf", + "C": "C_ONCV_PBE-1.0.upf", + "N": "N_ONCV_PBE-1.0.upf", + "O": "O_ONCV_PBE-1.0.upf", + "F": "F_ONCV_PBE-1.0.upf", + "Na": "Na_ONCV_PBE-1.0.upf", + "Mg": "Mg_ONCV_PBE-1.0.upf", + "Al": "Al_ONCV_PBE-1.0.upf", + "Si": "Si_ONCV_PBE-1.0.upf", + "P": "P_ONCV_PBE-1.0.upf", + "S": "S_ONCV_PBE-1.0.upf", + "Cl": "Cl_ONCV_PBE-1.0.upf", + "K": "K_ONCV_PBE-1.0.upf", + "Ca": "Ca_ONCV_PBE-1.0.upf", + "Ti": "Ti_ONCV_PBE-1.0.upf", + "V": "V_ONCV_PBE-1.0.upf", + "Cr": "Cr_ONCV_PBE-1.0.upf", + "Mn": "Mn_ONCV_PBE-1.0.upf", + "Fe": "Fe_ONCV_PBE-1.0.upf", + "Co": "Co_ONCV_PBE-1.0.upf", + "Ni": "Ni_ONCV_PBE-1.0.upf", + "Cu": "Cu_ONCV_PBE-1.0.upf", + "Zn": "Zn_ONCV_PBE-1.0.upf", + "Ga": "Ga_ONCV_PBE-1.0.upf", + "Ge": "Ge_ONCV_PBE-1.0.upf", + "As": "As_ONCV_PBE-1.0.upf", + "Se": "Se_ONCV_PBE-1.0.upf", + "Br": "Br_ONCV_PBE-1.0.upf", + "Sr": "Sr_ONCV_PBE-1.0.upf", + "Y": "Y_ONCV_PBE-1.0.upf", + "Zr": "Zr_ONCV_PBE-1.0.upf", + "Nb": "Nb_ONCV_PBE-1.0.upf", + "Mo": "Mo_ONCV_PBE-1.0.upf", + "Ag": "Ag_ONCV_PBE-1.0.upf", + "Sn": "Sn_ONCV_PBE-1.0.upf", + "Ba": "Ba_ONCV_PBE-1.0.upf", + "W": "W_ONCV_PBE-1.0.upf", + "Pt": "Pt_ONCV_PBE-1.0.upf", + "Au": "Au_ONCV_PBE-1.0.upf", + "Pb": "Pb_ONCV_PBE-1.0.upf", +} + +ABACUS_ORB_MAP = { + "H": "H_gga_6au_100Ry_2s1p.orb", + "Li": "Li_gga_7au_100Ry_4s1p.orb", + "C": "C_gga_7au_100Ry_2s2p1d.orb", + "N": "N_gga_7au_100Ry_2s2p1d.orb", + "O": "O_gga_7au_100Ry_2s2p1d.orb", + "Al": "Al_gga_7au_100Ry_4s4p1d.orb", + "Si": "Si_gga_7au_100Ry_2s2p1d.orb", + "Ti": "Ti_gga_8au_100Ry_4s2p2d1f.orb", + "V": "V_gga_8au_100Ry_4s2p2d1f.orb", + "Cr": "Cr_gga_8au_100Ry_4s2p2d1f.orb", + "Mn": "Mn_gga_8au_100Ry_4s2p2d1f.orb", + "Fe": "Fe_gga_8au_100Ry_4s2p2d1f.orb", + "Co": "Co_gga_8au_100Ry_4s2p2d1f.orb", + "Ni": "Ni_gga_8au_100Ry_4s2p2d1f.orb", + "Cu": "Cu_gga_9au_100Ry_4s2p2d1f.orb", + "Zn": "Zn_gga_8au_100Ry_4s2p2d1f.orb", + "Ga": "Ga_gga_8au_100Ry_4s2p2d1f.orb", + "Ge": "Ge_gga_8au_100Ry_2s2p2d1f.orb", + "Mo": "Mo_gga_9au_100Ry_4s2p2d1f.orb", + "Ag": "Ag_gga_9au_100Ry_4s2p2d1f.orb", + "W": "W_gga_8au_100Ry_4s2p2d1f.orb", + "Au": "Au_gga_9au_100Ry_4s2p2d1f.orb", +} + + +def download_abacus_pp_orb(elements: list, output_dir: Path) -> tuple: + """Download ABACUS pseudopotential and orbital files for given elements. + + Downloads from Gitee mirror of ABACUS test PP_ORB directory. + Returns (potcars_dict, orb_files_dict) with filenames relative to pp_orb/. + """ + pp_orb_dir = output_dir / "pp_orb" + pp_orb_dir.mkdir(parents=True, exist_ok=True) + + potcars = {} + orb_files = {} + missing_pp = [] + missing_orb = [] + + for elem in elements: + # Download pseudopotential + pp_name = ABACUS_PP_MAP.get(elem) + if not pp_name: + missing_pp.append(elem) + continue + pp_path = pp_orb_dir / pp_name + if not pp_path.exists(): + url = f"{ABACUS_PP_ORB_BASE}/{pp_name}" + print(f" Downloading PP for {elem}: {pp_name} ...", end=" ") + try: + req = Request(url, headers={"User-Agent": "APEX-generate-config"}) + with urlopen(req, timeout=30) as resp: + pp_path.write_bytes(resp.read()) + print("OK") + except (URLError, HTTPError, OSError) as exc: + print(f"FAILED ({exc})") + missing_pp.append(elem) + continue + potcars[elem] = pp_name + + # Download orbital file + orb_name = ABACUS_ORB_MAP.get(elem) + if not orb_name: + missing_orb.append(elem) + continue + orb_path = pp_orb_dir / orb_name + if not orb_path.exists(): + url = f"{ABACUS_PP_ORB_BASE}/{orb_name}" + print(f" Downloading ORB for {elem}: {orb_name} ...", end=" ") + try: + req = Request(url, headers={"User-Agent": "APEX-generate-config"}) + with urlopen(req, timeout=30) as resp: + orb_path.write_bytes(resp.read()) + print("OK") + except (URLError, HTTPError, OSError) as exc: + print(f"FAILED ({exc})") + missing_orb.append(elem) + continue + orb_files[elem] = orb_name + + if missing_pp: + print(f" WARNING: No PP mapping for elements: {missing_pp}") + print(" Provide --potcars manually for these elements.") + if missing_orb: + print(f" WARNING: No ORB mapping for elements: {missing_orb}") + print(" Provide --orb-files manually or use PW basis.") + + return potcars, orb_files + + def stage_vasp_potcars( output_dir: Path, source_prefix: str, potcars: dict ) -> tuple: @@ -765,7 +1327,10 @@ def plan_structure_layout(sources: list) -> list: def build_param_json(structure_paths, interaction: dict, properties: list, flow_type: str = "joint", - relaxation_settings: dict = None) -> dict: + relaxation_settings: dict = None, + gamma_overrides: dict = None, + property_configs: list = None, + melting_overrides: dict = None) -> dict: """Build param.json configuration. structure_paths: one path string or a list of path strings for `structures`. @@ -805,14 +1370,26 @@ def build_param_json(structure_paths, interaction: dict, # Properties if flow_type in ("joint", "props"): + if property_configs is not None: + param["properties"] = copy.deepcopy(property_configs) + return param prop_configs = [] is_dft = interaction.get("type") in ("abacus", "vasp") for prop_name in properties: if prop_name in PROPERTY_DEFAULTS: - prop_config = PROPERTY_DEFAULTS[prop_name].copy() + prop_config = copy.deepcopy(PROPERTY_DEFAULTS[prop_name]) # Apply DFT-specific overrides (smaller supercells, fewer points) if is_dft and prop_name in DFT_PROPERTY_OVERRIDES: prop_config.update(DFT_PROPERTY_OVERRIDES[prop_name]) + if prop_name in {"gamma", "gamma_surface"}: + prop_config.update( + gamma_overrides_for_property( + prop_name, gamma_overrides or {} + ) + ) + if prop_name == "melting_point" and melting_overrides: + cal_setting = prop_config.setdefault("cal_setting", {}) + cal_setting.update(copy.deepcopy(melting_overrides)) prop_configs.append(prop_config) else: print(f"Warning: Unknown property '{prop_name}', skipping", @@ -822,6 +1399,315 @@ def build_param_json(structure_paths, interaction: dict, return param +def load_confirmed_property_configs(path: str, requested: list) -> list: + """Load an approved property list and select entries in requested order.""" + source = Path(path).expanduser() + payload = json.loads(source.read_text(encoding="utf-8")) + configs = payload.get("properties") if isinstance(payload, dict) else payload + if not isinstance(configs, list) or not all( + isinstance(item, dict) and isinstance(item.get("type"), str) + for item in configs + ): + raise ValueError( + "--property-config must be a JSON list of property objects or " + "an object containing a 'properties' list" + ) + by_type = {} + for item in configs: + prop_type = item["type"] + if prop_type in by_type: + raise ValueError( + f"--property-config contains duplicate type: {prop_type}" + ) + by_type[prop_type] = item + missing = [prop_type for prop_type in requested if prop_type not in by_type] + if missing: + raise ValueError( + "--property-config is missing requested properties: " + + ", ".join(missing) + ) + return [copy.deepcopy(by_type[prop_type]) for prop_type in requested] + + +def _normalized_numeric_sequence(values): + """Keep integer-looking CLI values as ints while allowing transformed vectors.""" + normalized = [] + for value in values: + number = float(value) + normalized.append(int(number) if number.is_integer() else number) + return normalized + + +def gamma_overrides_from_args(args) -> dict: + """Collect only explicitly supplied ``--gamma-*`` options.""" + option_map = { + "parent_lattice": "gamma_parent_lattice", + "plane_miller": "gamma_plane_miller", + "slip_direction": "gamma_slip_direction", + "supercell_size": "gamma_supercell_size", + "vacuum_size": "gamma_vacuum_size", + "require_orthogonal_cell": "gamma_require_orthogonal_cell", + "min_slab_height": "gamma_min_slab_height", + "max_atoms": "gamma_max_atoms", + "min_distance": "gamma_min_distance", + "displacement_points": "gamma_displacement_points", + "n_steps": "gamma_n_steps", + "n_steps_x": "gamma_n_steps_x", + "n_steps_y": "gamma_n_steps_y", + "closed_loop": "gamma_closed_loop", + } + overrides = {} + for key, attr in option_map.items(): + value = getattr(args, attr, None) + if value is None: + continue + if key in {"plane_miller", "slip_direction", "supercell_size"}: + value = _normalized_numeric_sequence(value) + elif key == "displacement_points": + value = sorted(float(item) for item in value) + overrides[key] = value + return overrides + + +def gamma_overrides_for_property(prop_name: str, overrides: dict) -> dict: + """Return the subset of Gamma CLI overrides valid for one property type.""" + common = { + "parent_lattice", + "plane_miller", + "slip_direction", + "supercell_size", + "vacuum_size", + "min_slab_height", + "max_atoms", + "min_distance", + "require_orthogonal_cell", + } + allowed = common | ( + {"n_steps", "displacement_points"} if prop_name == "gamma" + else {"n_steps_x", "n_steps_y", "closed_loop"} + ) + return {key: copy.deepcopy(value) for key, value in overrides.items() + if key in allowed} + + +def validate_gamma_cli_options(properties: list, overrides: dict) -> list: + """Reject ambiguous or misplaced Gamma-only command-line options.""" + if not overrides: + return [] + errors = [] + requested = set(properties) + gamma_types = requested & {"gamma", "gamma_surface"} + if not gamma_types: + return [ + "--gamma-* options require --properties gamma and/or gamma_surface" + ] + if "n_steps" in overrides and "gamma" not in gamma_types: + errors.append("--gamma-n-steps requires --properties gamma") + if "displacement_points" in overrides and "gamma" not in gamma_types: + errors.append( + "--gamma-displacement-points requires --properties gamma" + ) + parent_lattice = overrides.get("parent_lattice") + if parent_lattice is not None and parent_lattice not in {"bcc", "fcc", "hcp"}: + errors.append( + "--gamma-parent-lattice must be one of bcc, fcc, or hcp" + ) + surface_only = {"n_steps_x", "n_steps_y", "closed_loop"} + if surface_only & overrides.keys() and "gamma_surface" not in gamma_types: + errors.append( + "--gamma-n-steps-x/--gamma-n-steps-y/--gamma-closed-loop " + "require --properties gamma_surface" + ) + plane = overrides.get("plane_miller") + direction = overrides.get("slip_direction") + if plane is not None and not 3 <= len(plane) <= 4: + errors.append("--gamma-plane-miller requires 3 or 4 components") + if direction is not None and not 3 <= len(direction) <= 4: + errors.append("--gamma-slip-direction requires 3 or 4 components") + if ( + plane is not None + and direction is not None + and len(plane) != len(direction) + ): + errors.append( + "--gamma-plane-miller and --gamma-slip-direction dimensions differ" + ) + for values, option in ( + (plane, "--gamma-plane-miller"), + (direction, "--gamma-slip-direction"), + ): + if values is not None and ( + any(not math.isfinite(float(value)) for value in values) + or not any(float(value) != 0 for value in values) + ): + errors.append(f"{option} must be a finite, non-zero vector") + for prop_name in gamma_types: + effective = copy.deepcopy(PROPERTY_DEFAULTS[prop_name]) + effective.update(gamma_overrides_for_property(prop_name, overrides)) + effective_plane = effective["plane_miller"] + effective_direction = effective["slip_direction"] + if ( + len(effective_plane) == len(effective_direction) + and all( + math.isfinite(float(value)) + for value in effective_plane + effective_direction + ) + and not math.isclose( + sum( + float(p) * float(d) + for p, d in zip(effective_plane, effective_direction) + ), + 0.0, + abs_tol=1.0e-10, + ) + ): + errors.append( + f"{prop_name}: --gamma-slip-direction must lie on " + "--gamma-plane-miller in the selected crystallographic " + "basis (the parent basis when --gamma-parent-lattice is set)" + ) + supercell = overrides.get("supercell_size") + if supercell is not None: + for index, value in enumerate(supercell[:2]): + if ( + not isinstance(value, int) + or isinstance(value, bool) + or value <= 0 + ): + errors.append( + f"--gamma-supercell-size component {index + 1} " + "must be a positive integer" + ) + if ( + not math.isfinite(float(supercell[2])) + or float(supercell[2]) <= 0 + ): + errors.append( + "--gamma-supercell-size plane count must be positive and finite" + ) + for key, option in ( + ("min_slab_height", "--gamma-min-slab-height"), + ("min_distance", "--gamma-min-distance"), + ): + if key in overrides: + value = float(overrides[key]) + minimum_ok = value > 0 if key == "min_slab_height" else value >= 0 + if not math.isfinite(value) or not minimum_ok: + qualifier = "positive" if key == "min_slab_height" else "non-negative" + errors.append(f"{option} must be {qualifier} and finite") + if "vacuum_size" in overrides: + value = float(overrides["vacuum_size"]) + if not math.isfinite(value) or value < 0: + errors.append( + "--gamma-vacuum-size must be non-negative and finite" + ) + if "require_orthogonal_cell" in overrides and not isinstance( + overrides["require_orthogonal_cell"], bool + ): + errors.append("--gamma-require-orthogonal-cell must be a boolean flag") + if "max_atoms" in overrides and ( + not isinstance(overrides["max_atoms"], int) + or isinstance(overrides["max_atoms"], bool) + or overrides["max_atoms"] <= 0 + ): + errors.append("--gamma-max-atoms must be a positive integer") + for key, option in ( + ("n_steps", "--gamma-n-steps"), + ("n_steps_x", "--gamma-n-steps-x"), + ("n_steps_y", "--gamma-n-steps-y"), + ): + if key in overrides and ( + not isinstance(overrides[key], int) + or isinstance(overrides[key], bool) + or overrides[key] <= 0 + ): + errors.append(f"{option} must be a positive integer") + if "closed_loop" in overrides and not isinstance( + overrides["closed_loop"], bool + ): + errors.append("--gamma-closed-loop must be a boolean flag") + if "displacement_points" in overrides: + points = overrides["displacement_points"] + if ( + not points + or any( + not math.isfinite(float(value)) + or not 0.0 <= float(value) <= 1.0 + for value in points + ) + or len(set(points)) != len(points) + or 0.0 not in points + ): + errors.append( + "--gamma-displacement-points must include 0 and contain " + "unique finite values in [0, 1]" + ) + return errors + + +def melting_overrides_from_args(args) -> dict: + """Collect explicitly supplied melting-point settings.""" + overrides = {} + temperatures = getattr(args, "melting_temperatures", None) + if temperatures is not None: + overrides["temperature"] = _normalized_numeric_sequence(temperatures) + replicas = getattr(args, "melting_replicas", None) + if replicas is not None: + overrides["replicas"] = replicas + restart_files = getattr(args, "melting_restart_files", None) + if restart_files is not None: + overrides["restart_files"] = list(restart_files) + return overrides + + +def validate_melting_cli_options(properties: list, overrides: dict) -> list: + """Validate scoped melting-point CLI options before ticket conversion.""" + if not overrides: + return [] + if "melting_point" not in set(properties): + return [ + "--melting-* options require --properties melting_point" + ] + errors = [] + temperatures = overrides.get( + "temperature", + PROPERTY_DEFAULTS["melting_point"]["cal_setting"]["temperature"], + ) + if ( + not temperatures + or any( + isinstance(value, bool) + or not isinstance(value, (int, float)) + or not math.isfinite(float(value)) + or float(value) <= 0 + for value in temperatures + ) + ): + errors.append( + "--melting-temperatures must contain positive finite values" + ) + replicas = overrides.get("replicas", 1) + if ( + not isinstance(replicas, int) + or isinstance(replicas, bool) + or replicas < 1 + ): + errors.append("--melting-replicas must be a positive integer") + restart_files = overrides.get("restart_files") + if restart_files is not None: + if len(restart_files) != len(temperatures): + errors.append( + "--melting-restart-files requires exactly one file per " + "melting temperature" + ) + missing = [path for path in restart_files if not Path(path).is_file()] + if missing: + errors.append( + "--melting-restart-files not found: " + ", ".join(missing) + ) + return errors + + # ============================================================================= # Validation # ============================================================================= @@ -896,8 +1782,114 @@ def main(): help="LAMMPS potential type (deepmd/mace/nep/eam_alloy/...)") create.add_argument("--model", "-m", help="Model/potential file path") + create.add_argument( + "--runtime-profile", + choices=[DPA4_RUNTIME_PROFILE], + help=( + "Use an immutable image-resident runtime contract. The DPA4 " + "profile remains unavailable until its exact image digest passes " + "post-snapshot qualification." + ), + ) create.add_argument("--properties", nargs="+", required=True, help="Property types to calculate") + create.add_argument( + "--property-config", + help=( + "Confirmed JSON property list (or object with a properties list). " + "Requested --properties are selected in command-line order." + ), + ) + create.add_argument( + "--gamma-parent-lattice", choices=("bcc", "fcc", "hcp"), + help=( + "Explicit parent lattice for RSS/disordered Gamma inputs; Miller " + "indices are interpreted in this parent basis" + ), + ) + create.add_argument( + "--gamma-plane-miller", nargs="+", type=float, + help=( + "Gamma slip-plane indices in the selected crystallographic basis " + "(parent basis when --gamma-parent-lattice is set)" + ), + ) + create.add_argument( + "--gamma-slip-direction", nargs="+", type=float, + help=( + "Gamma slip-direction components in the selected crystallographic " + "basis (parent basis when --gamma-parent-lattice is set)" + ), + ) + create.add_argument( + "--gamma-supercell-size", nargs=3, type=float, + metavar=("X", "Y", "PLANES"), + help="Gamma in-plane repeats and target Miller-plane spacings", + ) + create.add_argument( + "--gamma-vacuum-size", type=float, metavar="ANGSTROM", + help="Gamma/GammaSurface vacuum thickness (core default: 20 A)", + ) + create.add_argument( + "--gamma-require-orthogonal-cell", + action="store_true", + default=None, + help=( + "Fail unless Gamma/GammaSurface generates an orthogonal " + "Cartesian-z zero-tilt slab; never changes periodic boundaries" + ), + ) + create.add_argument( + "--gamma-min-slab-height", type=float, metavar="ANGSTROM", + help="Minimum material thickness for Gamma slabs", + ) + create.add_argument( + "--gamma-max-atoms", type=int, metavar="COUNT", + help="Maximum allowed atom count for a generated Gamma slab", + ) + create.add_argument( + "--gamma-min-distance", type=float, metavar="ANGSTROM", + help="Minimum allowed periodic atom-pair distance", + ) + create.add_argument( + "--gamma-n-steps", type=int, metavar="COUNT", + help="Gamma-line increments (the generated point count is COUNT + 1)", + ) + create.add_argument( + "--gamma-displacement-points", nargs="+", type=float, + metavar="FRACTION", + help=( + "Explicit Gamma-line fractions in [0,1]; must be unique and " + "include 0, and overrides the n_steps grid" + ), + ) + create.add_argument( + "--gamma-n-steps-x", type=int, metavar="COUNT", + help="Gamma-surface increments along the x slip vector", + ) + create.add_argument( + "--gamma-n-steps-y", type=int, metavar="COUNT", + help="Gamma-surface increments along the y slip vector", + ) + create.add_argument( + "--gamma-closed-loop", action="store_true", default=None, + help="Derive periodic in-plane vectors for gamma_surface", + ) + create.add_argument( + "--melting-temperatures", nargs="+", type=float, metavar="K", + help="Target temperatures for melting_point", + ) + create.add_argument( + "--melting-replicas", type=int, metavar="COUNT", + help="Independent velocity-seed replicas per melting temperature", + ) + create.add_argument( + "--melting-restart-files", nargs="+", metavar="PATH", + help=( + "One existing coexistence restart per melting temperature; each " + "is staged and forwarded as restart.coexistence.start" + ), + ) create.add_argument("--flow-type", default="joint", choices=["joint", "relax", "props"], help="Workflow type (default: joint)") @@ -913,10 +1905,21 @@ def main(): "--project-id", type=int, help="Bohrium project ID (or set BOHRIUM_PROJECT_ID)", ) + create.add_argument("--sandbox", action="store_true", + help="Use OpenAPI Sandbox mode (access_key auth, no ticket)") + create.add_argument("--machine-type", + help="Override machine_type for sandbox (e.g. 'c16_m120_1 * NVIDIA L20')") create.add_argument("--scass-type", - help="Override scass_type for inner dflow containers") + help="Override scass_type for inner dflow containers (legacy Bohrium)") create.add_argument("--run-command", help="Override calculator run command") + create.add_argument( + "--lammps-image", + help=( + "Explicit LAMMPS calculator image. The image × machine pair is " + "validated before config generation." + ), + ) create.add_argument( "--vasp-image", help=( @@ -1004,12 +2007,70 @@ def main(): # ------------------------------------------------------------------------- # Validate # ------------------------------------------------------------------------- - errors = validate_config(args.backend, args.potential, args.properties) + gamma_overrides = gamma_overrides_from_args(args) + melting_overrides = melting_overrides_from_args(args) + errors = validate_config( + args.backend, + "deepmd" if args.runtime_profile else args.potential, + args.properties, + ) + if args.lammps_image and args.backend != "lammps": + errors.append("--lammps-image is only valid with --backend lammps") + runtime_profile = None + if args.runtime_profile: + try: + runtime_profile = _load_dpa4_runtime_profile(args.runtime_profile) + except (OSError, ValueError, RuntimeError) as exc: + errors.append(str(exc)) + if args.backend != "lammps": + errors.append( + f"--runtime-profile {DPA4_RUNTIME_PROFILE} requires " + "--backend lammps" + ) + if args.potential not in (None, "deepmd"): + errors.append( + f"--runtime-profile {DPA4_RUNTIME_PROFILE} requires " + "--potential deepmd" + ) + if args.model: + errors.append( + f"--runtime-profile {DPA4_RUNTIME_PROFILE} uses the " + "image-resident PT2 artifact; do not pass --model" + ) + if args.lammps_image: + errors.append( + f"--runtime-profile {DPA4_RUNTIME_PROFILE} fixes the image; " + "do not pass --lammps-image" + ) + if getattr(args, "sandbox", False) or os.environ.get( + "BOHRIUM_USE_SANDBOX" + ) == "1": + errors.append( + f"--runtime-profile {DPA4_RUNTIME_PROFILE} is not qualified " + "for the OpenAPI Sandbox backend" + ) + args.potential = "deepmd" + errors.extend( + validate_gamma_cli_options(args.properties, gamma_overrides) + ) + errors.extend( + validate_melting_cli_options(args.properties, melting_overrides) + ) if errors: for err in errors: print(f"ERROR: {err}", file=sys.stderr) sys.exit(1) + confirmed_property_configs = None + if args.property_config: + try: + confirmed_property_configs = load_confirmed_property_configs( + args.property_config, args.properties + ) + except (OSError, ValueError, json.JSONDecodeError) as exc: + print(f"ERROR: {exc}", file=sys.stderr) + sys.exit(1) + # ------------------------------------------------------------------------- # Sanitize workflow name (RFC 1123) # ------------------------------------------------------------------------- @@ -1027,7 +2088,7 @@ def main(): print(f"Auto-generated workflow name: '{workflow_name}'") # ------------------------------------------------------------------------- - # Build global.json (includes ticket conversion) + # Build global.json (includes ticket conversion or sandbox direct auth) # ------------------------------------------------------------------------- if args.backend == "vasp" and not (args.vasp_image or "").strip(): print( @@ -1039,18 +2100,36 @@ def main(): ) sys.exit(1) - print("Converting access_key to dflow ticket...") - global_config = build_global_json( - backend=args.backend, - potential=args.potential, - access_key=args.access_key, - project_id=args.project_id, - scass_type=args.scass_type, - run_command=args.run_command, - vasp_image=args.vasp_image, - ) - validate_project_id_types(global_config) - print(f"Ticket obtained: {global_config['bohrium_config']['ticket'][:8]}...") + use_sandbox = getattr(args, "sandbox", False) or os.environ.get("BOHRIUM_USE_SANDBOX") == "1" + + if use_sandbox: + print("Building OpenAPI Sandbox config (no ticket needed)...") + global_config = build_global_json_sandbox( + backend=args.backend, + potential=args.potential, + access_key=args.access_key, + project_id=args.project_id, + machine_type=getattr(args, "machine_type", None), + run_command=args.run_command, + vasp_image=args.vasp_image, + lammps_image=args.lammps_image, + ) + print(f"Sandbox config built: machine_type={global_config['machine_type']}") + else: + print("Converting access_key to dflow ticket...") + global_config = build_global_json( + backend=args.backend, + potential=args.potential, + access_key=args.access_key, + project_id=args.project_id, + scass_type=args.scass_type, + run_command=args.run_command, + vasp_image=args.vasp_image, + lammps_image=args.lammps_image, + runtime_profile=runtime_profile, + ) + validate_project_id_types(global_config) + print(f"Ticket obtained: {global_config['bohrium_config']['ticket'][:8]}...") # ------------------------------------------------------------------------- # Create output directory and copy files @@ -1065,11 +2144,28 @@ def main(): shutil.copy2(item["source"], dest_file) print(f"Copied structure to {dest_file}") - # Copy model file if specified - if args.model and os.path.exists(args.model): - model_dest = output_dir / os.path.basename(args.model) - shutil.copy2(args.model, model_dest) - print(f"Copied model to {model_dest}") + staged_model = args.model + if args.backend == "lammps" and runtime_profile is None: + try: + staged_model = stage_lammps_model(args.model, output_dir) + except (OSError, ValueError) as exc: + print(f"ERROR: {exc}", file=sys.stderr) + sys.exit(1) + + # Melting restart inputs are a temperature-indexed transport mechanism. + # Stage them in the job root and let MeltingPoint copy the matching input + # to each replica as restart.coexistence.start. FiniteTlatt never receives + # these files. + if "restart_files" in melting_overrides: + staged_restarts = [] + for index, source_text in enumerate(melting_overrides["restart_files"]): + source = Path(source_text) + dest_name = f"melting_restart_{index:06d}{source.suffix}" + destination = output_dir / dest_name + shutil.copy2(source, destination) + staged_restarts.append(dest_name) + print(f"Copied melting restart to {destination}") + melting_overrides["restart_files"] = staged_restarts # Stage VASP POTCAR (+ INCAR) into the job. Absolute host libraries such as # /share/PAW_PBE are invisible inside Bohrium/dflow containers. @@ -1103,20 +2199,77 @@ def main(): file=sys.stderr, ) + # ------------------------------------------------------------------------- + # ABACUS: auto-download pp/orb if not provided + # ------------------------------------------------------------------------- + if args.backend == "abacus" and not staged_potcars: + # Detect elements from structure files + from pymatgen.core import Structure + elements_set = set() + for item in structure_layout: + try: + struct = Structure.from_file(str(item["source"])) + for site in struct.sites: + elements_set.add(str(site.specie)) + except Exception: + pass + if elements_set: + print(f"ABACUS: auto-downloading PP/ORB for elements: {sorted(elements_set)}") + staged_potcars, orb_files = download_abacus_pp_orb( + sorted(elements_set), output_dir + ) + staged_prefix = "pp_orb" + else: + print("WARNING: Could not detect elements; provide --potcars manually") + + # ------------------------------------------------------------------------- + # ABACUS: auto-generate INPUT template if not provided + # ------------------------------------------------------------------------- + if args.backend == "abacus" and not staged_incar: + input_path = output_dir / "INPUT" + if not input_path.exists(): + input_content = """INPUT_PARAMETERS +suffix ABACUS +ntype {ntype} +ecutwfc 100 +scf_thr 1e-7 +scf_nmax 100 +basis_type lcao +calculation scf +cal_stress 1 +cal_force 1 +kspacing 0.12 +smearing_method gaussian +smearing_sigma 0.01 +mixing_type broyden +mixing_beta 0.7 +""".format(ntype=len(staged_potcars) if staged_potcars else 1) + input_path.write_text(input_content) + staged_incar = "INPUT" + print(f"Generated ABACUS INPUT template: {input_path}") + print(" NOTE: suffix=ABACUS (required by fpop); ecutwfc=100Ry; kspacing=0.12") + # ------------------------------------------------------------------------- # Build interaction + param.json (after staging so paths are job-relative) # ------------------------------------------------------------------------- interaction = build_interaction( backend=args.backend, potential=args.potential, - model=args.model, + model=staged_model, incar=staged_incar, potcar_prefix=staged_prefix, potcars=staged_potcars, orb_files=orb_files, + runtime_profile=runtime_profile, ) param_config = build_param_json( - structure_paths, interaction, args.properties, args.flow_type + structure_paths, + interaction, + args.properties, + args.flow_type, + gamma_overrides=gamma_overrides, + property_configs=confirmed_property_configs, + melting_overrides=melting_overrides, ) # ------------------------------------------------------------------------- @@ -1208,14 +2361,55 @@ def main(): print(f"Potential: {args.potential}") print(f"Structures: {len(structure_paths)} ({', '.join(structure_paths)})") print(f"Properties: {', '.join(args.properties)}") + for prop in param_config.get("properties", []): + if prop.get("type") in {"gamma", "gamma_surface"}: + if prop["type"] == "gamma": + task_count = ( + len(prop["displacement_points"]) + if prop.get("displacement_points") is not None + else int(prop.get("n_steps", 10)) + 1 + ) + else: + task_count = ( + int(prop.get("n_steps_x", prop.get("n_steps", 10))) + 1 + ) * ( + int( + prop.get( + "n_steps_y", + prop.get("n_steps_x", prop.get("n_steps", 10)), + ) + ) + 1 + ) + print( + "Gamma config: " + + json.dumps(prop, sort_keys=True) + ) + print(f"Gamma tasks: {task_count}") + elif prop.get("type") == "melting_point": + cal = prop.get("cal_setting", {}) + task_count = len(cal.get("temperature", [])) * int( + cal.get("replicas", 1) + ) + print( + "Melting config: " + + json.dumps(prop, sort_keys=True) + ) + print(f"Melting tasks: {task_count}") print(f"Flow type: {args.flow_type}") print(f"Workflow name: {workflow_name}") - print(f"scass_type: {global_config['scass_type']}") + if use_sandbox: + print(f"machine_type: {global_config.get('machine_type', 'N/A')}") + print(f"context_type: OpenAPI (Sandbox)") + else: + print(f"scass_type: {global_config.get('scass_type', 'N/A')}") print(f"Output dir: {output_dir}") print(f"\nBohrium submit command (for outer job):") print(f" cmd: {cmd}") print(f"\nOuter job image: {APEX_IMAGE}") - print(f"Outer job machine: c1_m2_cpu (recommended lightweight client)") + if use_sandbox: + print(f"Outer job machine: c2_m4_cpu (sandbox lightweight client)") + else: + print(f"Outer job machine: c1_m2_cpu (recommended lightweight client)") if __name__ == "__main__": diff --git a/apex/skills/apex-flow/scripts/validate_apex_combo.py b/apex/skills/apex-flow/scripts/validate_apex_combo.py index b0bbf1d8..7f415b4a 100644 --- a/apex/skills/apex-flow/scripts/validate_apex_combo.py +++ b/apex/skills/apex-flow/scripts/validate_apex_combo.py @@ -8,8 +8,8 @@ Usage: python validate_apex_combo.py list-combos --backend lammps --prefer gpu python validate_apex_combo.py check \\ - --image registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3 \\ - --scass "c8_m31_1 * NVIDIA T4" + --image registry.dp.tech/dptech/dp/native/prod-16664/dpa4-phonolammps:0.0.2 \\ + --scass "c16_m120_1 * NVIDIA L20" python validate_apex_combo.py recommend --backend lammps --prefer cpu """ @@ -18,10 +18,19 @@ import argparse import json import sys +from pathlib import Path from typing import Optional REGISTRY_PREFIX = "registry.dp.tech/dptech/" +DPA4_LAMMPS_IMAGE = ( + "registry.dp.tech/dptech/dp/native/prod-16664/" + "dpa4-phonolammps:0.0.2" +) +CPU_LAMMPS_IMAGE = ( + "registry.dp.tech/dptech/dp/native/prod-397637/" + "apex-flow:1.3.0.post" +) # Normalized tag → reason (always reject) BLOCKED_IMAGES = { @@ -46,12 +55,13 @@ # Recommended allow-lists (short tags or full image refs) RECOMMENDED_LAMMPS_IMAGES = { "cpu": [ + CPU_LAMMPS_IMAGE, "registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3", "registry.dp.tech/dptech/deepmd-kit:3.1.1", "registry.dp.tech/dptech/deepmd-kit:2024Q1-d23cf3e", ], "gpu": [ - "registry.dp.tech/dptech/dp/native/prod-397637/deepmd-kit-phonolammps:3.1.3", + DPA4_LAMMPS_IMAGE, "registry.dp.tech/dptech/deepmd-kit:3.1.1", "registry.dp.tech/dptech/deepmd-kit:2024Q1-d23cf3e", "registry.dp.tech/dptech/deepmd-kit:3.1.0-cuda12.1", @@ -68,6 +78,8 @@ "c8_m32_cpu", ], "lammps_gpu": [ + "c16_m120_1 * NVIDIA L20", + "c8_m32_1 * NVIDIA 4090", "c8_m31_1 * NVIDIA T4", "c4_m15_1 * NVIDIA T4", "c16_m62_1 * NVIDIA T4", @@ -79,6 +91,16 @@ # Triclinic / non-orthogonal cells: avoid deepmd-kit:3.1.1 (segfault) TRICLINIC_UNSAFE_TAGS = {"deepmd-kit:3.1.1"} +DPA4_RUNTIME_PROFILE = "dpa4-alloytongqi-t4" + + +def _dpa4_profile(*, require_published: bool) -> dict: + script_dir = Path(__file__).resolve().parent + if str(script_dir) not in sys.path: + sys.path.insert(0, str(script_dir)) + from dpa4_profile import load_dpa4_profile # noqa: WPS433 + + return load_dpa4_profile(require_published=require_published) def normalize_image(image: str) -> str: @@ -103,11 +125,72 @@ def check_combo( scass: str, *, triclinic: bool = False, + runtime_profile: str = None, + mpi_ranks: int = 1, + gpu_count: int = 1, ) -> tuple[bool, list[str]]: """ Return (ok, messages). ok=False means do not submit. """ errors: list[str] = [] + if runtime_profile is not None: + if runtime_profile != DPA4_RUNTIME_PROFILE: + return False, [ + f"unknown runtime profile {runtime_profile!r}; fail closed" + ] + try: + profile = _dpa4_profile(require_published=True) + except RuntimeError as exc: + return False, [str(exc)] + script_dir = Path(__file__).resolve().parent + if str(script_dir) not in sys.path: + sys.path.insert(0, str(script_dir)) + from dpa4_profile import ( # noqa: WPS433 + dpa4_image_name, + dpa4_recommended_combo, + ) + + expected_image = dpa4_image_name(profile) + combo = dpa4_recommended_combo(profile) + expected_scass = combo["scass_type"] + if (image or "").strip() != expected_image: + errors.append( + "DPA4 profile requires the exact qualified image " + f"{expected_image!r}; tags, other digests, and unknown images " + "are prohibited" + ) + if (scass or "").strip() != expected_scass: + compatibility = profile["machine_compatibility"] + reason = compatibility.get("prohibited_exact", {}).get(scass) + if reason is None and "V100" in (scass or ""): + reason = profile["machine_compatibility"][ + "prohibited_classes" + ]["sm70_or_older"] + if reason is None and "NVIDIA T4" in (scass or ""): + reason = ( + "this T4 SKU is unverified for the exact image; " + "the DPA4 profile fails closed" + ) + if reason is None and "NVIDIA" in (scass or ""): + reason = ( + "this GPU/PT2 architecture is unverified; the DPA4 " + "profile fails closed" + ) + if reason is None: + reason = ( + "this machine is prohibited or unknown for the T4 PT2 " + "profile" + ) + errors.append( + f"DPA4 profile rejects scass_type {scass!r}: {reason}; " + f"exact qualified value is {expected_scass!r}" + ) + if mpi_ranks != combo["mpi_ranks"]: + errors.append("DPA4 profile requires exactly one MPI rank") + if gpu_count != combo["gpu_count"]: + errors.append("DPA4 profile requires exactly one visible GPU") + return (len(errors) == 0, errors) + tag = normalize_image(image) scass = (scass or "").strip() @@ -120,10 +203,16 @@ def check_combo( errors.append( f"blocked image × accelerator '{tag}' × '{accelerator}': {reason}" ) + if tag.endswith("dpa4-phonolammps:0.0.2") and "_cpu" in scass.lower(): + errors.append( + "blocked DPA4 image × CPU machine: image 0.0.2 did not finish " + "container preparation in sequential c8_m32_cpu validation; use " + "the apex-flow CPU image or a dedicated CPU DeepMD image" + ) if triclinic and tag in TRICLINIC_UNSAFE_TAGS: errors.append( f"image '{tag}' is unsafe for triclinic/non-orthogonal cells " - "(known segfault); use deepmd-kit:3.1.3 or later" + "(known segfault); use the validated phonoLAMMPS image" ) return (len(errors) == 0, errors) @@ -131,11 +220,41 @@ def check_combo( def recommend( backend: str = "lammps", prefer: str = "cpu", + runtime_profile: str = None, ) -> dict: """Return a recommended image + scass_type for the backend.""" prefer = prefer.lower() backend = backend.lower() + if runtime_profile is not None: + if runtime_profile != DPA4_RUNTIME_PROFILE: + raise ValueError(f"Unknown runtime profile: {runtime_profile}") + profile = _dpa4_profile(require_published=True) + script_dir = Path(__file__).resolve().parent + if str(script_dir) not in sys.path: + sys.path.insert(0, str(script_dir)) + from dpa4_profile import ( # noqa: WPS433 + dpa4_image_name, + dpa4_recommended_combo, + ) + + combo = dpa4_recommended_combo(profile) + return { + "runtime_profile": runtime_profile, + "backend": "lammps", + "potential": "deepmd", + "image": dpa4_image_name(profile), + "scass_type": combo["scass_type"], + "mpi_ranks": combo["mpi_ranks"], + "gpu_count": combo["gpu_count"], + "run_command": profile["calculator"]["run_command"], + "phonolammps_run_command": profile["calculator"][ + "phonolammps_command" + ], + "outer_machine": RECOMMENDED_SCASS["outer"][0], + "evidence": combo["evidence"], + } + if backend == "lammps": key = "lammps_gpu" if prefer == "gpu" else "lammps_cpu" images = RECOMMENDED_LAMMPS_IMAGES["gpu" if prefer == "gpu" else "cpu"] @@ -171,7 +290,39 @@ def recommend( raise ValueError(f"Unknown backend: {backend}") -def list_combos(backend: str = "lammps", prefer: str = "cpu") -> dict: +def list_combos( + backend: str = "lammps", + prefer: str = "cpu", + runtime_profile: str = None, +) -> dict: + if runtime_profile is not None: + if runtime_profile != DPA4_RUNTIME_PROFILE: + raise ValueError(f"Unknown runtime profile: {runtime_profile}") + profile = _dpa4_profile(require_published=False) + compatibility = profile["machine_compatibility"] + candidate = dict(compatibility["recommended"][0]) + candidate["image_ref"] = profile["image"]["ref"] + candidate["image_digest"] = profile["image"]["digest"] + return { + "runtime_profile": runtime_profile, + "status": ( + "published" if profile["published"] else "unpublished" + ), + "qualification_status": profile.get("qualification_status"), + "recommended": ( + recommend(runtime_profile=runtime_profile) + if profile["published"] else None + ), + "candidate_after_publish": candidate, + "prohibited_exact": compatibility["prohibited_exact"], + "prohibited_classes": compatibility["prohibited_classes"], + "unverified": compatibility["unverified"], + "unknown_policy": compatibility["unknown_policy"], + "note": ( + "Pre-snapshot evidence is not a recommendation. Exact " + "ref@digest post-snapshot qualification is still required." + ), + } rec = recommend(backend, prefer) return { "recommended": rec, @@ -184,17 +335,43 @@ def list_combos(backend: str = "lammps", prefer: str = "cpu") -> dict: "notes": [ "Always validate image×scass before writing global.json / submitting.", "Outer Bohrium job should use c1_m2_cpu, never GPU.", - "For triclinic cells, avoid deepmd-kit:3.1.1; prefer 3.1.3.", + "DPA4 image 0.0.2 is validated on NVIDIA L20 and RTX 4090; L20 is the default GPU resource.", + "DPA4 image 0.0.2 is blocked on CPU machines after sequential deployment timeouts.", + "For triclinic cells, avoid deepmd-kit:3.1.1; prefer the DPA4 0.0.2 image.", + "The bundled dpa4-alloytongqi-t4 profile remains fail-closed until its immutable image ref@digest passes post-snapshot qualification.", ], } def cmd_list_combos(args: argparse.Namespace) -> int: - data = list_combos(args.backend, args.prefer) + data = list_combos(args.backend, args.prefer, args.runtime_profile) if args.format == "json": print(json.dumps(data, indent=2, ensure_ascii=False)) else: rec = data["recommended"] + if args.runtime_profile: + candidate = data["candidate_after_publish"] + print( + f"Runtime profile: {data['runtime_profile']} " + f"status={data['status']}" + ) + if rec is None: + print(" recommendation: LOCKED") + else: + print(f" image: {rec['image']}") + print(f" scass_type: {rec['scass_type']}") + print(f" candidate scass_type: {candidate['scass_type']}") + print(f" note: {data['note']}") + print("\nProhibited exact combinations:") + for key, value in data["prohibited_exact"].items(): + print(f" - {key}: {value}") + print("\nProhibited classes:") + for key, value in data["prohibited_classes"].items(): + print(f" - {key}: {value}") + print("\nUnverified:") + for value in data["unverified"]: + print(f" - {value}") + return 0 print(f"Backend: {rec['backend']} prefer={rec['prefer']}") print(f" image: {rec['image']}") print(f" scass_type: {rec['scass_type']}") @@ -212,7 +389,14 @@ def cmd_list_combos(args: argparse.Namespace) -> int: def cmd_check(args: argparse.Namespace) -> int: - ok, errors = check_combo(args.image, args.scass, triclinic=args.triclinic) + ok, errors = check_combo( + args.image, + args.scass, + triclinic=args.triclinic, + runtime_profile=args.runtime_profile, + mpi_ranks=args.mpi_ranks, + gpu_count=args.gpu_count, + ) payload = {"ok": ok, "image": full_image(args.image), "scass_type": args.scass, "errors": errors} if args.format == "json": print(json.dumps(payload, indent=2, ensure_ascii=False)) @@ -227,7 +411,7 @@ def cmd_check(args: argparse.Namespace) -> int: def cmd_recommend(args: argparse.Namespace) -> int: - rec = recommend(args.backend, args.prefer) + rec = recommend(args.backend, args.prefer, args.runtime_profile) if args.format == "json": print(json.dumps(rec, indent=2, ensure_ascii=False)) else: @@ -247,6 +431,7 @@ def main(argv: Optional[list[str]] = None) -> int: p_list.add_argument("--backend", default="lammps", choices=["lammps", "abacus", "vasp"]) p_list.add_argument("--prefer", default="cpu", choices=["cpu", "gpu"]) p_list.add_argument("--format", choices=["json", "text"], default="text") + p_list.add_argument("--runtime-profile", choices=[DPA4_RUNTIME_PROFILE]) p_list.set_defaults(func=cmd_list_combos) p_check = sub.add_parser("check", help="Check one image × scass pair") @@ -255,16 +440,24 @@ def main(argv: Optional[list[str]] = None) -> int: p_check.add_argument("--triclinic", action="store_true", help="Cell is triclinic / non-orthogonal") p_check.add_argument("--format", choices=["json", "text"], default="text") + p_check.add_argument("--runtime-profile", choices=[DPA4_RUNTIME_PROFILE]) + p_check.add_argument("--mpi-ranks", type=int, default=1) + p_check.add_argument("--gpu-count", type=int, default=1) p_check.set_defaults(func=cmd_check) p_rec = sub.add_parser("recommend", help="Print a safe default combo") p_rec.add_argument("--backend", default="lammps", choices=["lammps", "abacus", "vasp"]) p_rec.add_argument("--prefer", default="cpu", choices=["cpu", "gpu"]) p_rec.add_argument("--format", choices=["json", "text"], default="text") + p_rec.add_argument("--runtime-profile", choices=[DPA4_RUNTIME_PROFILE]) p_rec.set_defaults(func=cmd_recommend) args = parser.parse_args(argv) - return args.func(args) + try: + return args.func(args) + except (OSError, ValueError, RuntimeError) as exc: + print(f"ERROR: {exc}", file=sys.stderr) + return 1 if __name__ == "__main__": diff --git a/apex/skills/apex-flow/scripts/validate_inputs.py b/apex/skills/apex-flow/scripts/validate_inputs.py index a6a4aa11..f73f281b 100644 --- a/apex/skills/apex-flow/scripts/validate_inputs.py +++ b/apex/skills/apex-flow/scripts/validate_inputs.py @@ -9,11 +9,17 @@ """ import argparse +import copy import glob +import hashlib +import importlib.util import json +import math import os import re +import shutil import sys +import tempfile from pathlib import Path @@ -22,11 +28,11 @@ "eos", "cohesive", "elastic", "surface", "vacancy", "interstitial", "phonon", "gamma", "gamma_surface", "decohesive", "finite_t_latt", "finite_t_elastic", - "gruneisen", "annealing", + "gruneisen", "annealing", "melting_point", } # LAMMPS-only properties -LAMMPS_ONLY_PROPERTIES = {"finite_t_elastic"} +LAMMPS_ONLY_PROPERTIES = {"finite_t_elastic", "melting_point"} # Valid LAMMPS potential types VALID_LAMMPS_TYPES = { @@ -37,9 +43,290 @@ # Valid backend types VALID_BACKENDS = {"vasp", "abacus"} | VALID_LAMMPS_TYPES -# Bare PATH-based vasp_std is unreliable in Bohrium VASP images. +BUNDLED_DPA4_SHA256 = ( + "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad" +) +DPA4_RUNTIME_KIND = "dpa4_pt2" +DPA4_RUNTIME_MODEL_PATH = ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/" + "alloytongqi.t4-sm75.pt2" +) +DPA4_RUNTIME_MODEL_SHA256 = ( + "2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf" +) +DPA4_SOURCE_CHECKPOINT_PATH = ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt" +) +DPA4_SCASS_TYPE = "c4_m15_1 * NVIDIA T4" +DPA4_LAMMPS_RUN_COMMAND = "/usr/local/bin/dpa4-lmp -in in.lammps" +DPA4_PHONOLAMMPS_RUN_COMMAND = ( + "/usr/local/bin/dpa4-phonolammps {input_file} -c {poscar} " + "--dim {dim} {primitive_axes}" +) +DPA4_GROUP_SIZE = 1 +DPA4_POOL_SIZE = 1 +DPA4_DISPATCHER_COMMAND = "python3" +DPA4_JOB_TYPE = "container" +DPA4_PLATFORM = "ali" +_HEX_SHA256_RE = re.compile(r"[0-9a-f]{64}") + + +def _load_dpa4_profile(*, require_published: bool = True) -> dict: + """Load the canonical profile shared by generation and validation.""" + profile_module_path = Path(__file__).resolve().parent / "dpa4_profile.py" + spec = importlib.util.spec_from_file_location( + "apex_skill_bundled_dpa4_profile", + profile_module_path, + ) + if spec is None or spec.loader is None: + raise RuntimeError( + f"Cannot load DPA4 runtime profile helper: {profile_module_path}" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module.load_dpa4_profile(require_published=require_published) + + +def _dpa4_image_name() -> str | None: + try: + profile = _load_dpa4_profile(require_published=True) + except RuntimeError: + return None + image = profile["image"] + return f"{image['ref']}@{str(image['digest']).lower()}" + + +def _sha256_path(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _iter_effective_interactions(param_config: dict): + base = param_config.get("interaction") + properties = param_config.get("properties") or [] + if not properties: + if isinstance(base, dict): + yield "interaction", base + return + for index, prop in enumerate(properties): + if not isinstance(prop, dict): + continue + overwrite = (prop.get("cal_setting") or {}).get("overwrite_interaction") + if isinstance(overwrite, dict): + yield ( + f"properties[{index}].cal_setting.overwrite_interaction", + overwrite, + ) + elif isinstance(base, dict): + yield f"properties[{index}].interaction", base + + +def _dpa4_intent(interaction: dict) -> bool: + return ( + "deepmd_runtime" in interaction + or interaction.get("model_in_image") is True + or interaction.get("model") == DPA4_RUNTIME_MODEL_PATH + or "runtime_model_sha256" in interaction + or "source_checkpoint" in interaction + or "source_checkpoint_sha256" in interaction + ) + + +def _uses_dpa4_phonolammps(param_config: dict) -> bool: + return any( + isinstance(prop, dict) + and prop.get("type") in {"phonon", "gruneisen"} + for prop in param_config.get("properties", []) or [] + ) + + +def _nested_scass(machine: dict | None) -> str | None: + if not isinstance(machine, dict): + return None + remote_profile = machine.get("remote_profile") + if not isinstance(remote_profile, dict): + return None + input_data = remote_profile.get("input_data") + if not isinstance(input_data, dict): + return None + return input_data.get("scass_type") + + +def _nested_image_name(machine: dict | None) -> str | None: + if not isinstance(machine, dict): + return None + remote_profile = machine.get("remote_profile") + if not isinstance(remote_profile, dict): + return None + input_data = remote_profile.get("input_data") + if not isinstance(input_data, dict): + return None + return input_data.get("image_name") + + +def _validate_dpa4_global_contract( + param_config: dict, + global_config: dict | None, + expected_image: str | None, +) -> list[str]: + """Validate image, hardware, grouping, and audited entry points.""" + if not isinstance(global_config, dict): + return [ + "DPA4/PT2 requires global.json so the exact image, T4 SKU, and " + "single-rank wrappers can be verified" + ] + + errors = [] + if expected_image is not None and ( + global_config.get("lammps_image_name") != expected_image + ): + errors.append( + "DPA4 lammps_image_name must equal the published immutable image " + f"{expected_image!r}" + ) + + context_type = str(global_config.get("context_type") or "").lower() + batch_type = str(global_config.get("batch_type") or "").lower() + if "bohrium" not in context_type or "bohrium" not in batch_type: + errors.append( + "DPA4/PT2 production validation requires Bohrium context_type " + "and batch_type" + ) + if global_config.get("scass_type") != DPA4_SCASS_TYPE: + errors.append( + f"DPA4 scass_type must equal {DPA4_SCASS_TYPE!r}; other T4 SKUs, " + "CPU, and non-T4 GPUs are unverified" + ) + if global_config.get("job_type", DPA4_JOB_TYPE) != DPA4_JOB_TYPE: + errors.append( + f"DPA4 job_type must equal {DPA4_JOB_TYPE!r} so the immutable " + "container image is used" + ) + if global_config.get("platform", DPA4_PLATFORM) != DPA4_PLATFORM: + errors.append( + f"DPA4 platform must equal the qualified Bohrium value " + f"{DPA4_PLATFORM!r}" + ) + + for label in ("machine", "dispatcher_config", "resources", "task"): + value = global_config.get(label) + if value not in (None, {}): + errors.append( + f"DPA4 {label} overrides are prohibited; use the generated " + "top-level Bohrium profile without nested dispatcher/resource " + "configuration" + ) + + machine = global_config.get("machine") + if isinstance(machine, dict): + nested_scass = _nested_scass(machine) + if nested_scass is not None and nested_scass != DPA4_SCASS_TYPE: + errors.append( + "machine.remote_profile.input_data.scass_type may not " + "override the qualified DPA4 T4 SKU" + ) + for key in ("context_type", "batch_type"): + value = machine.get(key) + if value is not None and "bohrium" not in str(value).lower(): + errors.append( + f"machine.{key} may not override the DPA4 Bohrium profile" + ) + nested_image = _nested_image_name(machine) + if nested_image is not None and nested_image != expected_image: + errors.append( + "machine.remote_profile.input_data.image_name may be absent " + "or equal the published immutable DPA4 image; nested image " + "overrides are prohibited" + ) + + dispatcher = global_config.get("dispatcher_config") + if isinstance(dispatcher, dict) and dispatcher.get("json_file") not in ( + None, + "", + ): + errors.append( + "DPA4 dispatcher_config.json_file is prohibited because it can " + "inject machine or resource overrides after validation" + ) + if isinstance(dispatcher, dict) and "machine_dict" in dispatcher: + dispatcher_machine = dispatcher.get("machine_dict") + nested_scass = _nested_scass(dispatcher_machine) + if nested_scass != DPA4_SCASS_TYPE: + errors.append( + "dispatcher_config.machine_dict must explicitly retain " + f"scass_type={DPA4_SCASS_TYPE!r}" + ) + for key in ("context_type", "batch_type"): + value = ( + dispatcher_machine.get(key) + if isinstance(dispatcher_machine, dict) + else None + ) + if not isinstance(value, str) or "bohrium" not in value.lower(): + errors.append( + "dispatcher_config.machine_dict must retain Bohrium " + f"{key}" + ) + nested_image = _nested_image_name(dispatcher_machine) + if nested_image is not None and nested_image != expected_image: + errors.append( + "dispatcher_config.machine_dict remote image_name may be " + "absent or equal the published immutable DPA4 image" + ) + + effective_dispatcher_command = global_config.get( + "dispatcher_command", DPA4_DISPATCHER_COMMAND + ) + effective_remote_command = global_config.get("dispatcher_remote_command") + if isinstance(dispatcher, dict): + if "command" in dispatcher: + effective_dispatcher_command = dispatcher.get("command") + if "remote_command" in dispatcher: + effective_remote_command = dispatcher.get("remote_command") + if effective_dispatcher_command != DPA4_DISPATCHER_COMMAND: + errors.append( + "DPA4 effective dispatcher command must remain the single-process " + f"default {DPA4_DISPATCHER_COMMAND!r}" + ) + if effective_remote_command not in (None, ""): + errors.append( + "DPA4 dispatcher remote_command is prohibited because it can " + "bypass the audited one-rank wrapper" + ) + + if global_config.get("lammps_run_command") != DPA4_LAMMPS_RUN_COMMAND: + errors.append( + "DPA4 lammps_run_command must equal the audited single-rank " + f"wrapper {DPA4_LAMMPS_RUN_COMMAND!r}" + ) + if _uses_dpa4_phonolammps(param_config) and ( + global_config.get("phonolammps_run_command") + != DPA4_PHONOLAMMPS_RUN_COMMAND + ): + errors.append( + "DPA4 phonon/Gruneisen requires phonolammps_run_command=" + f"{DPA4_PHONOLAMMPS_RUN_COMMAND!r}" + ) + if type(global_config.get("group_size")) is not int or ( + global_config.get("group_size") != DPA4_GROUP_SIZE + ): + errors.append( + f"DPA4 group_size must equal {DPA4_GROUP_SIZE} for one task per T4" + ) + if type(global_config.get("pool_size")) is not int or ( + global_config.get("pool_size") != DPA4_POOL_SIZE + ): + errors.append( + f"DPA4 pool_size must equal {DPA4_POOL_SIZE}" + ) + return errors + +# Bare PATH-based VASP executables are unreliable in Bohrium VASP images. _BARE_VASP_RUN_RE = re.compile( - r"^\s*mpirun\b.*\bvasp_std\s*$", re.IGNORECASE + r"^\s*mpirun\b.*\bvasp_(?:std|gam|ncl)\s*$", re.IGNORECASE ) @@ -209,6 +496,186 @@ def _input_has_kspacing(text: str) -> bool: return bool(re.search(r"(?im)^\s*kspacing\b", text)) +def _parse_incar_values(text: str) -> dict: + """Parse simple INCAR assignments, including semicolon-separated tags.""" + values = {} + for match in re.finditer( + r"(?im)(?:^|;)\s*([A-Za-z][A-Za-z0-9_]*)\s*=\s*([^;!#\n]+)", + text, + ): + values[match.group(1).upper()] = match.group(2).strip() + return values + + +def _parse_vasp_bool(value, default=False) -> bool: + if isinstance(value, bool): + return value + if value is None: + return default + normalized = str(value).strip().strip(".").lower() + if normalized in {"true", "t", "yes", "y", "1"}: + return True + if normalized in {"false", "f", "no", "n", "0"}: + return False + raise ValueError(f"invalid VASP boolean value: {value!r}") + + +def _positive_incar_int(values: dict, key: str, errors: list): + raw = values.get(key) + if raw is None: + return None + try: + value = int(raw) + except (TypeError, ValueError): + errors.append(f"VASP: {key} must be a positive integer, got {raw!r}") + return None + if value <= 0: + errors.append(f"VASP: {key} must be a positive integer, got {value}") + return None + return value + + +def _effective_vasp_sampling(prop: dict, incar_values: dict): + cal_setting = prop.get("cal_setting") or {} + kspacing = cal_setting.get( + "kspacing", cal_setting.get("KSPACING", incar_values.get("KSPACING")) + ) + kgamma = cal_setting.get( + "kgamma", cal_setting.get("KGAMMA", incar_values.get("KGAMMA")) + ) + if kspacing is None: + return None, None + if isinstance(kspacing, list): + spacing = [float(value) for value in kspacing] + else: + spacing = float(kspacing) + return spacing, _parse_vasp_bool(kgamma, default=False) + + +def _kpoint_summary( + poscar: Path, kspacing, kgamma: bool, supercell_size=None +) -> dict: + from apex.core.calculator.lib import vasp_utils + + sampling_poscar = poscar + temporary_dir = None + if supercell_size is not None: + factors = [int(value) for value in supercell_size] + if len(factors) != 3 or any(value <= 0 for value in factors): + raise ValueError( + f"invalid VASP supercell factors: {supercell_size!r}" + ) + lines = poscar.read_text( + encoding="utf-8", errors="replace" + ).splitlines() + if len(lines) < 5: + raise ValueError(f"unreadable POSCAR: {poscar}") + for line_index, factor in zip(range(2, 5), factors): + vector = [float(value) for value in lines[line_index].split()[:3]] + lines[line_index] = " ".join( + f"{factor * value:.16g}" for value in vector + ) + temporary_dir = tempfile.TemporaryDirectory() + sampling_poscar = Path(temporary_dir.name) / "POSCAR" + sampling_poscar.write_text("\n".join(lines) + "\n", encoding="utf-8") + try: + text = vasp_utils.make_kspacing_kpoints( + str(sampling_poscar), kspacing, kgamma + ) + finally: + if temporary_dir is not None: + temporary_dir.cleanup() + lines = [line.strip() for line in text.splitlines() if line.strip()] + if len(lines) < 4: + raise RuntimeError("APEX generated an unreadable KPOINTS payload") + return { + "style": lines[2], + "grid": [int(value) for value in lines[3].split()[:3]], + } + + +def _property_supercell_size(prop: dict): + prop_type = prop.get("type") + if prop_type in {"vacancy", "interstitial"}: + return prop.get("supercell", [1, 1, 1]) + if prop_type in {"phonon", "gruneisen", "finite_t_latt", "annealing", "melting_point"}: + return prop.get("supercell_size", [2, 2, 2]) + return prop.get("supercell_size") + + +def _collect_vasp_sampling_reports( + param_config: dict, + base_dir: Path, + incar_values: dict, + gamma_reports: list, +) -> tuple: + """Collect representative grids; runtime still decides from each KPOINTS.""" + reports = list(gamma_reports) + errors = [] + structure_paths = [ + path for path in _iter_structure_poscars(param_config, base_dir) + if path.name != "STRU" + ] + + relaxation = param_config.get("relaxation") + if ( + isinstance(relaxation, dict) + and relaxation.get("req_calc", True) is not False + ): + try: + kspacing, kgamma = _effective_vasp_sampling( + relaxation, incar_values + ) + for structure_path in structure_paths: + reports.append( + { + "label": f"relaxation for {structure_path}", + "kpoints": ( + None if kspacing is None + else _kpoint_summary( + structure_path, kspacing, kgamma + ) + ), + } + ) + except Exception as exc: + errors.append( + f"VASP: cannot verify relaxation KPOINTS: {exc}" + ) + + for prop_index, prop in enumerate(param_config.get("properties", [])): + if not isinstance(prop, dict) or prop.get("req_calc", True) is False: + continue + if prop.get("type") in {"gamma", "gamma_surface"}: + continue + try: + kspacing, kgamma = _effective_vasp_sampling(prop, incar_values) + supercell_size = _property_supercell_size(prop) + for structure_path in structure_paths: + reports.append( + { + "label": ( + f"properties[{prop_index}] {prop.get('type')} " + f"for {structure_path}" + ), + "kpoints": ( + None if kspacing is None + else _kpoint_summary( + structure_path, + kspacing, + kgamma, + supercell_size=supercell_size, + ) + ), + } + ) + except Exception as exc: + errors.append( + f"VASP: cannot verify properties[{prop_index}] KPOINTS: {exc}" + ) + return reports, errors + + def validate_dft_kspacing( param_config: dict, base_dir: Path, interaction: dict ) -> tuple: @@ -288,8 +755,8 @@ def validate_vasp_run_command(global_config: dict) -> tuple: errors.append( "VASP: vasp_run_command must use the Bohrium template " '(source /opt/intel/oneapi/setvars.sh && ulimit -s unlimited && ' - "mpirun -n /opt/vasp.5.4.4/bin/vasp_std); " - "bare 'mpirun -n N vasp_std' is not allowed" + "mpirun -n /opt/vasp.5.4.4/bin/); " + "a bare PATH-based VASP executable is not allowed" ) else: if missing_setvars: @@ -304,9 +771,150 @@ def validate_vasp_run_command(global_config: dict) -> tuple: ) if missing_abs: warnings.append( - "VASP: prefer absolute binary " - "/opt/vasp.5.4.4/bin/vasp_std in vasp_run_command" + "VASP: prefer an absolute vasp_std/vasp_gam binary path " + "in vasp_run_command" + ) + + return errors, warnings + + +def _read_vasp_incar_values(interaction: dict, base_dir: Path) -> tuple: + incar_name = interaction.get("incar") or "INCAR" + incar_path = _resolve_under_base(str(incar_name), base_dir) + if not incar_path.is_file(): + return {}, incar_path + text = incar_path.read_text(encoding="utf-8", errors="replace") + return _parse_incar_values(text), incar_path + + +def _vasp_command_details(global_config: dict) -> dict: + command = ( + global_config.get("vasp_run_command") + or global_config.get("run_command") + or "" + ) + executable_match = re.search( + r"(?:^|[/\s])(?Pvasp_(?:std|gam))(?=$|[\s\"'])", + command, + re.IGNORECASE, + ) + ranks_match = re.search( + r"\bmpirun\b[^;&|\n]*?(?:-n|-np)\s+(?P\d+)", + command, + re.IGNORECASE, + ) + scass_match = re.search( + r"\bc(?P\d+)_", str(global_config.get("scass_type") or "") + ) + return { + "command": command, + "executable": ( + executable_match.group("name").lower() + if executable_match else None + ), + "ranks": int(ranks_match.group("ranks")) if ranks_match else None, + "scass_cores": ( + int(scass_match.group("cores")) if scass_match else None + ), + } + + +def _uses_bohrium_backend(global_config: dict) -> bool: + """Return whether the global configuration selects Bohrium execution.""" + context_values = ( + global_config.get("context_type"), + global_config.get("batch_type"), + ) + return ( + global_config.get("dflow_host") == "https://workflows.deepmodeling.com" + or isinstance(global_config.get("bohrium_config"), dict) + or any( + isinstance(value, str) and "bohrium" in value.lower() + for value in context_values + ) + ) + + +def validate_vasp_parallel_settings( + param_config: dict, + global_config: dict, + base_dir: Path, + gamma_reports: list, +) -> tuple: + """Validate executable, MPI, and INCAR parallel settings together.""" + errors = [] + warnings = [] + interaction = param_config.get("interaction") or {} + incar_values, incar_path = _read_vasp_incar_values(interaction, base_dir) + details = _vasp_command_details(global_config) + + executable = details["executable"] + if executable is None: + errors.append( + "VASP: cannot parse vasp_std/vasp_gam from vasp_run_command" + ) + ranks = details["ranks"] + if ranks is None: + errors.append( + "VASP: cannot parse a positive MPI rank count from " + "'mpirun -n '" + ) + + scass_cores = details["scass_cores"] + if ( + _uses_bohrium_backend(global_config) + and ranks is not None + and scass_cores is not None + and ranks != scass_cores + ): + errors.append( + f"VASP: MPI ranks ({ranks}) must match Bohrium CPU count " + f"from scass_type ({scass_cores})" + ) + + ncore = _positive_incar_int(incar_values, "NCORE", errors) + npar = _positive_incar_int(incar_values, "NPAR", errors) + kpar = _positive_incar_int(incar_values, "KPAR", errors) + if "NCORE" not in incar_values: + warnings.append( + f"VASP: NCORE is not set in '{incar_path}'; choose it explicitly " + "after considering MPI ranks and KPAR" + ) + if ncore is not None and npar is not None: + errors.append("VASP: do not set NCORE and NPAR at the same time") + + effective_kpar = 1 if kpar is None else kpar + if ranks is not None: + if ranks % effective_kpar: + errors.append( + f"VASP: KPAR={effective_kpar} must divide MPI ranks={ranks}" ) + elif ncore is not None: + ranks_per_kgroup = ranks // effective_kpar + if ranks_per_kgroup % ncore: + errors.append( + f"VASP: NCORE={ncore} must divide " + f"MPI ranks/KPAR={ranks_per_kgroup}" + ) + + sampling_reports, sampling_errors = _collect_vasp_sampling_reports( + param_config, base_dir, incar_values, gamma_reports + ) + errors.extend(sampling_errors) + gamma_only_reports = [ + report for report in sampling_reports + if report.get("kpoints") + and str(report["kpoints"].get("style", "")).lower() == "gamma" + and report["kpoints"].get("grid") == [1, 1, 1] + ] + if gamma_only_reports and effective_kpar != 1: + labels = ", ".join( + report.get("label", "VASP task") for report in gamma_only_reports + ) + errors.append( + "VASP: Gamma-centered 1x1x1 tasks are executed with vasp_gam " + f"and require KPAR=1; detected KPAR={effective_kpar} for {labels}" + ) return errors, warnings @@ -316,32 +924,89 @@ def validate_global(global_config: dict) -> list: errors = [] warnings = [] - # generate_config.py writes the current, top-level APEX schema. Keep - # accepting the older nested-machine schema for existing user configs. - is_current_schema = any( - key in global_config - for key in ("dflow_host", "context_type", "bohrium_config", "program_id") - ) + context_type = str(global_config.get("context_type", "")).lower() + batch_type = str(global_config.get("batch_type", "")).lower() + if "openapi" in {context_type, batch_type}: + for key in ("batch_type", "context_type", "machine_type", "image_address"): + if not global_config.get(key): + errors.append(f"OpenAPI: missing non-empty '{key}' in global.json") + + project_id = global_config.get("project_id") + if type(project_id) is not int or project_id <= 0: + errors.append( + "OpenAPI: project_id must be a positive unquoted JSON integer" + ) + + access_key = global_config.get("access_key") + if not isinstance(access_key, str) or not access_key.strip(): + errors.append("OpenAPI: missing non-empty access_key") + + bohrium_config = global_config.get("bohrium_config") + if not isinstance(bohrium_config, dict): + errors.append("OpenAPI: missing bohrium_config object") + else: + nested_id = bohrium_config.get("project_id") + if type(nested_id) is not int or nested_id <= 0: + errors.append( + "OpenAPI: bohrium_config.project_id must be a positive " + "JSON integer" + ) + elif nested_id != project_id: + errors.append("OpenAPI: project_id values must match") + nested_key = bohrium_config.get("access_key") + if not isinstance(nested_key, str) or not nested_key.strip(): + errors.append("OpenAPI: missing bohrium_config.access_key") + + dflow_config = global_config.get("dflow_config") + if not isinstance(dflow_config, dict): + errors.append("OpenAPI: missing dflow_config object") + else: + if dflow_config.get("namespace") != "dflow": + errors.append("OpenAPI: dflow_config.namespace must be 'dflow'") + if dflow_config.get("token") != "": + errors.append("OpenAPI: dflow_config.token must be an empty string") + + lammps_image = global_config.get("lammps_image_name") + image_address = global_config.get("image_address") + if lammps_image and image_address != lammps_image: + errors.append("OpenAPI: image_address and lammps_image_name must match") + if not global_config.get("dispatcher_image"): + errors.append("OpenAPI: missing dispatcher_image") + + run_commands = ( + "lammps_run_command", "abacus_run_command", "vasp_run_command" + ) + if not any(global_config.get(key) for key in run_commands): + errors.append("OpenAPI: missing calculator run command in global.json") + + rc_errors, rc_warnings = validate_vasp_run_command(global_config) + errors.extend(rc_errors) + warnings.extend(rc_warnings) + return errors, warnings + + machine = global_config.get("machine") + machine = machine if isinstance(machine, dict) else {} + is_bohrium = _uses_bohrium_backend(global_config) - if is_current_schema: + if is_bohrium: for key in ("batch_type", "context_type"): if not global_config.get(key): errors.append(f"Missing '{key}' in global.json") program_id = global_config.get("program_id") - if not isinstance(program_id, int) or isinstance(program_id, bool): + if program_id is not None and ( + not isinstance(program_id, int) or isinstance(program_id, bool) + ): errors.append( "'program_id' must be an unquoted JSON integer, not a string; " - "generate global.json with " - "generate_config.py" + "use `apex account` or regenerate global.json" ) - elif program_id <= 0: + elif isinstance(program_id, int) and program_id <= 0: errors.append("'program_id' must be a positive integer") bohrium_config = global_config.get("bohrium_config") - if not isinstance(bohrium_config, dict): - errors.append("Missing 'bohrium_config' section in global.json") - else: + ticket_mode = isinstance(bohrium_config, dict) + if ticket_mode: project_id = bohrium_config.get("project_id") if not isinstance(project_id, int) or isinstance(project_id, bool): errors.append( @@ -364,22 +1029,29 @@ def validate_global(global_config: dict) -> list: "Missing non-empty 'bohrium_config.ticket'; regenerate " "global.json with generate_config.py" ) + if program_id is None: + errors.append( + "Ticket-based Bohrium config requires integer 'program_id'" + ) + else: + warnings.append( + "Bohrium direct-submit profile uses credentials from " + "`apex account`; verify them with `apex account --show`" + ) - machine = global_config.get("machine") - if isinstance(machine, dict): - remote_profile = machine.get("remote_profile") - if isinstance(remote_profile, dict) and "program_id" in remote_profile: - nested_id = remote_profile["program_id"] - if not isinstance(nested_id, int) or isinstance(nested_id, bool): - errors.append( - "'machine.remote_profile.program_id' must be a JSON " - "integer, not a quoted string" - ) - elif nested_id != program_id: - errors.append( - "'machine.remote_profile.program_id' must match " - "'program_id'" - ) + remote_profile = machine.get("remote_profile") + if isinstance(remote_profile, dict) and "program_id" in remote_profile: + nested_id = remote_profile["program_id"] + if not isinstance(nested_id, int) or isinstance(nested_id, bool): + errors.append( + "'machine.remote_profile.program_id' must be a JSON " + "integer, not a quoted string" + ) + elif program_id is not None and nested_id != program_id: + errors.append( + "'machine.remote_profile.program_id' must match " + "'program_id'" + ) if not global_config.get("scass_type"): errors.append("Missing 'scass_type' in global.json") @@ -398,20 +1070,34 @@ def validate_global(global_config: dict) -> list: errors.extend(rc_errors) warnings.extend(rc_warnings) else: - # Legacy nested-machine schema. - if "machine" not in global_config: - errors.append("Missing 'machine' section in global.json") - else: - machine = global_config["machine"] - if "batch_type" not in machine: + # Local debug and DPDispatcher cluster modes do not use Bohrium auth. + top_batch = global_config.get("batch_type") + nested_batch = machine.get("batch_type") + batch_type = nested_batch or top_batch + if not batch_type: + if not machine: + errors.append( + "Missing local execution configuration: set top-level " + "'batch_type' or provide 'machine.batch_type'" + ) + else: errors.append("Missing 'machine.batch_type'") - if "resources" not in global_config: + if ( + isinstance(batch_type, str) + and batch_type.lower() not in {"shell"} + and "resources" not in global_config + ): warnings.append("No 'resources' section - will use defaults") if "run_command" not in global_config: errors.append("Missing 'run_command' in global.json") + if re.search(r"<[^>]+>", json.dumps(global_config)): + errors.append( + "Replace all <...> placeholders in local/cluster global.json" + ) + return errors, warnings @@ -436,6 +1122,20 @@ def validate_interaction(interaction: dict) -> list: errors.append(f"LAMMPS potential '{int_type}' requires 'model' field") if "type_map" not in interaction: errors.append(f"LAMMPS potential '{int_type}' requires 'type_map' field") + if "model_in_image" in interaction and not isinstance( + interaction["model_in_image"], bool + ): + errors.append("interaction.model_in_image must be a boolean") + runtime = interaction.get("deepmd_runtime") + if runtime is not None and runtime != DPA4_RUNTIME_KIND: + errors.append( + f"interaction.deepmd_runtime must equal {DPA4_RUNTIME_KIND!r}" + ) + if ( + runtime == DPA4_RUNTIME_KIND + and int_type != "deepmd" + ): + errors.append("deepmd_runtime=dpa4_pt2 requires interaction.type=deepmd") # ABACUS-specific checks elif int_type == "abacus": @@ -454,7 +1154,250 @@ def validate_interaction(interaction: dict) -> list: return errors, warnings -def validate_properties(properties: list, interaction_type: str) -> list: +def validate_bundled_dpa4_runtime( + param_config: dict, + global_config: dict | None, + base_dir: Path, +) -> tuple[list[str], list[str]]: + """Require the exact image-resident, T4-only DPA4 production contract.""" + interactions = list(_iter_effective_interactions(param_config)) + if not interactions: + return [], [] + + errors = [] + kinds = [] + expected_fields = { + "type": "deepmd", + "deepmd_runtime": DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint": DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": BUNDLED_DPA4_SHA256, + } + + for label, interaction in interactions: + local_checkpoint = False + model = interaction.get("model") + if isinstance(model, str) and interaction.get("model_in_image") is not True: + model_path = _resolve_under_base(model, base_dir) + try: + local_checkpoint = ( + model_path.is_file() + and _sha256_path(model_path) == BUNDLED_DPA4_SHA256 + ) + except OSError: + local_checkpoint = False + + if local_checkpoint: + kinds.append(DPA4_RUNTIME_KIND) + errors.append( + f"{label}.model is the bundled DPA4 source checkpoint; " + f"LAMMPS must use image-resident {DPA4_RUNTIME_MODEL_PATH!r}, " + "never model.pt" + ) + continue + + if not _dpa4_intent(interaction): + kinds.append("legacy") + continue + + kinds.append(DPA4_RUNTIME_KIND) + for key in ("runtime_model_sha256", "source_checkpoint_sha256"): + declared = interaction.get(key) + if not isinstance(declared, str) or not _HEX_SHA256_RE.fullmatch(declared): + errors.append( + f"{label}.{key} must be a lowercase SHA-256 hex digest" + ) + for key, expected in expected_fields.items(): + if interaction.get(key) != expected: + errors.append( + f"{label}.{key} must equal {expected!r} for the bundled " + "DPA4 T4/PT2 runtime" + ) + type_map = interaction.get("type_map") + valid_type_map = type_map == "auto" or ( + isinstance(type_map, dict) + and bool(type_map) + and all( + isinstance(symbol, str) + and bool(symbol.strip()) + and type(index) is int + and index >= 0 + for symbol, index in type_map.items() + ) + and set(type_map.values()) == set(range(len(type_map))) + ) + if not valid_type_map: + errors.append( + f"{label}.type_map must be 'auto' before CLI expansion or a " + "non-empty contiguous element-to-index mapping afterwards" + ) + + kind_set = set(kinds) + if DPA4_RUNTIME_KIND in kind_set and "legacy" in kind_set: + errors.append( + "A single APEX parameter set cannot mix legacy LAMMPS interactions " + "with bundled DPA4/PT2 (including overwrite_interaction)" + ) + + if DPA4_RUNTIME_KIND in kind_set: + base_interaction = param_config.get("interaction") + if ( + not isinstance(base_interaction, dict) + or base_interaction.get("type") not in VALID_LAMMPS_TYPES + ): + errors.append( + "DPA4/PT2 overwrite_interaction requires a LAMMPS base " + "interaction; VASP/ABACUS base calculators cannot execute the " + "DPA4 runtime" + ) + expected_image = _dpa4_image_name() + if expected_image is None: + errors.append( + "DPA4 image identity is not finalized: replace " + "__DPA4_IMAGE_REF__ and __DPA4_IMAGE_DIGEST__ before submission" + ) + errors.extend( + _validate_dpa4_global_contract( + param_config, + global_config, + expected_image, + ) + ) + + return errors, [] + + +def validate_gamma_settings(prop: dict, prefix: str) -> tuple: + """Mirror the public validation rules in ``gamma_slab.py``.""" + errors = [] + warnings = [] + parent_lattice = prop.get("parent_lattice") + if parent_lattice is not None and ( + not isinstance(parent_lattice, str) + or parent_lattice.strip().lower() not in {"bcc", "fcc", "hcp"} + ): + errors.append( + f"{prefix}: parent_lattice must be one of bcc, fcc, or hcp" + ) + supercell = prop.get("supercell_size", [1, 1, 5]) + if not isinstance(supercell, (list, tuple)) or len(supercell) != 3: + errors.append(f"{prefix}: gamma supercell_size must contain 3 values") + else: + for index, value in enumerate(supercell[:2]): + if ( + not isinstance(value, int) + or isinstance(value, bool) + or value <= 0 + ): + errors.append( + f"{prefix}: gamma supercell_size[{index}] " + "must be a positive integer" + ) + plane_count = supercell[2] + if ( + isinstance(plane_count, bool) + or not isinstance(plane_count, (int, float)) + or not math.isfinite(float(plane_count)) + or plane_count <= 0 + ): + errors.append( + f"{prefix}: gamma supercell_size[2] must be a positive " + "finite number of Miller-plane spacings" + ) + + min_height = prop.get("min_slab_height") + if min_height is None: + warnings.append( + f"{prefix}: min_slab_height is not set; confirm the generated " + "material thickness before submission" + ) + elif ( + isinstance(min_height, bool) + or not isinstance(min_height, (int, float)) + or not math.isfinite(float(min_height)) + or min_height <= 0 + ): + errors.append(f"{prefix}: min_slab_height must be a positive finite number") + + max_atoms = prop.get("max_atoms") + if max_atoms is None: + warnings.append( + f"{prefix}: max_atoms is not set; confirm the generated atom " + "count before submission" + ) + elif ( + not isinstance(max_atoms, int) + or isinstance(max_atoms, bool) + or max_atoms <= 0 + ): + errors.append(f"{prefix}: max_atoms must be a positive integer") + + min_distance = prop.get("min_distance", 0.2) + if ( + isinstance(min_distance, bool) + or not isinstance(min_distance, (int, float)) + or not math.isfinite(float(min_distance)) + or min_distance < 0 + ): + errors.append(f"{prefix}: min_distance must be non-negative and finite") + + vacuum_size = prop.get("vacuum_size", 20) + if ( + isinstance(vacuum_size, bool) + or not isinstance(vacuum_size, (int, float)) + or not math.isfinite(float(vacuum_size)) + or vacuum_size < 0 + ): + errors.append(f"{prefix}: vacuum_size must be non-negative and finite") + + require_orthogonal = prop.get( + "require_orthogonal_cell", prop.get("orthogonalize_cell", False) + ) + if not isinstance(require_orthogonal, bool): + errors.append(f"{prefix}: require_orthogonal_cell must be a boolean") + if ( + "require_orthogonal_cell" in prop + and "orthogonalize_cell" in prop + and prop["require_orthogonal_cell"] != prop["orthogonalize_cell"] + ): + errors.append( + f"{prefix}: require_orthogonal_cell and orthogonalize_cell disagree" + ) + + if prop.get("type") == "gamma": + n_steps = prop.get("n_steps", 10) + if ( + not isinstance(n_steps, int) + or isinstance(n_steps, bool) + or n_steps <= 0 + ): + errors.append(f"{prefix}: n_steps must be a positive integer") + displacement_points = prop.get("displacement_points") + if displacement_points is not None and ( + not isinstance(displacement_points, list) + or not displacement_points + or any( + isinstance(value, bool) + or not isinstance(value, (int, float)) + or not math.isfinite(float(value)) + or not 0.0 <= float(value) <= 1.0 + for value in displacement_points + ) + or len(set(displacement_points)) != len(displacement_points) + or 0.0 not in displacement_points + ): + errors.append( + f"{prefix}: displacement_points must include 0 and contain " + "unique finite values in [0, 1]" + ) + return errors, warnings + + +def validate_properties( + properties: list, interaction_type: str, base_dir: Path = None +) -> list: """Validate property configurations.""" errors = [] warnings = [] @@ -475,6 +1418,18 @@ def validate_properties(properties: list, interaction_type: str) -> list: errors.append(f"{prefix}: unknown property type '{prop_type}'") continue + cal_setting = prop.get("cal_setting", {}) + if ( + prop_type != "melting_point" + and isinstance(cal_setting, dict) + and "restart_files" in cal_setting + ): + errors.append( + f"{prefix}: cal_setting.restart_files is supported only by " + "melting_point; finite_t_latt and other properties do not " + "forward restart.coexistence.start" + ) + # Check LAMMPS-only constraint if prop_type in LAMMPS_ONLY_PROPERTIES: if interaction_type in ("vasp", "abacus"): @@ -541,18 +1496,168 @@ def validate_properties(properties: list, interaction_type: str) -> list: f"{prefix}: finite_t_elastic only supports method='paired_langevin'" ) + elif prop_type == "melting_point": + method = str(prop.get("method", "two_phase")).lower().replace("-", "_") + if method not in { + "two_phase", "coexistence", "two_phase_coexistence", + "direct_coexistence", + }: + errors.append( + f"{prefix}: melting_point only supports method='two_phase'" + ) + supercell = prop.get("supercell_size", [1, 1, 1]) + if ( + not isinstance(supercell, (list, tuple)) + or len(supercell) != 3 + or any( + not isinstance(value, int) + or isinstance(value, bool) + or value <= 0 + for value in supercell + ) + ): + errors.append( + f"{prefix}: supercell_size must contain 3 positive integers" + ) + cal = prop.get("cal_setting", {}) + temperatures = cal.get("temperature") + if not isinstance(temperatures, list) or not temperatures: + errors.append( + f"{prefix}: cal_setting.temperature must be a non-empty list" + ) + elif any( + isinstance(value, bool) + or not isinstance(value, (int, float)) + or not math.isfinite(float(value)) + or value <= 0 + for value in temperatures + ): + errors.append(f"{prefix}: all temperatures must be positive finite numbers") + axis = cal.get("interface_axis", "z") + if axis not in {"x", "y", "z"}: + errors.append(f"{prefix}: interface_axis must be x, y, or z") + liquid_fraction = cal.get("liquid_fraction", 0.5) + if ( + isinstance(liquid_fraction, bool) + or not isinstance(liquid_fraction, (int, float)) + or not 0.1 <= float(liquid_fraction) <= 0.9 + ): + errors.append( + f"{prefix}: liquid_fraction must be between 0.1 and 0.9" + ) + replicas = cal.get("replicas", 1) + if ( + not isinstance(replicas, int) + or isinstance(replicas, bool) + or replicas < 1 + ): + errors.append(f"{prefix}: replicas must be a positive integer") + restart_files = cal.get("restart_files") + if restart_files is not None: + if not isinstance(restart_files, list): + errors.append( + f"{prefix}: restart_files must be a list with one " + "entry per temperature" + ) + elif isinstance(temperatures, list) and len(restart_files) != len( + temperatures + ): + errors.append( + f"{prefix}: restart_files must contain exactly one " + "entry per temperature" + ) + else: + invalid_paths = [ + path for path in restart_files + if not isinstance(path, str) or not path.strip() + ] + if invalid_paths: + errors.append( + f"{prefix}: restart_files entries must be " + "non-empty paths" + ) + elif base_dir is not None: + missing = [ + path for path in restart_files + if not ( + Path(path) + if Path(path).is_absolute() + else base_dir / path + ).is_file() + ] + if missing: + errors.append( + f"{prefix}: melting restart file(s) not found: " + + ", ".join(missing) + ) + for key in ( + "premelt_steps", "conditioning_steps", "production_steps", + "dump_step", "thermo_step", "restart_interval", + ): + value = cal.get(key, { + "premelt_steps": 5000, + "conditioning_steps": 5000, + "production_steps": 100000, + "dump_step": 100, + "thermo_step": 100, + "restart_interval": 10000, + }[key]) + if ( + not isinstance(value, int) + or isinstance(value, bool) + or value <= 0 + ): + errors.append( + f"{prefix}: melting_point {key} must be a positive integer" + ) + if prop_type in {"gamma", "gamma_surface"}: + gamma_errors, gamma_warnings = validate_gamma_settings(prop, prefix) + errors.extend(gamma_errors) + warnings.extend(gamma_warnings) plane = prop.get("plane_miller") direction = prop.get("slip_direction") if not plane or not direction: errors.append( f"{prefix}: {prop_type} requires plane_miller and slip_direction" ) + elif not isinstance(plane, (list, tuple)) or not isinstance( + direction, (list, tuple) + ): + errors.append( + f"{prefix}: plane_miller and slip_direction must be sequences" + ) elif len(plane) != len(direction): errors.append( f"{prefix}: plane_miller and slip_direction dimensions differ" ) - elif sum(p * d for p, d in zip(plane, direction)) != 0: + elif not 3 <= len(plane) <= 4: + errors.append( + f"{prefix}: plane_miller and slip_direction require " + "3 or 4 components" + ) + elif not all( + isinstance(value, (int, float)) + and not isinstance(value, bool) + and math.isfinite(float(value)) + for value in list(plane) + list(direction) + ): + errors.append( + f"{prefix}: plane_miller and slip_direction must be " + "finite numeric vectors" + ) + elif not any(float(value) != 0 for value in plane) or not any( + float(value) != 0 for value in direction + ): + errors.append( + f"{prefix}: plane_miller and slip_direction must be " + "non-zero vectors" + ) + elif not math.isclose( + sum(p * d for p, d in zip(plane, direction)), + 0.0, + abs_tol=1.0e-10, + ): errors.append( f"{prefix}: slip_direction must lie on plane_miller" ) @@ -577,6 +1682,207 @@ def validate_properties(properties: list, interaction_type: str) -> list: return errors, warnings +def _gamma_task_count(prop: dict) -> int: + if prop.get("type") == "gamma": + if prop.get("displacement_points") is not None: + return len(prop["displacement_points"]) + return int(prop.get("n_steps", 10)) + 1 + n_steps_x = int(prop.get("n_steps_x", prop.get("n_steps", 10))) + n_steps_y = int(prop.get("n_steps_y", n_steps_x)) + return (n_steps_x + 1) * (n_steps_y + 1) + + +def preflight_gamma_structures( + param_config: dict, base_dir: Path +) -> tuple: + """Generate one representative slab per structure and Gamma property.""" + reports = [] + errors = [] + warnings = [] + properties = [ + (index, prop) + for index, prop in enumerate(param_config.get("properties", [])) + if isinstance(prop, dict) + and prop.get("type") in {"gamma", "gamma_surface"} + ] + if not properties: + return reports, errors, warnings + + structure_paths = _iter_structure_poscars(param_config, base_dir) + if not structure_paths: + errors.append( + "Gamma preflight: no local POSCAR/CONTCAR structure was resolved" + ) + return reports, errors, warnings + + interaction = param_config.get("interaction") or {} + if interaction.get("type") == "abacus": + warnings.append( + "Gamma preflight: representative STRU generation is not performed; " + "run apex preview before submission" + ) + return reports, errors, warnings + + incar_values = {} + if interaction.get("type") == "vasp": + incar_values, _ = _read_vasp_incar_values(interaction, base_dir) + + from pymatgen.core.structure import Structure + from apex.core.property.Gamma import Gamma + from apex.core.property.GammaSurface import GammaSurface + + for structure_path in structure_paths: + if structure_path.name == "STRU": + warnings.append( + f"Gamma preflight: skipped unsupported structure file " + f"'{structure_path}'" + ) + continue + try: + parent_atom_count = len(Structure.from_file(structure_path)) + except Exception as exc: + errors.append( + f"Gamma preflight: cannot read '{structure_path}': {exc}" + ) + continue + + for prop_index, original_prop in properties: + prop = copy.deepcopy(original_prop) + prop_type = prop["type"] + label = ( + f"properties[{prop_index}] {prop_type} " + f"for {structure_path}" + ) + prop["reproduce"] = False + for key in ( + "init_from_suffix", + "output_suffix", + "init_data_path", + "start_confs_path", + ): + prop.pop(key, None) + if prop_type == "gamma": + prop["n_steps"] = 1 + if prop.get("displacement_points") is not None: + prop["displacement_points"] = [0.0] + else: + prop["n_steps_x"] = 1 + prop["n_steps_y"] = 1 + + previous_cwd = Path.cwd() + try: + with tempfile.TemporaryDirectory( + prefix="apex-gamma-preflight-" + ) as temp_name: + root = Path(temp_name) + equi = root / "relaxation" / "relax_task" + work = root / prop_type + equi.mkdir(parents=True) + shutil.copy2(structure_path, equi / "CONTCAR") + (equi / "result.json").write_text( + "{}\n", encoding="utf-8" + ) + + if prop_type == "gamma": + task_paths = Gamma( + prop, interaction + ).make_confs(str(work), str(equi)) + else: + task_paths = GammaSurface( + prop, interaction + ).make_confs( + str(work), + str(equi), + require_relaxation_result=False, + ) + if not task_paths: + raise RuntimeError("no representative task was generated") + + metadata_path = work / "slab_generation.json" + metadata = json.loads( + metadata_path.read_text(encoding="utf-8") + ) + task_poscar = Path(task_paths[0]) / "POSCAR" + kpoints = None + if interaction.get("type") == "vasp": + kspacing, kgamma = _effective_vasp_sampling( + original_prop, incar_values + ) + if kspacing is not None: + kpoints = _kpoint_summary( + task_poscar, kspacing, kgamma + ) + + min_height = original_prop.get("min_slab_height") + if ( + min_height is not None + and metadata["slab_height"] + 1.0e-10 + < float(min_height) + ): + raise RuntimeError( + f"generated material thickness " + f"{metadata['slab_height']:.6f} A is below " + f"min_slab_height={float(min_height):.6f} A" + ) + + reports.append( + { + "label": label, + "property_index": prop_index, + "property_type": prop_type, + "structure": str(structure_path), + "parent_atom_count": parent_atom_count, + "atom_count": metadata["atom_count"], + "slab_height": metadata["slab_height"], + "effective_plane_spacings": metadata[ + "effective_plane_spacings" + ], + "oriented_cell_repeats": metadata[ + "oriented_cell_repeats" + ], + "minimum_pair_distance": metadata[ + "minimum_pair_distance" + ], + "expected_task_count": _gamma_task_count( + original_prop + ), + "kpoints": kpoints, + } + ) + except Exception as exc: + errors.append(f"Gamma preflight failed for {label}: {exc}") + finally: + os.chdir(previous_cwd) + + if not reports and not errors: + errors.append("Gamma preflight did not produce a representative slab") + return reports, errors, warnings + + +def print_gamma_preflight_reports(reports: list): + if not reports: + return + print("Gamma preflight:") + for report in reports: + kpoints = report.get("kpoints") + kpoint_text = ( + "not available" + if not kpoints + else f"{kpoints['style']} {kpoints['grid']}" + ) + print(f" {report['label']}") + print( + " atoms(parent/final)=" + f"{report['parent_atom_count']}/{report['atom_count']}; " + f"thickness={report['slab_height']:.6f} A; " + f"plane spacings={report['effective_plane_spacings']:.8g}; " + f"layers={report['oriented_cell_repeats']}; " + f"minimum distance={report['minimum_pair_distance']:.6f} A; " + f"expected tasks={report['expected_task_count']}; " + f"KPOINTS={kpoint_text}" + ) + + def validate_structures(param_config: dict, base_dir: Path) -> list: """Validate structure paths.""" errors = [] @@ -611,6 +1917,7 @@ def main(): all_errors = [] all_warnings = [] global_config = None + gamma_reports = [] # Load param.json param_path = Path(args.param) @@ -646,7 +1953,28 @@ def main(): # Validate properties properties = param_config.get("properties", []) interaction_type = interaction.get("type", "unknown") - errors, warnings = validate_properties(properties, interaction_type) + # Relaxation-only parameter files intentionally omit ``properties`` (or + # may leave it empty). Keep rejecting an empty property-only input, but + # do not turn a valid relaxation flow into a validation failure. + if properties or not isinstance(param_config.get("relaxation"), dict): + errors, warnings = validate_properties( + properties, interaction_type, base_dir=base_dir + ) + else: + errors, warnings = [], [] + all_errors.extend(errors) + all_warnings.extend(warnings) + + for label, overwrite in _iter_effective_interactions(param_config): + if label == "interaction" or label.endswith(".interaction"): + continue + errors, warnings = validate_interaction(overwrite) + all_errors.extend(f"{label}: {error}" for error in errors) + all_warnings.extend(f"{label}: {warning}" for warning in warnings) + + errors, warnings = validate_bundled_dpa4_runtime( + param_config, global_config, base_dir + ) all_errors.extend(errors) all_warnings.extend(warnings) @@ -655,6 +1983,19 @@ def main(): all_errors.extend(errors) all_warnings.extend(warnings) + # Build representative Gamma slabs before any remote submission. + if any( + isinstance(prop, dict) + and prop.get("type") in {"gamma", "gamma_surface"} + for prop in properties + ): + reports, errors, warnings = preflight_gamma_structures( + param_config, base_dir + ) + gamma_reports.extend(reports) + all_errors.extend(errors) + all_warnings.extend(warnings) + # VASP POTCAR filesystem check (prefix + per-element files) if interaction.get("type") == "vasp": required_elements = collect_structure_elements(param_config, base_dir) @@ -677,6 +2018,13 @@ def main(): all_errors.extend(errors) all_warnings.extend(warnings) + if interaction.get("type") == "vasp" and global_config is not None: + errors, warnings = validate_vasp_parallel_settings( + param_config, global_config, base_dir, gamma_reports + ) + all_errors.extend(errors) + all_warnings.extend(warnings) + # VASP image is license-gated: require an explicit user-provided image. if interaction.get("type") == "vasp" and global_config is not None: image = global_config.get("vasp_image_name") @@ -696,6 +2044,7 @@ def main(): ) # Report + print_gamma_preflight_reports(gamma_reports) if all_warnings: print("WARNINGS:", file=sys.stderr) for w in all_warnings: @@ -731,6 +2080,22 @@ def main(): f" bohrium_config.project_id={project_id!r} " f"type={type(project_id).__name__}" ) + elif global_config and str( + global_config.get("context_type", "") + ).lower() == "openapi": + project_id = global_config.get("project_id") + nested_id = global_config.get( + "bohrium_config", {} + ).get("project_id") + print(" Hard OpenAPI project ID type check:") + print( + f" project_id={project_id!r} " + f"type={type(project_id).__name__}" + ) + print( + f" bohrium_config.project_id={nested_id!r} " + f"type={type(nested_id).__name__}" + ) if __name__ == "__main__": diff --git a/apex/skills/apex-flow/variants/local/SKILL.md b/apex/skills/apex-flow/variants/local/SKILL.md new file mode 100644 index 00000000..2c4ade9a --- /dev/null +++ b/apex/skills/apex-flow/variants/local/SKILL.md @@ -0,0 +1,137 @@ +--- +name: apex-flow +description: Run locally installed APEX relaxation and 15-property workflows through Bohrium direct, local debug, or Slurm/PBS with VASP, ABACUS, or LAMMPS. Includes parent-aware Gamma line/surface and diagnostic views, finite-temperature and two-phase melting-point workflows, restart transport, RSS/high-entropy generation, monitoring, reporting, and result extraction. +--- + +# APEX Flow — Local Agent Edition + +Use APEX for automated relaxation and batch property workflows. This edition is +installed on the same machine as the Agent and must contain +`reference/execution-profile.md`, selected during installation. + +## Start Here + +1. Read `reference/execution-profile.md`. If it is missing, stop and ask the + user to reinstall with `apex skill`. +2. Follow that profile's authentication, global configuration, and submit + command. Do not mix commands from another profile. +3. Confirm the calculator backend: LAMMPS, ABACUS, or VASP. +4. Read the input structure and report formula, atom count, lattice lengths, + and whether it is already a supercell. +5. Present the complete property JSON and wait for user approval. + +## Execution Profiles + +- **Bohrium cloud (direct)**: use `apex account`; run `apex submit` directly + from this machine. A saved AccessKey is exchanged by dflow for a short-lived + ticket, but no ticket is serialized into `global.json` and no outer Bohrium + submission job is created. +- **local**: use `apex submit -d` with Local/Shell configuration. No Bohrium + login, ticket, cloud image, or Argo server is required. +- **local cluster**: use `apex submit -d` and DPDispatcher with Local + + Slurm/PBS from a login node. Confirm scheduler resources and modules first. + +The exact installed choice is in `reference/execution-profile.md`. + +## Workflow + +1. Prepare structures, calculator inputs, models/potentials, `param.json`, and + the profile-specific `global.json`. +2. Validate inside the job directory: + + ```bash + python /scripts/validate_inputs.py \ + --param param.json --global global.json + ``` + + For Gamma properties, read the printed representative-slab report and stop + on atom-limit, thickness, distance, or KPOINTS errors. Missing + `min_slab_height`/`max_atoms` is a compatibility warning that requires + manual confirmation. +3. For `gamma_surface`, run `apex preview param.json` and stop if stderr + contains `Generated Gamma surface contains overlapping atoms.` + The default `--gif-view auto` writes both Gamma/GammaSurface projections; + human-requested diagnostics can select `default`, `slip-plane`, `parent-bc`, + or `both`. Agents still decide safety from text validation and the overlap + warning, not by opening the GIF. +4. Run the submit command from the installed execution profile without `-s` + unless the user explicitly requests submit-only lifecycle management. +5. Preserve the workflow ID, monitor it when applicable, and verify result + retrieval. +6. Read `confs//_00/result.json`. Use + `scripts/parse_results.py` for multi-result summaries. Use `apex archive` + only when a consolidated `all_result.json`, database storage, or + `apex report` is required. + +## Calculator Rules + +- **LAMMPS + DeePMD/DPA**: treat the bundled DPA4 single-task checkpoint + `models/DPA4-alloytongqi/model.pt` as provenance, not a direct LAMMPS input. + Use `"type_map": "auto"`. The old image's Bohrium registry mirror (digest + `sha256:43a27ca4a7bba7f774bbd56104d205a6a80cd9d65928f249f6109e9ef37b8402`) + fails with `Unknown model type: dpa4`, so do not submit this model with the + default old LAMMPS image. The candidate `dpa4-alloytongqi-t4` profile is + `pre_snapshot_only` and must fail closed until an exact image ref/digest + passes the packaged benchmark. Its only tested candidate is one rank on one + `c4_m15_1 * NVIDIA T4`; other T4 SKUs, non-T4 GPUs, CPU, multi-rank, and + multi-GPU remain unverified/prohibited. Do not route automatically. + After publication, require absolute `/usr/local/bin/dpa4-lmp` and + `/usr/local/bin/dpa4-phonolammps` entrypoints from the generated profile. +- **VASP**: confirm a licensed executable/image appropriate to the selected + profile. Verify every POTCAR locally. Cloud-only image discovery rules apply + only to the Bohrium profile. +- **ABACUS/VASP k-points**: VASP must set `KSPACING`; ABACUS must set + `kspacing` or `cal_setting.K_POINTS`. +- **VASP executable selection**: APEX reads each generated task `KPOINTS`. + Gamma-centered `1x1x1` always uses `vasp_gam`; every other grid uses + `vasp_std`, for all properties and relaxation. `KGAMMA=True` alone is not + proof. A task selected for `vasp_gam` requires `KPAR=1`. MPI ranks must + match the Bohrium CPU count; `KPAR` must divide ranks and `NCORE` must + divide ranks/KPAR. Do not combine `NCORE` with `NPAR`; a missing `NCORE` + is a warning. +- Model, pseudopotential, orbital, and structure paths must be valid from the + execution environment; do not assume a host path exists in a container or + on a compute node. + +See `reference/calculators.md` and `reference/lammps_potentials.md`. + +## Property Rules + +- Supported types and complete defaults are in `reference/properties.md`. +- `finite_t_elastic` and `melting_point` are LAMMPS-only. +- `finite_t_latt` and `annealing` support LAMMPS and VASP, but not ABACUS. VASP uses `MDALGO=3` and requires a binary compiled with `-Dtbdyn`; annealing `protocol="coexistence"` is a fixed-temperature equilibration plus production run. +- For vacancy/interstitial/phonon/Grüneisen/finite-T calculations, confirm the + final atom count after expansion. Avoid expanding an existing supercell + twice; use `[1,1,1]` after user confirmation when appropriate. +- For gamma/gamma-surface, start from the recommended crystallographic tables + in `reference/properties.md`. Non-tabulated systems are allowed only when the + direction lies on the plane; report the warning and inspect generated + geometry. For disordered RSS/SQS cells, set `parent_lattice` explicitly. +- Gamma and GammaSurface default to 20 Å vacuum. Gamma line supports explicit + `displacement_points`, which must include `0`; the task count is the number + of supplied fractions. +- For `melting_point`, confirm temperatures, replicas, final atom count, stage + lengths, task count, and resources. Optional `restart_files` must contain one + existing file per temperature and are forwarded to each matching replica as + `restart.coexistence.start`; the generated input is not changed to + `read_restart`. `finite_t_latt` never receives this file. Missing, duplicate, + or disagreeing configured replicas keep the bracket `inconclusive`. +- Generate optional Gamma overrides with the material-independent + `--gamma-*` arguments documented in `reference/properties.md`. Always show + the final Gamma JSON and expected task count before submission. + +## RSS + +For random solid solutions and high-entropy materials, read +`reference/rss_workflow.md`. Set `"show_progress": false`, run `apex rss`, and +judge success from generated `conf_*/POSCAR` files plus `rss_metadata.json`. + +## Safety + +- Never store credentials in `param.json`, commit them, or print passwords. +- Never silently choose a backend, potential, VASP license resource, + crystallographic plane, temperature range, or cluster queue. +- Stop on failed validation. Do not describe a workflow as successful until + expected result files exist. +- To terminate a cloud workflow, terminate the inner dflow workflow before any + wrapper process. See `reference/workflow-control.md`. diff --git a/apex/skills/apex-flow/variants/local/profiles/bohrium-direct.md b/apex/skills/apex-flow/variants/local/profiles/bohrium-direct.md new file mode 100644 index 00000000..40bae406 --- /dev/null +++ b/apex/skills/apex-flow/variants/local/profiles/bohrium-direct.md @@ -0,0 +1,54 @@ +# Installed Profile: Bohrium Cloud (Direct Local Submit) + +The Agent and APEX client run on this machine. Calculations run in Bohrium +containers through dflow. Do not create an outer Bohrium submission job. + +## Authentication + +Check the saved account without exposing the password or AccessKey: + +```bash +apex account --show +``` + +Configure either email/password or AccessKey authentication, together with a +program ID: + +```bash +apex account +# or, non-interactively: +apex account --access-key YOUR_ACCESS_KEY --program-id YOUR_PROGRAM_ID +``` + +Interactive setup first asks whether to use email/password or AccessKey; +whitespace-only field input keeps the saved value. Clear saved login methods +without removing the rest of the cloud profile with: + +```bash +apex account --clear # email/password and AccessKey +apex account --clear access-key # AccessKey only +apex account --clear email # email/password only +``` + +The account is stored in `~/.apex/account.json` and merged into Bohrium +configuration by APEX. A saved AccessKey is passed to dflow, which exchanges +it for a short-lived ticket used by DPDispatcher's Bohrium context; APEX does +not write that ticket into `global.json`. +This profile does not require the `BOHRIUM_ACCESS_KEY` environment variable, +`generate_config.py refresh-global`, or an outer `c1_m2_cpu` job. + +## Configuration + +Copy `data/global_bohrium_direct.json` to the job as `global.json`. Select the +calculator image/run command and `scass_type` for the approved backend. VASP +requires a user-authorized licensed image. + +## Submit + +```bash +cd +apex submit param.json -c global.json -f -n +``` + +Keep the process running for automatic monitoring, retrieval, and archival. +Preserve the printed dflow workflow ID. diff --git a/apex/skills/apex-flow/variants/local/profiles/local-cluster.md b/apex/skills/apex-flow/variants/local/profiles/local-cluster.md new file mode 100644 index 00000000..9d5bd790 --- /dev/null +++ b/apex/skills/apex-flow/variants/local/profiles/local-cluster.md @@ -0,0 +1,38 @@ +# Installed Profile: Local Slurm/PBS Cluster + +The Agent runs APEX on a cluster login node. dflow uses local debug mode while +DPDispatcher submits calculator tasks to the site's scheduler. + +## Required Questions + +Before writing `global.json`, ask for: + +- scheduler (`Slurm` or `PBS`); +- partition/queue and account/project flags; +- nodes, tasks/cores, GPUs, memory, and walltime; +- required modules or environment activation; +- calculator `run_command`; +- cluster-visible work/scratch paths. + +Do not invent scheduler flags or credentials. + +## Configuration + +For Slurm, copy `data/global_local_cluster_slurm.json` to `global.json` and +replace every `<...>` placeholder. For PBS, translate scheduler directives to +the site's PBS syntax and set `machine.batch_type` to `PBS`. + +The default assumes the Agent is already on the login node and therefore uses +`Local` context. If APEX runs on a workstation and connects remotely, use +DPDispatcher `SSHContext` instead and obtain hostname, username, port, and +remote root from the user. + +## Submit + +```bash +cd +apex submit -d param.json -c global.json -f -n +``` + +Monitor scheduler jobs and inspect `dpdispatcher.log` on failure. No Bohrium +account or ticket is used. diff --git a/apex/skills/apex-flow/variants/local/profiles/local-debug.md b/apex/skills/apex-flow/variants/local/profiles/local-debug.md new file mode 100644 index 00000000..34d5eadc --- /dev/null +++ b/apex/skills/apex-flow/variants/local/profiles/local-debug.md @@ -0,0 +1,24 @@ +# Installed Profile: Local Workstation + +APEX and calculator executables run on this workstation in dflow debug mode. +No Bohrium account, ticket, cloud image, Kubernetes, or outer job is required. + +## Preflight + +Confirm the selected calculator command works directly in the current shell, +including required modules/environment variables and model/potential files. + +## Configuration + +Copy `data/global_local_debug.json` to the job as `global.json` and replace +`run_command` with the approved local calculator command. + +## Submit + +```bash +cd +apex submit -d param.json -c global.json -f -n +``` + +Do not add Bohrium fields to this configuration. Results and debug artifacts +remain on the local filesystem. diff --git a/apex/skills/apex-flow/variants/local/reference/submission.md b/apex/skills/apex-flow/variants/local/reference/submission.md new file mode 100644 index 00000000..db1888e9 --- /dev/null +++ b/apex/skills/apex-flow/variants/local/reference/submission.md @@ -0,0 +1,58 @@ +# Local Agent Submission Reference + +This skill is installed on the machine that runs the APEX client. The selected +mode is recorded in `execution-profile.md`; read it before preparing a job. + +## Mode Boundary + +| Profile | Authentication | Submit client | Calculator execution | +|---|---|---|---| +| Bohrium cloud | `apex account` | Local Agent machine | Bohrium/dflow containers | +| local | None | Local Agent machine | Same workstation | +| local cluster | Scheduler/user login | Cluster login node | Slurm/PBS compute nodes | + +Do not require the `BOHRIUM_ACCESS_KEY` environment variable or serialize a +ticket in this local edition. Bohrium-direct reads the masked credentials saved +by `apex account`; when AccessKey authentication is selected, dflow exchanges +that key for a short-lived ticket at runtime. Explicit ticket packaging is used +by the separate `apex skill --zip` Cloud/MatMaster edition, where an outer +container cannot read the user's local account file. + +## Common Preparation + +1. Confirm execution profile, calculator backend, structure, and property + parameters. +2. Copy all required model/potential/input files into locations visible from + the selected execution environment. +3. Build `param.json` from `reference/properties.md` and calculator references. +4. Start from the selected profile's audited global template under `data/`. +5. Validate: + + ```bash + python /scripts/validate_inputs.py \ + --param param.json --global global.json + ``` + +6. Submit without `-s` for automatic monitoring, retrieval, and local + `all_result.json` generation. + +## Results + +Prefer property-level files for Agent answers: + +```text +confs//_00/result.json +``` + +Use `scripts/parse_results.py --work-dir --format summary` when many +results must be summarized. `apex archive` is optional and is appropriate when +`all_result.json` is missing, database archival is requested, or a subsequent +`apex report` needs consolidated data. + +## Failure Handling + +- Preserve the exact workflow ID for Bohrium mode. +- For local/cluster debug mode, inspect the local dflow debug directory and + `dpdispatcher.log`. +- Do not retry with changed inputs in place without revalidating. +- A zero exit code is not sufficient: verify expected `result.json` files. diff --git a/apex/submit.py b/apex/submit.py index c3977736..77bbe5ba 100644 --- a/apex/submit.py +++ b/apex/submit.py @@ -6,6 +6,10 @@ import logging import copy import json +import hashlib +import importlib.util +import re +from pathlib import Path from typing import List from multiprocessing import Pool from monty.serialization import loadfn @@ -32,9 +36,412 @@ LAMMPS_PHONON_IMAGE = ( - "registry.dp.tech/dptech/dp/native/prod-397637/" - "deepmd-kit-phonolammps:3.1.3" + "registry.dp.tech/dptech/dp/native/prod-16664/" + "dpa4-phonolammps:0.0.2" ) +GPU_LAMMPS_INTERACTIONS = {"deepmd", "mace", "nep"} + +DPA4_RUNTIME_KIND = "dpa4_pt2" +DPA4_RUNTIME_MODEL_PATH = ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/" + "alloytongqi.t4-sm75.pt2" +) +DPA4_RUNTIME_MODEL_SHA256 = ( + "2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf" +) +DPA4_SOURCE_CHECKPOINT_PATH = ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt" +) +DPA4_SOURCE_CHECKPOINT_SHA256 = ( + "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad" +) +DPA4_SCASS_TYPE = "c4_m15_1 * NVIDIA T4" +DPA4_LAMMPS_RUN_COMMAND = "/usr/local/bin/dpa4-lmp -in in.lammps" +DPA4_PHONOLAMMPS_RUN_COMMAND = ( + "/usr/local/bin/dpa4-phonolammps {input_file} -c {poscar} " + "--dim {dim} {primitive_axes}" +) +DPA4_GROUP_SIZE = 1 +DPA4_POOL_SIZE = 1 +DPA4_DISPATCHER_COMMAND = "python3" +DPA4_JOB_TYPE = "container" +DPA4_PLATFORM = "ali" +_HEX_SHA256_RE = re.compile(r"[0-9a-f]{64}") + + +def _load_dpa4_profile(*, require_published: bool = True) -> dict: + """Load the canonical bundled DPA4 profile without importing a hyphenated package.""" + profile_module_path = ( + Path(__file__).resolve().parent + / "skills" + / "apex-flow" + / "scripts" + / "dpa4_profile.py" + ) + spec = importlib.util.spec_from_file_location( + "apex_bundled_dpa4_profile", + profile_module_path, + ) + if spec is None or spec.loader is None: + raise RuntimeError( + f"Cannot load DPA4 runtime profile helper: {profile_module_path}" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module.load_dpa4_profile(require_published=require_published) + + +def _dpa4_image_name() -> str: + """Return the immutable image identity from the canonical DPA4 profile.""" + profile = _load_dpa4_profile(require_published=True) + image = profile["image"] + return f"{image['ref']}@{str(image['digest']).lower()}" + + +def _sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _iter_effective_lammps_interactions( + relax_param: dict | None, + props_param: dict | None, + flow_type: str, +): + """Yield every interaction that can reach the single LAMMPS run image.""" + if flow_type in {"relax", "joint"} and relax_param: + interaction = relax_param.get("interaction") + if isinstance(interaction, dict): + yield "relax.interaction", interaction + + if flow_type not in {"props", "joint"} or not props_param: + return + + base_interaction = props_param.get("interaction") + properties = props_param.get("properties") or [] + if not properties: + if isinstance(base_interaction, dict): + yield "props.interaction", base_interaction + return + + for index, prop in enumerate(properties): + if not isinstance(prop, dict): + continue + overwrite = (prop.get("cal_setting") or {}).get("overwrite_interaction") + if isinstance(overwrite, dict): + yield ( + f"props.properties[{index}].cal_setting.overwrite_interaction", + overwrite, + ) + elif isinstance(base_interaction, dict): + yield f"props.properties[{index}].interaction", base_interaction + + +def _has_dpa4_runtime_intent(interaction: dict) -> bool: + return ( + "deepmd_runtime" in interaction + or interaction.get("model_in_image") is True + or interaction.get("model") == DPA4_RUNTIME_MODEL_PATH + or "runtime_model_sha256" in interaction + or "source_checkpoint" in interaction + or "source_checkpoint_sha256" in interaction + ) + + +def _validate_exact_dpa4_interaction(label: str, interaction: dict) -> list[str]: + expected = { + "type": "deepmd", + "deepmd_runtime": DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint": DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": DPA4_SOURCE_CHECKPOINT_SHA256, + } + errors = [] + for key in ("runtime_model_sha256", "source_checkpoint_sha256"): + declared = interaction.get(key) + if not isinstance(declared, str) or not _HEX_SHA256_RE.fullmatch(declared): + errors.append(f"{label}.{key} must be a lowercase SHA-256 hex digest") + for key, value in expected.items(): + if interaction.get(key) != value: + errors.append( + f"{label}.{key} must equal {value!r} for the bundled DPA4 " + "T4/PT2 production runtime" + ) + type_map = interaction.get("type_map") + if type_map != "auto": + valid_mapping = ( + isinstance(type_map, dict) + and bool(type_map) + and all( + isinstance(symbol, str) + and bool(symbol.strip()) + and type(index) is int + and index >= 0 + for symbol, index in type_map.items() + ) + and set(type_map.values()) == set(range(len(type_map))) + ) + if not valid_mapping: + errors.append( + f"{label}.type_map must be 'auto' before CLI expansion or a " + "non-empty contiguous element-to-index mapping afterwards" + ) + return errors + + +def _model_locations_with_sha256( + interaction: dict, + work_dir_list: List[os.PathLike], + expected_sha256: str, +) -> list[str]: + """Resolve a staged model across workdirs and return exact hash matches.""" + model = interaction.get("model") + if not isinstance(model, str) or interaction.get("model_in_image") is True: + return [] + model_path = Path(model).expanduser() + candidates = ( + [model_path] + if model_path.is_absolute() + else [Path(work_dir) / model_path for work_dir in work_dir_list] + ) + matches = [] + for candidate in candidates: + try: + if ( + candidate.is_file() + and _sha256_file(candidate) == expected_sha256 + ): + matches.append(str(candidate.resolve())) + except OSError: + continue + return matches + + +def _validate_lammps_runtime_contract( + relax_param: dict | None, + props_param: dict | None, + flow_type: str, + work_dir_list: List[os.PathLike], +) -> str: + """Validate one image-compatible runtime class for the whole workflow. + + FlowGenerator currently owns a single run image, so mixing the bundled + DPA4/PT2 runtime with legacy interactions is never safe, including through + property overwrite_interaction or different submitted work directories. + """ + classifications = [] + errors = [] + for label, interaction in _iter_effective_lammps_interactions( + relax_param, props_param, flow_type + ): + checkpoint_locations = _model_locations_with_sha256( + interaction, work_dir_list, DPA4_SOURCE_CHECKPOINT_SHA256 + ) + if checkpoint_locations: + errors.append( + f"{label}.model is the bundled DPA4 source checkpoint in " + f"{checkpoint_locations}; LAMMPS must receive the image-resident " + f"T4 .pt2 runtime at {DPA4_RUNTIME_MODEL_PATH!r}, never model.pt" + ) + classifications.append(DPA4_RUNTIME_KIND) + continue + + staged_runtime_locations = _model_locations_with_sha256( + interaction, work_dir_list, DPA4_RUNTIME_MODEL_SHA256 + ) + if staged_runtime_locations: + errors.append( + f"{label}.model resolves to the bundled DPA4 T4 .pt2 in " + f"{staged_runtime_locations}, but the production contract " + f"requires the image-resident path {DPA4_RUNTIME_MODEL_PATH!r} " + "with model_in_image=true" + ) + classifications.append(DPA4_RUNTIME_KIND) + continue + + if _has_dpa4_runtime_intent(interaction): + classifications.append(DPA4_RUNTIME_KIND) + errors.extend(_validate_exact_dpa4_interaction(label, interaction)) + else: + classifications.append("legacy") + + runtime_kinds = set(classifications) + if DPA4_RUNTIME_KIND in runtime_kinds and "legacy" in runtime_kinds: + errors.append( + "A single APEX workflow cannot mix legacy LAMMPS interactions with " + "the bundled DPA4/PT2 runtime (including relax/property overrides " + "or submitted work directories), because it has only one run image" + ) + + if DPA4_RUNTIME_KIND in runtime_kinds: + try: + _dpa4_image_name() + except RuntimeError as exc: + errors.append(str(exc)) + + if errors: + raise RuntimeError("Invalid LAMMPS runtime contract:\n- " + "\n- ".join(errors)) + return DPA4_RUNTIME_KIND if runtime_kinds == {DPA4_RUNTIME_KIND} else "legacy" + + +def _uses_phonolammps(props_param: dict | None) -> bool: + return bool( + props_param + and any( + isinstance(prop, dict) + and prop.get("type") in {"phonon", "gruneisen"} + for prop in props_param.get("properties", []) + ) + ) + + +def _effective_dispatcher_machine(wf_config: Config) -> dict | None: + """Read machine_dict after Config and dispatcher_config overrides.""" + machine = wf_config.dispatcher_config_dict.get("machine_dict") + return machine if isinstance(machine, dict) else None + + +def _effective_bohrium_scass(wf_config: Config) -> str | None: + """Read the effective dispatcher SKU after nested machine overrides.""" + machine = _effective_dispatcher_machine(wf_config) + if not isinstance(machine, dict): + return None + remote_profile = machine.get("remote_profile") + if not isinstance(remote_profile, dict): + return None + input_data = remote_profile.get("input_data") + if not isinstance(input_data, dict): + return None + return input_data.get("scass_type") + + +def _effective_bohrium_image(wf_config: Config) -> str | None: + """Read an optional nested Bohrium image override after all merges.""" + machine = _effective_dispatcher_machine(wf_config) + if not isinstance(machine, dict): + return None + remote_profile = machine.get("remote_profile") + if not isinstance(remote_profile, dict): + return None + input_data = remote_profile.get("input_data") + if not isinstance(input_data, dict): + return None + return input_data.get("image_name") + + +def _validate_dpa4_execution_config( + wf_config: Config, + props_param: dict | None, +) -> None: + """Require the one hardware/command profile covered by DPA4 evidence.""" + errors = [] + context_type = str(wf_config.context_type or "").lower() + batch_type = str(wf_config.batch_type or "").lower() + if "bohrium" not in context_type or "bohrium" not in batch_type: + errors.append( + "DPA4/PT2 production runs require Bohrium context_type and " + "batch_type" + ) + for label, value in ( + ("machine", wf_config.machine), + ("dispatcher_config", wf_config.dispatcher_config), + ("resources", wf_config.resources), + ("task", wf_config.task), + ): + if value not in (None, {}): + errors.append( + f"{label} overrides are prohibited for the immutable DPA4 " + "profile; use the generated top-level Bohrium configuration" + ) + effective_machine = _effective_dispatcher_machine(wf_config) or {} + effective_context = str(effective_machine.get("context_type") or "").lower() + effective_batch = str(effective_machine.get("batch_type") or "").lower() + if "bohrium" not in effective_context or "bohrium" not in effective_batch: + errors.append( + "effective dispatcher machine_dict must retain Bohrium " + "context_type and batch_type" + ) + if wf_config.scass_type != DPA4_SCASS_TYPE: + errors.append( + f"scass_type must equal {DPA4_SCASS_TYPE!r}; all other GPU/CPU " + "SKUs are unverified" + ) + if wf_config.job_type != DPA4_JOB_TYPE: + errors.append( + f"job_type must equal {DPA4_JOB_TYPE!r} so the immutable DPA4 " + "container image is actually used" + ) + if wf_config.platform != DPA4_PLATFORM: + errors.append( + f"platform must equal the qualified Bohrium value " + f"{DPA4_PLATFORM!r}" + ) + effective_scass = _effective_bohrium_scass(wf_config) + if effective_scass != DPA4_SCASS_TYPE: + errors.append( + "effective machine.remote_profile.input_data.scass_type must " + f"equal {DPA4_SCASS_TYPE!r}; nested machine overrides may not " + "change the qualified GPU" + ) + effective_image = _effective_bohrium_image(wf_config) + if effective_image is not None: + expected_image = _dpa4_image_name() + if effective_image != expected_image: + errors.append( + "effective machine.remote_profile.input_data.image_name may " + "be absent or equal the immutable DPA4 image " + f"{expected_image!r}; nested image overrides are prohibited" + ) + + dispatcher = wf_config.dispatcher_config_dict + if dispatcher.get("json_file") not in (None, ""): + errors.append( + "dispatcher_config.json_file is prohibited for DPA4 because it " + "can inject machine or resource overrides after validation" + ) + if dispatcher.get("command") != DPA4_DISPATCHER_COMMAND: + errors.append( + "effective dispatcher command must remain the single-process " + f"default {DPA4_DISPATCHER_COMMAND!r}" + ) + if dispatcher.get("remote_command") not in (None, ""): + errors.append( + "dispatcher remote_command is prohibited for DPA4 because it can " + "bypass the audited one-rank wrapper" + ) + + basic = wf_config.basic_config_dict + if basic.get("lammps_run_command") != DPA4_LAMMPS_RUN_COMMAND: + errors.append( + "lammps_run_command must equal the audited single-rank wrapper " + f"{DPA4_LAMMPS_RUN_COMMAND!r}" + ) + if _uses_phonolammps(props_param) and ( + basic.get("phonolammps_run_command") + != DPA4_PHONOLAMMPS_RUN_COMMAND + ): + errors.append( + "phonon/Gruneisen with DPA4 requires phonolammps_run_command=" + f"{DPA4_PHONOLAMMPS_RUN_COMMAND!r}" + ) + if basic.get("group_size") != DPA4_GROUP_SIZE: + errors.append( + f"group_size must equal {DPA4_GROUP_SIZE} for one task per T4" + ) + if basic.get("pool_size") != DPA4_POOL_SIZE: + errors.append( + f"pool_size must equal {DPA4_POOL_SIZE} for the qualified profile" + ) + + if errors: + raise RuntimeError( + "Invalid DPA4 execution profile:\n- " + "\n- ".join(errors) + ) def validate_submit_paths(parameter_dicts: List[dict]) -> None: @@ -60,14 +467,44 @@ def validate_submit_paths(parameter_dicts: List[dict]) -> None: ) -def _select_run_image(calculator: str, props_param: dict, run_image: str) -> str: - if calculator == "lammps" and props_param and any( +def _select_run_image( + calculator: str, + props_param: dict, + run_image: str, + machine_type: str = None, + runtime_contract: str = "legacy", +) -> str: + if runtime_contract == DPA4_RUNTIME_KIND: + if calculator != "lammps": + raise RuntimeError("DPA4/PT2 runtime contract requires calculator=lammps") + expected_image = _dpa4_image_name() + if run_image != expected_image: + logging.warning( + "Exact bundled DPA4/PT2 contract overrides LAMMPS run image " + "%r with immutable candidate %r.", + run_image, + expected_image, + ) + return expected_image + interaction = (props_param or {}).get("interaction", {}) + interaction_type = ( + interaction.get("type") if isinstance(interaction, dict) else None + ) + is_cpu_machine = "_cpu" in str(machine_type or "").lower() + if ( + calculator == "lammps" + and interaction_type in GPU_LAMMPS_INTERACTIONS + and not is_cpu_machine + and props_param + and any( prop.get("type") in {"phonon", "gruneisen"} for prop in props_param.get("properties", []) + ) ): if run_image != LAMMPS_PHONON_IMAGE: logging.warning( - "LAMMPS phonon/Gruneisen requires the validated phonoLAMMPS image; " + "GPU LAMMPS phonon/Gruneisen requires the validated " + "phonoLAMMPS image; " "overriding run image %r with %r.", run_image, LAMMPS_PHONON_IMAGE, @@ -113,8 +550,18 @@ def _infer_type_map_from_structure_file(structure_file: str) -> dict: return {symbol: idx for idx, symbol in enumerate(symbols)} -def _resolve_first_structure_file(param_path: str, structures: List[str]) -> str: +def _resolve_structure_files(param_path: str, structures: List[str]) -> List[str]: + """Resolve every structure file used to build a complete LAMMPS type map.""" base_dir = os.path.dirname(os.path.abspath(param_path)) + structure_files = [] + seen_files = set() + + def add_file(path: str) -> None: + absolute_path = os.path.abspath(path) + if absolute_path not in seen_files: + seen_files.add(absolute_path) + structure_files.append(absolute_path) + for pattern in structures: if os.path.isabs(pattern): search_patterns = [pattern] @@ -122,37 +569,84 @@ def _resolve_first_structure_file(param_path: str, structures: List[str]) -> str search_patterns = [os.path.join(base_dir, pattern), pattern] matches = [] + seen_matches = set() for search_pattern in search_patterns: - matches.extend(glob.glob(search_pattern)) - matches = sorted(set(matches)) + for match in sorted(glob.glob(search_pattern)): + absolute_match = os.path.abspath(match) + if absolute_match not in seen_matches: + seen_matches.add(absolute_match) + matches.append(absolute_match) for match in matches: if os.path.isdir(match): for candidate in ("POSCAR", "CONTCAR", "STRU"): candidate_path = os.path.join(match, candidate) if os.path.isfile(candidate_path): - return candidate_path - nested_poscars = sorted(glob.glob(os.path.join(match, "conf_*", "POSCAR"))) - if nested_poscars: - return nested_poscars[0] + add_file(candidate_path) + break + else: + nested_poscars = sorted( + glob.glob(os.path.join(match, "conf_*", "POSCAR")) + ) + for nested_poscar in nested_poscars: + add_file(nested_poscar) elif os.path.isfile(match): - return match + add_file(match) + + if structure_files: + return structure_files raise RuntimeError( "Cannot infer interaction.type_map automatically: no structure file found " f"for patterns {structures} from {param_path}" ) -def auto_fill_type_map_from_poscar(parameter_dict: dict, param_path: str) -> bool: - interaction = parameter_dict.get("interaction") - if not isinstance(interaction, dict): - return False - if interaction.get("type") in {"vasp", "abacus"}: - return False +def _infer_type_map_from_structure_files(structure_files: List[str]) -> dict: + symbols = [] + seen = set() + for structure_file in structure_files: + for symbol in _infer_type_map_from_structure_file(structure_file): + if symbol not in seen: + seen.add(symbol) + symbols.append(symbol) + if not symbols: + raise RuntimeError("Cannot infer interaction.type_map from empty structures") + return {symbol: idx for idx, symbol in enumerate(symbols)} - current_type_map = interaction.get("type_map") - if isinstance(current_type_map, dict) and current_type_map: - return False - if current_type_map not in (None, "", "auto"): + +def _iter_lammps_interactions_for_type_map(parameter_dict: dict): + """Yield each distinct LAMMPS interaction that may need CLI expansion.""" + seen = set() + interactions = [parameter_dict.get("interaction")] + for prop in parameter_dict.get("properties") or []: + if not isinstance(prop, dict): + continue + cal_setting = prop.get("cal_setting") or {} + if isinstance(cal_setting, dict): + interactions.append(cal_setting.get("overwrite_interaction")) + + for interaction in interactions: + if not isinstance(interaction, dict): + continue + if interaction.get("type") in {"vasp", "abacus"}: + continue + identity = id(interaction) + if identity in seen: + continue + seen.add(identity) + yield interaction + + +def auto_fill_type_map_from_poscar(parameter_dict: dict, param_path: str) -> bool: + interactions = [] + for interaction in _iter_lammps_interactions_for_type_map(parameter_dict): + current_type_map = interaction.get("type_map") + if ( + current_type_map is None + or current_type_map == "" + or current_type_map == "auto" + ): + interactions.append(interaction) + if not interactions: return False structures = parameter_dict.get("structures", []) @@ -161,8 +655,10 @@ def auto_fill_type_map_from_poscar(parameter_dict: dict, param_path: str) -> boo "Cannot infer interaction.type_map automatically because `structures` is empty" ) - structure_file = _resolve_first_structure_file(param_path, structures) - interaction["type_map"] = _infer_type_map_from_structure_file(structure_file) + structure_files = _resolve_structure_files(param_path, structures) + type_map = _infer_type_map_from_structure_files(structure_files) + for interaction in interactions: + interaction["type_map"] = dict(type_map) with open(param_path, "w", encoding="utf-8") as fp: json.dump(parameter_dict, fp, indent=4) @@ -237,6 +733,18 @@ def pack_upload_dir( if prop_prefix: prop_prefix_base = prop_prefix.split('/')[0] include_dirs.add(prop_prefix_base) + # Melting-point continuations may provide one restart per temperature. + # These files are scientific inputs, so stage their containing top-level + # directories alongside models and custom input files. + if prop_param: + for prop in prop_param.get("properties", []): + if prop.get("type") not in {"melting_point", "two_phase_melting"}: + continue + for restart_file in prop.get("cal_setting", {}).get("restart_files", []): + if not os.path.isabs(restart_file): + restart_prefix = Path(restart_file).parts[0] + if restart_prefix not in ("", ".", ".."): + include_dirs.add(restart_prefix) confs = relax_confs + prop_confs assert len(confs) > 0, "No configuration path indicated!" conf_dirs = [] @@ -581,18 +1089,63 @@ def submit_workflow( relax_param, props_param) = judge_flow(parameter_dicts, indicated_flow_type) print(f'Running APEX calculation via {calculator}') print(f'Submitting {flow_type} workflow...') + + # Resolve work directories before choosing the one run image shared by all + # generated LAMMPS steps. Runtime identity must be uniform across them. + work_dir_list = [] + for item in work_dirs: + work_dir_list.extend(glob.glob(os.path.abspath(item))) + work_dir_list = sorted(set(work_dir_list)) + if not work_dir_list: + raise NotADirectoryError('Empty work directory indicated, please check your argument') + + # Scan every effective interaction, even when the base calculator is VASP + # or ABACUS. A DPA4 overwrite must never bypass the one-image calculator + # contract by hiding under a non-LAMMPS base interaction. + runtime_contract = _validate_lammps_runtime_contract( + relax_param, + props_param, + flow_type, + work_dir_list, + ) + if runtime_contract == DPA4_RUNTIME_KIND: + if calculator != "lammps": + raise RuntimeError( + "DPA4/PT2 overwrite_interaction cannot run in a workflow whose " + f"base calculator is {calculator!r}; use one uniform LAMMPS " + "interaction" + ) + _validate_dpa4_execution_config(wf_config, props_param) + make_image = wf_config.basic_config_dict["apex_image_name"] run_image = wf_config.basic_config_dict[f"{calculator}_image_name"] if not run_image: run_image = wf_config.basic_config_dict["run_image_name"] - run_image = _select_run_image(calculator, props_param, run_image) + machine_type = ( + getattr(wf_config, "machine_type", None) + or getattr(wf_config, "scass_type", None) + ) + run_image = _select_run_image( + calculator, + props_param, + run_image, + machine_type=machine_type, + runtime_contract=runtime_contract, + ) run_command = wf_config.basic_config_dict[f"{calculator}_run_command"] if not run_command: run_command = wf_config.basic_config_dict["run_command"] - if calculator == "lammps": - run_command = _with_lammps_retry_env(run_command, wf_config) lammps_run_command = wf_config.basic_config_dict["lammps_run_command"] phonolammps_run_command = wf_config.basic_config_dict["phonolammps_run_command"] + if runtime_contract == DPA4_RUNTIME_KIND: + # Validation above makes this assignment an assertion of the audited + # entry points, rather than a silent repair of an unsafe global.json. + run_command = DPA4_LAMMPS_RUN_COMMAND + lammps_run_command = DPA4_LAMMPS_RUN_COMMAND + if _uses_phonolammps(props_param): + phonolammps_run_command = DPA4_PHONOLAMMPS_RUN_COMMAND + if calculator == "lammps": + run_command = _with_lammps_retry_env(run_command, wf_config) post_image = make_image group_size = wf_config.basic_config_dict["group_size"] pool_size = wf_config.basic_config_dict["pool_size"] @@ -629,11 +1182,6 @@ def submit_workflow( prop["lammps_run_command"] = lammps_run_command # submit the workflows - work_dir_list = [] - for ii in work_dirs: - glob_list = glob.glob(os.path.abspath(ii)) - work_dir_list.extend(glob_list) - work_dir_list.sort() if len(work_dir_list) > 1: n_processes = len(work_dir_list) print(f'Submitting via {n_processes} processes...') @@ -664,8 +1212,6 @@ def submit_workflow( wf_config, labels=labels, ) - else: - raise NotADirectoryError('Empty work directory indicated, please check your argument') def submit_from_args( @@ -684,7 +1230,8 @@ def submit_from_args( param_dict = loadfn(param_path) if auto_fill_type_map_from_poscar(param_dict, param_path): print( - f"Auto-filled interaction.type_map from structure file and updated: {param_path}" + "Auto-filled LAMMPS type_map field(s) from all resolved " + f"structure files and updated: {param_path}" ) parameter_dicts.append(param_dict) diff --git a/apex/superop/RelaxationFlow.py b/apex/superop/RelaxationFlow.py index ad5c4a35..3244bbca 100644 --- a/apex/superop/RelaxationFlow.py +++ b/apex/superop/RelaxationFlow.py @@ -26,6 +26,7 @@ Slices, ) from dflow.plugins.dispatcher import DispatcherExecutor +from apex.core.lib.vasp_runtime import build_kpoint_aware_vasp_command class RelaxationFlow(Steps): @@ -166,11 +167,14 @@ def _build( image=run_image ) if calculator == 'vasp': + kpoint_aware_run_command = build_kpoint_aware_vasp_command( + run_command + ) runcal = Step( name="RelaxVASP-Cal", template=run_fp, parameters={ - "run_image_config": {"command": run_command}, + "run_image_config": {"command": kpoint_aware_run_command}, "task_name": make.outputs.parameters["task_names"], "backward_list": ["INCAR", "POSCAR", "OUTCAR", "CONTCAR"], "backward_dir_name": "relax_task" @@ -245,5 +249,3 @@ def _build( = post.outputs.artifacts["output_all"] self.outputs.artifacts["retrieve_path"]._from \ = post.outputs.artifacts["retrieve_path"] - - diff --git a/apex/superop/SimplePropertySteps.py b/apex/superop/SimplePropertySteps.py index f77b837e..a5fa7038 100644 --- a/apex/superop/SimplePropertySteps.py +++ b/apex/superop/SimplePropertySteps.py @@ -1,4 +1,5 @@ import os +import shlex from pathlib import ( Path, ) @@ -25,6 +26,7 @@ Slices, ) from dflow.plugins.dispatcher import DispatcherExecutor +from apex.core.lib.vasp_runtime import build_kpoint_aware_vasp_command class SimplePropertySteps(Steps): @@ -161,6 +163,17 @@ def _build( # Step for property run if calculator in ['vasp', 'abacus']: + if calculator == "vasp": + property_run_command = build_kpoint_aware_vasp_command( + run_command, staged_run_command=True + ) + else: + quoted_command = shlex.quote(run_command) + property_run_command = ( + "if [ -f run_command ]; then " + f"APEX_RUN_COMMAND={quoted_command} bash run_command; " + f"else {run_command}; fi" + ) run_fp = PythonOPTemplate( run_op, slices=Slices( @@ -179,10 +192,10 @@ def _build( name="PropsVASP-Cal", template=run_fp, parameters={ - "run_image_config": {"command": run_command}, + "run_image_config": {"command": property_run_command}, "task_name": make.outputs.parameters["task_names"], - "backward_list": ["INCAR", "POSCAR", "OUTCAR", "CONTCAR", - "vasprun.xml"] + "backward_list": make.outputs.parameters["backward_list"], + "log_name": "outlog", }, artifacts={ "task_path": make.outputs.artifacts["task_paths"] @@ -196,9 +209,9 @@ def _build( name="PropsABACUS-Cal", template=run_fp, parameters={ - "run_image_config": {"command": run_command}, + "run_image_config": {"command": property_run_command}, "task_name": make.outputs.parameters["task_names"], - "backward_list": ["OUT.ABACUS", "log"], + "backward_list": make.outputs.parameters["backward_list"], "log_name": "log" }, artifacts={ diff --git a/apex/task_failure.py b/apex/task_failure.py index 76af509e..b5c486ec 100644 --- a/apex/task_failure.py +++ b/apex/task_failure.py @@ -8,6 +8,7 @@ HEADER_ONLY_RETRY_REASON = "header_only_lammps_log_after_nonzero_exit" REMOTE_LAMMPS_STARTUP_FAILURE = "remote_lammps_startup_failure" +TRANSIENT_LAMMPS_RETRY_REASON = "transient_signal_or_timeout_retry" def is_lammps_header_only_text(text: str) -> bool: diff --git a/apex/utils.py b/apex/utils.py index 3cce602b..b905301b 100644 --- a/apex/utils.py +++ b/apex/utils.py @@ -11,9 +11,9 @@ from decimal import Decimal from dflow.python import OP from dflow.python import upload_packages -from fpop.vasp import RunVasp from fpop.abacus import RunAbacus from apex.op.RunLAMMPS import RunLAMMPS +from apex.op.RunVASP import RunVASP from apex.account import merge_bohrium_defaults from apex.core.calculator import LAMMPS_INTER_TYPE @@ -219,7 +219,7 @@ def get_task_type(d: dict) -> (str, Type[OP]): interaction_type = d['interaction']['type'] if interaction_type == 'vasp': task_type = 'vasp' - run_op = RunVasp + run_op = RunVASP elif interaction_type == 'abacus': task_type = 'abacus' run_op = RunAbacus diff --git a/docs/gui_dev.md b/docs/gui_dev.md index 4239b424..b68842be 100644 --- a/docs/gui_dev.md +++ b/docs/gui_dev.md @@ -114,11 +114,12 @@ Provides a visual overlay editor for `apex account` stored credentials within th - `email` - `program_id` - `password` (overwrite only, no echo) +- `access_key` (overwrite only, no echo) ### 6.3 Security Policy -- The page never displays plaintext passwords; only shows "Set / Not Set". -- The password input field is cleared after saving. +- The page never displays plaintext passwords or AccessKeys; only shows "Set / Not Set". +- The password and AccessKey input fields are cleared after saving. - The underlying file is still written by `save_account_config()`. ### 6.4 Related Functions diff --git a/docs/properties/melting_point.md b/docs/properties/melting_point.md new file mode 100644 index 00000000..0dc069b6 --- /dev/null +++ b/docs/properties/melting_point.md @@ -0,0 +1,37 @@ +# Two-phase coexistence melting point + +The APEX `melting_point` property implements a direct solid/liquid +coexistence bracket for LAMMPS potentials. + +Each task uses the same relaxed structure and supercell construction: + +1. The cell is divided along `interface_axis` (default `z`). +2. The upper `liquid_fraction` is heated to `premelt_temperature` while the + lower crystal is pinned. +3. The liquid seed is conditioned at the target temperature. +4. Both halves are released for `production_steps` using the requested NPT + barostat and pressure. +5. LAMMPS records local Steinhardt `q6`, thermodynamics, and seed-resolved MSD. +6. APEX normalizes the released trajectory against the prepared solid/liquid + q6 gap and fits 2 ps block means with a Theil-Sen slope. + +LAMMPS writes alternating binary checkpoints `restart.melting.1` and +`restart.melting.2` every `restart_interval` timesteps, starting during the +premelt stage. A successful task also writes `restart.melting.final`. APEX +retrieves all `restart.melting.*` files together with the trajectory and log; +the restart interval defaults to 10000 timesteps (10 ps at the default +0.001 ps timestep). + +A temperature is a solid-side endpoint only when the 95% slope interval lies +above zero and the projected solid-fraction change exceeds the configured +minimum. A liquid-side endpoint uses the corresponding negative criteria. +The highest solid-side and lowest liquid-side endpoints form the bracket. + +The workflow does not automatically submit new temperatures. If the bracket +is missing or too wide, use the reported recommended midpoint in a new APEX +property suffix after reviewing and approving the calculation. + +For reproducible finite-size studies, keep the structure orientation, +interface axis, liquid fraction, temperature protocol, seeds, timestep, +thermostat/barostat settings, and analysis parameters identical while changing +only `supercell_size`. diff --git a/images/dpa4-phonolammps-b95-t4/PRE_SNAPSHOT_EVIDENCE_BOUNDARY.md b/images/dpa4-phonolammps-b95-t4/PRE_SNAPSHOT_EVIDENCE_BOUNDARY.md new file mode 100644 index 00000000..37aafed2 --- /dev/null +++ b/images/dpa4-phonolammps-b95-t4/PRE_SNAPSHOT_EVIDENCE_BOUNDARY.md @@ -0,0 +1,32 @@ +# Pre-snapshot evidence boundary + +The benchmark manifest with SHA-256 +`aa8984b51655150c8fbac3340a1c40ef9ea58f1ad24735b8cc3aaa1803eadb97` +records a successful run on the source Bohrium node only. + +Its image reference and digest are synthetic pre-snapshot markers. They are +not a registry identity and must never unlock APEX submission or a Skill +recommendation. + +That run used benchmark v1 and predates the current v2 complete +`FORCE_CONSTANTS` parser: it proved +the file was non-empty, but did not record all `N^2` ordered 3x3 blocks and +finite values in the manifest. The old manifest therefore does not satisfy the +current verifier even as source-node qualification evidence. + +Two snapshot submissions through `lbg node tosnap` were rejected by the +Bohrium API with `record not found`: one used node ID `1508668`, and one used +the listed machine ID `1494489`. A filtered image readback found no matching +image record. Sensitive root metadata was restored after the failed attempt. +The snapshot and publication gates therefore remain open. + +Promotion requires all of the following against the published immutable +`ref@sha256:digest`: + +1. rerun the packaged 12 LAMMPS legs, six CPU/GPU parity checks, and the + phonoLAMMPS smoke test; +2. rerun the latest-APEX make, RunLAMMPS, and post-processing path; +3. record the exact image identity, `c4_m15_1 * NVIDIA T4`, one visible T4, + one MPI rank, wrapper hashes, PT2 hashes, and result evidence; +4. change the Skill profile from `pre_snapshot_only` to + `post_snapshot_passed` only after all checks pass. diff --git a/images/dpa4-phonolammps-b95-t4/README.md b/images/dpa4-phonolammps-b95-t4/README.md new file mode 100644 index 00000000..8675f553 --- /dev/null +++ b/images/dpa4-phonolammps-b95-t4/README.md @@ -0,0 +1,125 @@ +# DPA4 b95 + phonoLAMMPS T4 snapshot runtime + +This directory records the runtime contract for the DPA4 `alloytongqi` +LAMMPS image. The deliverable image is produced by a **Bohrium filesystem +snapshot** after the exact runtime, model copies, and wrappers below have been +verified. It is not produced from `docs/Dockerfile.deepmd-phonolammps`, which +remains the old DeepMD 3.1.3 runtime. + +The final registry reference, immutable digest, and Bohrium snapshot ID have +not been provided. They are therefore `null` in +`runtime-manifest.template.json`. Do not describe a tag, digest, snapshot, or +latest-APEX compatibility as final until publication and readback supply those +values and the exact digest passes the benchmark. + +## Frozen build identity + +| Component | Exact identity | +| --- | --- | +| Formal Python environment | `/opt/dpa4-phonolammps-3.2.0b0-py310` | +| DeepMD-kit | `3.2.0b0.post0+b95f21e9`, commit `b95f21e998f9c294f4d467cf1422d38fa6b5e80a` | +| PyTorch backend | `2.11.0+cu130`, commit `70d99e998b4955e0049d13a98d77ae1b14db1f45` | +| LAMMPS | `22 Jul 2025 Update 2`, tag `stable_22Jul2025_update2`, commit `a33449868448baf3c73d8eacdb2d329b13361696` | +| phonoLAMMPS | `0.10.1`; upstream version commit `c590fd77efdf5e196ea3c7b5245b168eec8e332f` | +| phono stack | phonopy `4.3.1`, dynaphopy `1.18.0` | + +The phonoLAMMPS wheel has no `direct_url.json` and no upstream `0.10.1` tag. +The listed commit is the upstream commit that introduces version `0.10.1`; +the installed distribution does not independently bind its bytes to that +commit. The manifest preserves this provenance limitation instead of claiming +stronger evidence. + +“Native CUDA OFF” means the DeepMD native core is the CPU variant +(`DP_VARIANT=cpu`); it does **not** mean the image is CPU-only. DPA4 GPU +inference uses the PyTorch backend built with CUDA 13.0 and the T4-specific +AOTI `.pt2`. The native DeepMD core does not link CUDA, while its PyTorch +backend does. + +## Final model layout + +Install and hash the three immutable model files before taking the snapshot: + +| Role | Final path | SHA-256 | +| --- | --- | --- | +| Checkpoint / identity | `/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt` | `c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad` | +| CPU diagnostic/parity only | `/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.cpu-x86_64.pt2` | `d24525ed454c181354397d46ea62b6376b4b79e1df66a160e89160dbb2284dc9` | +| Production T4 runtime | `/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.t4-sm75.pt2` | `2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf` | + +The checkpoint is not a LAMMPS runtime model. The CPU `.pt2` is not a +production fallback. The APEX DPA4 production contract references only the T4 +`.pt2` path. + +On the build host, the current source files are: + +```text +/opt/dpa4-phonolammps-3.2.0b0-py310/share/models/DPA4-alloytongqi/model.pt +/opt/dpa4-artifacts/cpu-x86_64/alloytongqi.pt2 +/opt/dpa4-artifacts/t4-sm75/alloytongqi.pt2 +``` + +From this directory, stage the snapshot payload with: + +```bash +MODEL_DIR=/opt/dpa4-runtime/models/DPA4-alloytongqi +install -d -m 0755 "$MODEL_DIR" /opt/dpa4-runtime /usr/local/bin +install -m 0644 /opt/dpa4-phonolammps-3.2.0b0-py310/share/models/DPA4-alloytongqi/model.pt "$MODEL_DIR/model.pt" +install -m 0644 /opt/dpa4-artifacts/cpu-x86_64/alloytongqi.pt2 "$MODEL_DIR/alloytongqi.cpu-x86_64.pt2" +install -m 0644 /opt/dpa4-artifacts/t4-sm75/alloytongqi.pt2 "$MODEL_DIR/alloytongqi.t4-sm75.pt2" +install -m 0755 dpa4-lmp dpa4-phonolammps dpa4-python3 /usr/local/bin/ +install -m 0755 dpa4-python3 /root/.bohrium/python3 +install -m 0644 runtime-manifest.template.json /opt/dpa4-runtime/ +``` + +Then verify all three model byte sizes and hashes against +`runtime-manifest.template.json`. Its +`models.final_paths_verified` field is `true` because the three final paths +were re-hashed on the snapshot source filesystem. This proves only the source +filesystem layout; it does not prove that a registry image contains those +bytes. + +## Runtime wrappers + +Install `dpa4-lmp`, `dpa4-phonolammps`, and `dpa4-python3` as executable files +in `/usr/local/bin`. Install the same `dpa4-python3` bytes at +`/root/.bohrium/python3`, which is the first `python3` on the snapshot image's +PATH. This is required because dflow's LAMMPS Python OP launches with the +literal command `python3`; without the shim it would resolve to the inherited +DeepMD 3.1.3 environment. All wrappers preserve every user argument with +`"$@"` and execute the exact formal-environment entry point. They set: + +- formal-venv `PATH` and all required `LD_LIBRARY_PATH` prefixes, including + the venv `lib` that supplies `libmpi.so.12`; +- `LAMMPS_PLUGIN_PATH` for `libdeepmd_lmpplugin.so` auto-loading; +- `DP_BACKEND_PLUGIN_PATH` for the PyTorch/PT-export backend; +- `DP_COMPILE_INFER`, `DP_AMP_INFER`, `DP_TRITON_INFER`, + `DP_CUTE_INFER`, and `DP_TF32_INFER` to `0`; +- OpenMP and both DeepMD thread counts to `1`. + +Generated DPA4 LAMMPS inputs must still place `atom_modify map yes` before +`read_data`. Do not add an explicit legacy plugin-load directive. + +## Snapshot and promotion gate + +Before asking Bohrium to snapshot the build host: + +1. Copy the three models and install all three wrappers, including the exact + `/root/.bohrium/python3` shim. +2. Re-hash every model and `critical_artifacts` entry in the manifest. +3. Confirm `dpa4-lmp` auto-loads `libdeepmd_lmpplugin.so` and exposes the + `deepmd` pair style. +4. Confirm `dpa4-phonolammps --version` and the Python LAMMPS API both load; + testing only `lmp -h` does not exercise the MPI loader path. +5. Run the small-cell benchmark in + `apex/skills/apex-flow/benchmarks/dpa4-alloytongqi/` and retain its complete + evidence workspace. +6. Take the Bohrium snapshot, publish it, read back the immutable registry + digest, and then create a filled runtime manifest alongside this template. +7. Re-run the benchmark from the published digest before marking the runtime + compatible. + +The qualified production envelope is exactly one LAMMPS rank on one NVIDIA +Tesla T4 (compute capability 7.5), with the recorded baseline flags and T4 +`.pt2`. Multi-rank execution, multiple GPUs, other GPU families, other flags, +and mutable/unidentified images are unqualified. The existing APEX default +image and the forced phonon/Gruneisen image remain the old image unless a +separate, evidence-backed routing change is made. diff --git a/images/dpa4-phonolammps-b95-t4/dpa4-lmp b/images/dpa4-phonolammps-b95-t4/dpa4-lmp new file mode 100755 index 00000000..bf23ef43 --- /dev/null +++ b/images/dpa4-phonolammps-b95-t4/dpa4-lmp @@ -0,0 +1,23 @@ +#!/usr/bin/env bash +set -euo pipefail + +readonly DPA4_VENV=/opt/dpa4-phonolammps-3.2.0b0-py310 +readonly DPA4_SITE_PACKAGES="$DPA4_VENV/lib/python3.10/site-packages" +readonly DPA4_DEEPMD_LIB="$DPA4_SITE_PACKAGES/deepmd/lib" +readonly DPA4_RUNTIME_LIBS="$DPA4_VENV/lib:$DPA4_DEEPMD_LIB:$DPA4_SITE_PACKAGES/lammps:$DPA4_SITE_PACKAGES/torch/lib:$DPA4_SITE_PACKAGES/nvidia/cu13/lib:$DPA4_SITE_PACKAGES/nvidia/cudnn/lib:$DPA4_SITE_PACKAGES/nvidia/cusparselt/lib:$DPA4_SITE_PACKAGES/nvidia/nccl/lib:$DPA4_SITE_PACKAGES/nvidia/nvshmem/lib" + +export PATH="$DPA4_VENV/bin:${PATH:-/usr/bin:/bin}" +export LD_LIBRARY_PATH="$DPA4_RUNTIME_LIBS${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" +export LAMMPS_PLUGIN_PATH="$DPA4_DEEPMD_LIB" +export DP_BACKEND_PLUGIN_PATH="$DPA4_DEEPMD_LIB" + +export DP_COMPILE_INFER=0 +export DP_AMP_INFER=0 +export DP_TRITON_INFER=0 +export DP_CUTE_INFER=0 +export DP_TF32_INFER=0 +export OMP_NUM_THREADS=1 +export DP_INTRA_OP_PARALLELISM_THREADS=1 +export DP_INTER_OP_PARALLELISM_THREADS=1 + +exec /opt/dpa4-phonolammps-3.2.0b0-py310/bin/lmp "$@" diff --git a/images/dpa4-phonolammps-b95-t4/dpa4-phonolammps b/images/dpa4-phonolammps-b95-t4/dpa4-phonolammps new file mode 100755 index 00000000..321b6652 --- /dev/null +++ b/images/dpa4-phonolammps-b95-t4/dpa4-phonolammps @@ -0,0 +1,23 @@ +#!/usr/bin/env bash +set -euo pipefail + +readonly DPA4_VENV=/opt/dpa4-phonolammps-3.2.0b0-py310 +readonly DPA4_SITE_PACKAGES="$DPA4_VENV/lib/python3.10/site-packages" +readonly DPA4_DEEPMD_LIB="$DPA4_SITE_PACKAGES/deepmd/lib" +readonly DPA4_RUNTIME_LIBS="$DPA4_VENV/lib:$DPA4_DEEPMD_LIB:$DPA4_SITE_PACKAGES/lammps:$DPA4_SITE_PACKAGES/torch/lib:$DPA4_SITE_PACKAGES/nvidia/cu13/lib:$DPA4_SITE_PACKAGES/nvidia/cudnn/lib:$DPA4_SITE_PACKAGES/nvidia/cusparselt/lib:$DPA4_SITE_PACKAGES/nvidia/nccl/lib:$DPA4_SITE_PACKAGES/nvidia/nvshmem/lib" + +export PATH="$DPA4_VENV/bin:${PATH:-/usr/bin:/bin}" +export LD_LIBRARY_PATH="$DPA4_RUNTIME_LIBS${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" +export LAMMPS_PLUGIN_PATH="$DPA4_DEEPMD_LIB" +export DP_BACKEND_PLUGIN_PATH="$DPA4_DEEPMD_LIB" + +export DP_COMPILE_INFER=0 +export DP_AMP_INFER=0 +export DP_TRITON_INFER=0 +export DP_CUTE_INFER=0 +export DP_TF32_INFER=0 +export OMP_NUM_THREADS=1 +export DP_INTRA_OP_PARALLELISM_THREADS=1 +export DP_INTER_OP_PARALLELISM_THREADS=1 + +exec /opt/dpa4-phonolammps-3.2.0b0-py310/bin/phonolammps "$@" diff --git a/images/dpa4-phonolammps-b95-t4/dpa4-python3 b/images/dpa4-phonolammps-b95-t4/dpa4-python3 new file mode 100755 index 00000000..7e2eae15 --- /dev/null +++ b/images/dpa4-phonolammps-b95-t4/dpa4-python3 @@ -0,0 +1,23 @@ +#!/usr/bin/env bash +set -euo pipefail + +readonly DPA4_VENV=/opt/dpa4-phonolammps-3.2.0b0-py310 +readonly DPA4_SITE_PACKAGES="$DPA4_VENV/lib/python3.10/site-packages" +readonly DPA4_DEEPMD_LIB="$DPA4_SITE_PACKAGES/deepmd/lib" +readonly DPA4_RUNTIME_LIBS="$DPA4_VENV/lib:$DPA4_DEEPMD_LIB:$DPA4_SITE_PACKAGES/lammps:$DPA4_SITE_PACKAGES/torch/lib:$DPA4_SITE_PACKAGES/nvidia/cu13/lib:$DPA4_SITE_PACKAGES/nvidia/cudnn/lib:$DPA4_SITE_PACKAGES/nvidia/cusparselt/lib:$DPA4_SITE_PACKAGES/nvidia/nccl/lib:$DPA4_SITE_PACKAGES/nvidia/nvshmem/lib" + +export PATH="$DPA4_VENV/bin:${PATH:-/usr/bin:/bin}" +export LD_LIBRARY_PATH="$DPA4_RUNTIME_LIBS${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" +export LAMMPS_PLUGIN_PATH="$DPA4_DEEPMD_LIB" +export DP_BACKEND_PLUGIN_PATH="$DPA4_DEEPMD_LIB" + +export DP_COMPILE_INFER=0 +export DP_AMP_INFER=0 +export DP_TRITON_INFER=0 +export DP_CUTE_INFER=0 +export DP_TF32_INFER=0 +export OMP_NUM_THREADS=1 +export DP_INTRA_OP_PARALLELISM_THREADS=1 +export DP_INTER_OP_PARALLELISM_THREADS=1 + +exec /opt/dpa4-phonolammps-3.2.0b0-py310/bin/python "$@" diff --git a/images/dpa4-phonolammps-b95-t4/runtime-manifest.template.json b/images/dpa4-phonolammps-b95-t4/runtime-manifest.template.json new file mode 100644 index 00000000..d2cd1ab4 --- /dev/null +++ b/images/dpa4-phonolammps-b95-t4/runtime-manifest.template.json @@ -0,0 +1,276 @@ +{ + "schema_version": "1.0.0", + "manifest_kind": "dpa4-phonolammps-b95-t4-runtime", + "status": "unpublished", + "observed_build_host_at": "2026-08-12", + "image": { + "build_method": "bohrium_snapshot", + "reference": null, + "digest": null, + "snapshot_id": null, + "identity_status": "not_provided" + }, + "runtime": { + "venv": "/opt/dpa4-phonolammps-3.2.0b0-py310", + "python": "/opt/dpa4-phonolammps-3.2.0b0-py310/bin/python", + "python_version": "3.10.12", + "lammps_binary": "/opt/dpa4-phonolammps-3.2.0b0-py310/bin/lmp", + "phonolammps_binary": "/opt/dpa4-phonolammps-3.2.0b0-py310/bin/phonolammps", + "installed_wrappers": { + "lammps": "/usr/local/bin/dpa4-lmp", + "phonolammps": "/usr/local/bin/dpa4-phonolammps", + "python": "/usr/local/bin/dpa4-python3", + "dflow_python3": "/root/.bohrium/python3" + }, + "fixed_environment": { + "LAMMPS_PLUGIN_PATH": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib", + "DP_BACKEND_PLUGIN_PATH": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib", + "DP_COMPILE_INFER": "0", + "DP_AMP_INFER": "0", + "DP_TRITON_INFER": "0", + "DP_CUTE_INFER": "0", + "DP_TF32_INFER": "0", + "OMP_NUM_THREADS": "1", + "DP_INTRA_OP_PARALLELISM_THREADS": "1", + "DP_INTER_OP_PARALLELISM_THREADS": "1" + }, + "path_prefix": [ + "/opt/dpa4-phonolammps-3.2.0b0-py310/bin" + ], + "ld_library_path_prefix": [ + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/lammps", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/torch/lib", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/nvidia/cu13/lib", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/nvidia/cudnn/lib", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/nvidia/cusparselt/lib", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/nvidia/nccl/lib", + "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/nvidia/nvshmem/lib" + ] + }, + "models": { + "checkpoint": { + "path": "/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt", + "build_host_source": "/opt/dpa4-phonolammps-3.2.0b0-py310/share/models/DPA4-alloytongqi/model.pt", + "bytes": 30403297, + "sha256": "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad", + "role": "identity_and_freeze_input" + }, + "cpu_pt2": { + "path": "/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.cpu-x86_64.pt2", + "build_host_source": "/opt/dpa4-artifacts/cpu-x86_64/alloytongqi.pt2", + "bytes": 73318573, + "sha256": "d24525ed454c181354397d46ea62b6376b4b79e1df66a160e89160dbb2284dc9", + "role": "diagnostic_and_cpu_gpu_parity_only", + "production_allowed": false + }, + "t4_pt2": { + "path": "/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.t4-sm75.pt2", + "build_host_source": "/opt/dpa4-artifacts/t4-sm75/alloytongqi.pt2", + "bytes": 78960581, + "sha256": "2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf", + "role": "production_runtime", + "production_allowed": true, + "target": { + "gpu_model": "NVIDIA Tesla T4", + "compute_capability": "7.5", + "sm": "sm75" + } + }, + "final_paths_verified": true + }, + "components": { + "deepmd_kit": { + "version": "3.2.0b0.post0+b95f21e9", + "repository": "https://github.com/deepmodeling/deepmd-kit.git", + "source_commit": "b95f21e998f9c294f4d467cf1422d38fa6b5e80a", + "source_describe": "v3.2.0b0-106-gb95f21e9", + "source_commit_date": "2026-07-07T01:30:07Z", + "source_archive": { + "build_host_path": "/opt/dpa4-src/deepmd-kit-b95f21e9-exact.tar.gz", + "bytes": 20444047, + "sha256": "73ad46bd257c48786df20279749bde3d6fe1883fa0cfef4d096303a37477122f" + }, + "python_wheel": { + "build_host_path": "/opt/dpa4-src/wheels-python-only/deepmd_kit-3.2.0b0.post0+b95f21e9-py37-none-linux_x86_64.whl", + "bytes": 2682404, + "sha256": "6d1749dc973541057b301df5f598ca31e353b34cf196cab4f225a2c855db8759" + }, + "native_core": { + "variant": "cpu", + "native_cuda": false, + "pytorch_backend_enabled": true, + "evidence": "deepmd/lib/run_config.ini: DP_VARIANT=cpu and ENABLE_PYTORCH=1" + } + }, + "torch": { + "metadata_version": "2.11.0", + "runtime_version": "2.11.0+cu130", + "repository": "https://github.com/pytorch/pytorch.git", + "source_commit": "70d99e998b4955e0049d13a98d77ae1b14db1f45", + "cuda_version": "13.0", + "cuda_enabled": true, + "dist_record": { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/torch-2.11.0.dist-info/RECORD", + "bytes": 1413875, + "sha256": "cf4e5a2e297b7fcadb2e2139ef748021611e367f86f3c8b51ca6ae8ce0b1bce4" + } + }, + "lammps": { + "package_version": "2025.7.22.2.0", + "runtime_version": "22 Jul 2025 Update 2", + "repository": "https://github.com/lammps/lammps.git", + "source_tag": "stable_22Jul2025_update2", + "source_tag_object": "9621ce9c6136874188a2f29c6659821a11d3bc47", + "source_commit": "a33449868448baf3c73d8eacdb2d329b13361696", + "source_archive": { + "build_host_path": "/opt/dpa4-src/lammps-stable_22Jul2025_update2.tar.gz", + "bytes": 152263953, + "sha256": "8fca775d9637139fc45ba5d2d0d14db2ff43fba1e73887f7c948e29253149d45" + }, + "dist_record": { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/lammps-2025.7.22.2.0.dist-info/RECORD", + "bytes": 48637, + "sha256": "f1f0353a0ab807da9ac37b0f9ff20580cb6b37f6b68dcf9530544dea3f5794f8" + } + }, + "phonolammps": { + "version": "0.10.1", + "repository": "https://github.com/abelcarreras/phonolammps.git", + "source_tag": null, + "source_commit": "c590fd77efdf5e196ea3c7b5245b168eec8e332f", + "source_commit_evidence": "upstream commit introducing __version__ 0.10.1; installed distribution has no direct_url.json, so the installed bytes do not independently attest the commit", + "dist_record": { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/phonolammps-0.10.1.dist-info/RECORD", + "bytes": 1496, + "sha256": "8c6f6bae7b40f029ab1c8a2d1b701cb1b54c937856a9164cd27de134ee31ac8a" + } + }, + "python_packages": { + "ase": "3.29.0", + "dpdata": "0.2.17", + "dynaphopy": "1.18.0", + "mpich": "5.0.1.post1", + "numpy": "1.26.4", + "phonopy": "4.3.1", + "scipy": "1.15.3", + "seekpath": "2.2.1", + "spglib": "2.7.0" + } + }, + "critical_artifacts": [ + { + "path": "/usr/local/bin/dpa4-lmp", + "source": "images/dpa4-phonolammps-b95-t4/dpa4-lmp", + "bytes": 1040, + "sha256": "ab151ba5197a2820da96ffc0b9ab09a81e4963b47563ef6a7ec508ebd6314697" + }, + { + "path": "/usr/local/bin/dpa4-phonolammps", + "source": "images/dpa4-phonolammps-b95-t4/dpa4-phonolammps", + "bytes": 1048, + "sha256": "429e962bc5024b55e71fcb42dc5c356dbd12a0ddbebc817a125e2f886d879a16" + }, + { + "path": "/usr/local/bin/dpa4-python3", + "source": "images/dpa4-phonolammps-b95-t4/dpa4-python3", + "bytes": 1043, + "sha256": "b2f898e39aaa2cd6ea7bb72b586cbf428866509ccdd50898f7550c7169c3b471" + }, + { + "path": "/root/.bohrium/python3", + "source": "images/dpa4-phonolammps-b95-t4/dpa4-python3", + "bytes": 1043, + "sha256": "b2f898e39aaa2cd6ea7bb72b586cbf428866509ccdd50898f7550c7169c3b471" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/bin/python", + "resolved_path": "/usr/bin/python3.10", + "bytes": 5917224, + "sha256": "7d51cd6b48b521277f5caa4610a82126e315fa2be4df069823a8b1eeb5bd4a86" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/bin/dp", + "bytes": 188, + "sha256": "92c0b1ef4f6ac6bfcd71e000543af6c7af26de18a26f77897e74053eada3411a" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/bin/lmp", + "bytes": 192, + "sha256": "a73138bf427e64b524f5ce9322f5ef388171b46ddeb98e9230b7eab8fcbe8385" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/lammps/lmp", + "bytes": 27744, + "sha256": "2636c5b16dd1e0b30b6681b733dfa1d21fb7ab896290102040c58b9dfe558a48" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/lammps/liblammps.so.0", + "bytes": 43337080, + "sha256": "2dad25619c9bd2a92310e7e5e9cf68d70bd9ade0f0dc3ae13acd575d54a69e1b" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/bin/phonolammps", + "bytes": 7145, + "sha256": "9f2a0a7337bcada3d4f7b4968533638b3c0d0629cfccf0cf6f551389a63fcafe" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/run_config.ini", + "bytes": 698, + "sha256": "6aa2acd9cac3c5c45bd19f84168640eca1b068560ff7f4d182c71ece82c9ffa3" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/libdeepmd.so", + "bytes": 346984, + "sha256": "40b6efbbfb283cd3535d936db2e60120972198bed769b6aabc449197985681ed" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/libdeepmd_c.so", + "bytes": 221160, + "sha256": "3c5c73c338254af7ad2545e0456d0b8cd65e968d4e9f6c3fc77e380ce9bc8927" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/libdeepmd_cc.so", + "bytes": 293712, + "sha256": "01994c564d90a29b2bfad701abca63d95241a76272f7e2fa2bd6b56e83bf9f71" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/libdeepmd_backend_pt.so", + "bytes": 437800, + "sha256": "8bc46dd864a7a5d7aa52289afd6ae0b7f8640ea417cca790ade6cc3f34fef52d" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/libdeepmd_backend_ptexpt.so", + "bytes": 507200, + "sha256": "67ad88f279f1b5ac57580364329203cb10e787f0cdbb8886fbdfe1aba0125bab" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/libdeepmd_lmpplugin.so", + "bytes": 595592, + "sha256": "16c2e2c5c594c8f559d3978c253104c97315ae76405b1a98ba898eab68a72094" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/python3.10/site-packages/deepmd/lib/libdeepmd_op_pt.so", + "bytes": 762496, + "sha256": "fb22f9496d6e64181ccbd98839ecfa9136d3619d965be2886b90294c67d4236c" + }, + { + "path": "/opt/dpa4-phonolammps-3.2.0b0-py310/lib/libmpi.so.12", + "bytes": 13306545, + "sha256": "d6b92a7f76b886b537066c65ef241f0d011afc79a0cb368d32d47c270ecd5e25" + } + ], + "qualification_boundary": { + "production_model": "/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.t4-sm75.pt2", + "cpu_model_is_diagnostic_only": true, + "mpi_ranks": 1, + "gpu_count": 1, + "gpu_model": "NVIDIA Tesla T4", + "compute_capability": "7.5", + "native_deepmd_cuda": false, + "pytorch_cuda_backend": true, + "final_image_benchmark_manifest": null, + "qualification_status": "awaiting_final_image_identity_and_benchmark" + } +} diff --git a/setup.py b/setup.py index c33ed6b1..4b4ec3d5 100644 --- a/setup.py +++ b/setup.py @@ -22,6 +22,10 @@ "skills/apex-flow/models/*/*", "skills/apex-flow/reference/*", "skills/apex-flow/scripts/*", + "skills/apex-flow/benchmarks/*/*", + "skills/apex-flow/variants/local/SKILL.md", + "skills/apex-flow/variants/local/profiles/*", + "skills/apex-flow/variants/local/reference/*", ], }, include_package_data=True, diff --git a/tests/test_abacus_property.py b/tests/test_abacus_property.py index 178c9d4d..40e37e37 100644 --- a/tests/test_abacus_property.py +++ b/tests/test_abacus_property.py @@ -310,7 +310,6 @@ def test_make_property_refine(self): pwd = os.getcwd() target_path_0 = "confs/fcc-Al/eos_00" target_path_2 = "confs/fcc-Al/eos_02" - path_to_work = os.path.abspath(target_path_0) make_property(self.jdata["structures"], self.jdata["interaction"], [property]) dfm_dirs_0 = glob.glob(os.path.join(target_path_0, "task.*")) @@ -336,7 +335,9 @@ def test_make_property_refine(self): make_property( self.jdata["structures"], self.jdata["interaction"], new_prop_list ) - self.assertTrue(os.path.isdir(path_to_work.replace("00", "02"))) + # Prefer relative target path: abspath(...).replace("00","02") breaks when + # the workspace path itself contains "00" (e.g. /personal/00_APEX/...). + self.assertTrue(os.path.isdir(target_path_2)) os.chdir(pwd) dfm_dirs_2 = glob.glob(os.path.join(target_path_2, "task.*")) self.assertEqual(len(dfm_dirs_2), len(dfm_dirs_0)) diff --git a/tests/test_account_cli.py b/tests/test_account_cli.py index 13c5b6a7..39242c5c 100644 --- a/tests/test_account_cli.py +++ b/tests/test_account_cli.py @@ -10,6 +10,7 @@ def test_account_parser(self): argv = [ "apex", "account", "--email", "demo@example.com", + "--access-key", "demo-access-key", "--program-id", "1234", "--show" ] @@ -18,5 +19,16 @@ def test_account_parser(self): self.assertEqual(args.cmd, "account") self.assertEqual(args.email, "demo@example.com") + self.assertEqual(args.access_key, "demo-access-key") self.assertEqual(args.program_id, 1234) self.assertTrue(args.show) + + def test_account_parser_clear_targets(self): + for argv, expected in ( + (["apex", "account", "--clear"], "all"), + (["apex", "account", "--clear", "access-key"], "access-key"), + (["apex", "account", "--clear", "email"], "email"), + ): + with self.subTest(argv=argv), patch.object(sys, "argv", argv): + _, args = parse_args() + self.assertEqual(expected, args.clear) diff --git a/tests/test_account_config.py b/tests/test_account_config.py index 111e70ac..dbb5b64e 100644 --- a/tests/test_account_config.py +++ b/tests/test_account_config.py @@ -2,17 +2,35 @@ import os import tempfile import unittest +from argparse import Namespace from unittest.mock import patch from apex.account import ( BOHRIUM_WORKFLOWS_HOST, DEFAULT_BOHRIUM_CONFIG, + DEFAULT_OPENAPI_CONFIG, + account_from_args, merge_bohrium_defaults, + prompt_for_account_fields, ) +from apex.config import Config from apex.utils import load_config_file class TestAccountConfig(unittest.TestCase): + def test_config_default_lammps_image_is_cpu_safe_runtime(self): + self.assertEqual( + Config().lammps_image_name, + "registry.dp.tech/dptech/dp/native/prod-397637/" + "apex-flow:1.3.0.post", + ) + + def test_openapi_default_gpu_machine_is_l20(self): + self.assertEqual( + DEFAULT_OPENAPI_CONFIG["machine_type"], + "c16_m120_1 * NVIDIA L20", + ) + def test_merge_bohrium_defaults_for_bohrium_config_file(self): with patch.dict(os.environ, {"APEX_ACCOUNT_FILE": "/tmp/does-not-exist.json"}): merged = merge_bohrium_defaults( @@ -109,3 +127,114 @@ def test_ignore_broken_account_file(self): config_file="global_bohrium.json" ) self.assertEqual(merged["dflow_host"], BOHRIUM_WORKFLOWS_HOST) + + def test_access_key_auth_does_not_require_email_or_password(self): + with tempfile.TemporaryDirectory() as tmpdir: + account_file = os.path.join(tmpdir, "account.json") + with open(account_file, "w", encoding="utf-8") as fp: + json.dump({"access_key": "secret-key", "program_id": 1111}, fp) + with patch.dict(os.environ, {"APEX_ACCOUNT_FILE": account_file}): + merged = merge_bohrium_defaults({}, config_file="global_bohrium.json") + + self.assertEqual(merged["access_key"], "secret-key") + self.assertNotIn("email", merged) + self.assertNotIn("password", merged) + + def test_access_key_uses_dflow_and_ticket_compatible_dispatcher_auth(self): + config = Config( + dflow_host=BOHRIUM_WORKFLOWS_HOST, + context_type="Bohrium", + batch_type="Bohrium", + access_key="secret-key", + program_id=1111, + scass_type="c8_m31_1 * NVIDIA T4", + ) + + self.assertEqual(config.bohrium_config_dict["access_key"], "secret-key") + self.assertEqual(config.bohrium_config_dict["app_key"], "") + self.assertEqual(config.machine_dict["context_type"], "Bohrium") + self.assertEqual(config.machine_dict["batch_type"], "Bohrium") + self.assertEqual( + config.machine_dict["remote_profile"], + { + "email": None, + "password": None, + "program_id": 1111, + "input_data": { + "job_type": "container", + "platform": "ali", + "scass_type": "c8_m31_1 * NVIDIA T4", + }, + }, + ) + + def test_clear_login_targets(self): + expected_remaining = { + "all": set(), + "access-key": {"email", "password"}, + "email": {"access_key"}, + } + for clear_target, remaining in expected_remaining.items(): + with self.subTest(clear=clear_target), tempfile.TemporaryDirectory() as tmpdir: + account_file = os.path.join(tmpdir, "account.json") + with open(account_file, "w", encoding="utf-8") as fp: + json.dump( + { + "email": "saved@example.com", + "password": "saved-password", + "access_key": "saved-key", + "program_id": 1111, + }, + fp, + ) + args = Namespace( + file=account_file, + reset=False, + show=False, + non_interactive=True, + clear=clear_target, + dflow_host=None, + k8s_api_server=None, + batch_type=None, + context_type=None, + email=None, + password=None, + access_key=None, + program_id=None, + apex_image_name=None, + ) + + account_from_args(args) + with open(account_file, encoding="utf-8") as fp: + saved = json.load(fp) + + present = {key for key in ("email", "password", "access_key") if key in saved} + self.assertEqual(remaining, present) + self.assertEqual(1111, saved["program_id"]) + + def test_interactive_email_choice_clears_access_key_and_keeps_blanks(self): + current = { + "access_key": "stale-key", + "email": "saved@example.com", + "password": "saved-password", + "program_id": 1111, + } + with patch("apex.account.getpass", return_value=" "), patch( + "builtins.input", side_effect=["1", " ", " "] + ): + updated = prompt_for_account_fields(current) + + self.assertIsNone(updated["access_key"]) + self.assertEqual("saved@example.com", updated["email"]) + self.assertEqual("saved-password", updated["password"]) + self.assertEqual(1111, updated["program_id"]) + + def test_interactive_access_key_choice_keeps_blank_value(self): + current = {"access_key": "saved-key", "program_id": 1111} + with patch("apex.account.getpass", return_value=" "), patch( + "builtins.input", side_effect=["2", " "] + ): + updated = prompt_for_account_fields(current) + + self.assertEqual("saved-key", updated["access_key"]) + self.assertEqual(1111, updated["program_id"]) diff --git a/tests/test_annealing.py b/tests/test_annealing.py index 425e88c7..e49204e6 100644 --- a/tests/test_annealing.py +++ b/tests/test_annealing.py @@ -8,6 +8,7 @@ from monty.serialization import dumpfn, loadfn from apex.archive import ResultStorage +from apex.core.calculator.VASP import VASP from apex.core.calculator.lib import lammps_utils from apex.core.property.Annealing import Annealing from apex.reporter.DashReportApp import DashReportApp, return_prop_class, return_prop_type @@ -19,6 +20,17 @@ ) TYPE_MAP = {"Ti": 0} PARAM = {"type": "deepmd"} +TEST_POSCAR = """Si +1.0 +5.0 0.0 0.0 +0.0 5.0 0.0 +0.0 0.0 5.0 +Si +2 +Direct +0.0 0.0 0.0 +0.5 0.5 0.5 +""" def dummy_interaction(param): @@ -51,7 +63,7 @@ def test_annealing_default_parameter_parsing(): assert prop.supercell_size == [2, 2, 2] task_param = prop.task_param() - assert task_param["cal_type"] == "annealing" + assert task_param["cal_type"] == "static" assert task_param["cal_setting"]["rdf_bins"] == 100 assert task_param["cal_setting"]["rdf_cutoff"] == 6.0 assert task_param["cal_setting"]["req_compute_rdf"] is True @@ -342,6 +354,245 @@ def test_annealing_compute_lower_extracts_rdf_msd_and_volume_temperature(tmp_pat } +def test_annealing_compute_lower_extracts_vasp_trajectory(tmp_path): + task_dir = tmp_path / "task.000000" + task_dir.mkdir() + dumpfn( + { + "start_temp": 300, + "target_temp": 600, + "end_temp": 300, + "timestep_fs": 1.0, + "rdf_bins": 4, + "rdf_cutoff": 4.0, + }, + task_dir / "Annealing.json", + ) + + def xdat_frame(step, x): + return ( + "Si\n1.0\n5 0 0\n0 5 0\n0 0 5\nSi\n2\n" + f"Direct configuration= {step}\n" + "0.0 0.0 0.0\n" + f"{x} 0.0 0.0\n" + ) + + (task_dir / "XDATCAR").write_text( + "APEX_STAGE ramp\n" + + xdat_frame(1, 0.4) + + xdat_frame(2, 0.42) + + "APEX_STAGE decline\n" + + xdat_frame(1, 0.42) + + xdat_frame(2, 0.4), + encoding="utf-8", + ) + (task_dir / "OUTCAR").write_text( + "APEX_STAGE ramp\n" + " direct lattice vectors reciprocal lattice vectors\n" + " 4 0 0 0.25 0 0\n 0 4 0 0 0.25 0\n 0 0 4 0 0 0.25\n" + " volume of cell : 64\n" + " temperature = 400\n external pressure = 1 kB\n" + " direct lattice vectors reciprocal lattice vectors\n" + " 6 0 0 0.1667 0 0\n 0 6 0 0 0.1667 0\n 0 0 6 0 0 0.1667\n" + " volume of cell : 216\n" + " temperature = 600\n external pressure = 2 kB\n" + "APEX_STAGE decline\n" + " temperature = 500\n external pressure = 2 kB\n" + " temperature = 300\n external pressure = 1 kB\n", + encoding="utf-8", + ) + frames = Annealing._parse_vasp_xdatcar(task_dir / "XDATCAR") + Annealing._attach_vasp_thermo(frames, task_dir / "OUTCAR") + assert [frames["ramp"][0]["cell"][i][i] for i in range(3)] == [4.0] * 3 + assert [frames["ramp"][1]["cell"][i][i] for i in range(3)] == [6.0] * 3 + + result, _ = Annealing({"type": "annealing"})._compute_lower( + str(tmp_path / "result.json"), [str(task_dir)], {} + ) + task = result["tasks"]["task.000000"] + assert task["volume_temperature"]["heating"]["temperature"] == [400.0, 600.0] + assert task["volume_temperature"]["heating"]["total_volume"] == [64.0, 216.0] + assert task["volume_temperature"]["cooling"]["pressure"] == [2.0, 1.0] + assert task["msd"]["T_ramp_300K_600K"]["msd_total"][-1] > 0 + + +def test_annealing_dft_respects_disabled_rdf_and_msd(tmp_path): + task_dir = tmp_path / "task.000000" + task_dir.mkdir() + dumpfn( + { + "start_temp": 300, + "target_temp": 600, + "end_temp": 300, + "timestep_fs": 1.0, + "req_compute_rdf": False, + "req_compute_msd": False, + }, + task_dir / "Annealing.json", + ) + (task_dir / "XDATCAR").write_text( + "APEX_STAGE ramp\n" + "Si\n1.0\n5 0 0\n0 5 0\n0 0 5\nSi\n1\n" + "Direct configuration= 1\n0 0 0\n", + encoding="utf-8", + ) + + result, _ = Annealing({"type": "annealing"})._compute_lower( + str(tmp_path / "result.json"), [str(task_dir)], {} + ) + task = result["tasks"]["task.000000"] + assert task["rdf"] == {} + assert task["msd"] == {} + assert "heating" in task["volume_temperature"] + + +def test_annealing_vasp_inputs_and_abacus_rejection(tmp_path): + dft_defaults = Annealing( + {"type": "annealing"}, {"type": "vasp"} + ).task_param()["cal_setting"] + assert dft_defaults["equi_step"] == 100 + assert dft_defaults["ramp_step"] == 200 + assert dft_defaults["cool_step"] == 200 + assert dft_defaults["hold_step"] == 100 + + metadata = { + "start_temp": 300, + "target_temp": 900, + "end_temp": 400, + "equi_step": 10, + "ramp_step": 20, + "cool_step": 40, + "final_equi_step": 30, + "timestep_fs": 1.0, + } + task_param = { + "type": "annealing", + "cal_type": "static", + "cal_setting": { + "pressure_kbar": 0.0, + "langevin_gamma": 10.0, + "K_POINTS": [1, 1, 1, 0, 0, 0], + }, + } + + vasp_dir = tmp_path / "vasp" + vasp_dir.mkdir() + (vasp_dir / "POSCAR").write_text(TEST_POSCAR) + (vasp_dir / "INCAR.base").write_text("ENCUT=300\nKSPACING=0.5\nSMASS=0\n") + dumpfn(metadata, vasp_dir / "Annealing.json") + vasp = VASP( + { + "type": "vasp", + "incar": str(vasp_dir / "INCAR.base"), + "potcars": {"Si": "Si"}, + }, + str(vasp_dir / "POSCAR"), + ) + vasp.make_input_file(str(vasp_dir), "annealing", task_param) + vasp_ramp = (vasp_dir / "INCAR.ramp").read_text() + assert "MDALGO = 3" in vasp_ramp + assert "TEBEG = 300.0" in vasp_ramp + assert "TEEND = 900.0" in vasp_ramp + assert "NSW = 20" in vasp_ramp + assert "SMASS" not in vasp_ramp + assert vasp.backward_files("annealing") == [ + "OUTCAR", + "outlog", + "OSZICAR", + "XDATCAR", + "CONTCAR", + ] + stage_plan = loadfn(vasp_dir / "apex_vasp_stage_plan.json") + assert stage_plan["task_type"] == "annealing" + assert [stage["expected_ionic_steps"] for stage in stage_plan["stages"]] == [ + 10, + 20, + 40, + 30, + ] + + with pytest.raises(NotImplementedError, match="does not support.*ABACUS"): + Annealing({"type": "annealing"}, {"type": "abacus"}) + + +def test_annealing_vasp_coexistence_protocol(tmp_path): + with pytest.raises(ValueError, match="only VASP"): + Annealing({"type": "annealing", "protocol": "coexistence"}) + + equi_dir = make_equi_dir(tmp_path) + prop = Annealing( + { + "type": "annealing", + "protocol": "coexistence", + "cal_setting": {"target_temp": 900, "nblock": 5}, + }, + {"type": "vasp"}, + ) + task = Path(prop.make_confs(str(tmp_path / "coexistence"), str(equi_dir))[0]) + metadata = loadfn(task / "Annealing.json") + assert metadata["protocol"] == "coexistence" + assert metadata["equi_step"] == 5000 + assert metadata["production_step"] == 10000 + + incar = tmp_path / "INCAR.base" + incar.write_text("ENCUT=400\nKSPACING=0.5\n") + calculator = VASP( + {"type": "vasp", "incar": str(incar), "potcars": {"Ti": "Ti"}}, + str(task / "POSCAR"), + ) + calculator.make_input_file( + str(task), "annealing", prop.task_param() + ) + assert "TEBEG = 900.0" in (task / "INCAR.equi").read_text() + assert (task / "INCAR").read_text() == (task / "INCAR.equi").read_text() + assert "NSW = 10000" in (task / "INCAR.production").read_text() + assert "NBLOCK = 5" in (task / "INCAR.production").read_text() + command = (task / "run_command").read_text() + assert "APEX_STAGE equi" in command + assert "APEX_STAGE production" in command + + +def test_annealing_coexistence_collects_production_thermo(tmp_path): + task = tmp_path / "task.000000" + task.mkdir() + dumpfn( + { + "protocol": "coexistence", + "target_temp": 900, + "timestep_fs": 1.0, + "req_compute_rdf": False, + "req_compute_msd": False, + }, + task / "Annealing.json", + ) + frame = ( + "Si\n1.0\n5 0 0\n0 5 0\n0 0 5\nSi\n2\n" + "Direct configuration= {step}\n" + "0.0 0.0 0.0\n0.5 0.5 0.5\n" + ) + (task / "XDATCAR").write_text( + "APEX_STAGE production\n" + + frame.format(step=1) + + frame.format(step=2) + ) + (task / "OUTCAR").write_text( + "APEX_STAGE production\n" + " temperature = 890\n external pressure = 2 kB\n" + " free energy TOTEN = -10 eV\n total energy ETOTAL = -9 eV\n" + " temperature = 910\n external pressure = 3 kB\n" + " free energy TOTEN = -11 eV\n total energy ETOTAL = -10 eV\n" + ) + result, _ = Annealing({"type": "annealing"})._compute_lower( + str(tmp_path / "result.json"), [str(task)], {} + ) + production = result["tasks"]["task.000000"]["volume_temperature"]["production"] + assert production["temperature"] == [890.0, 910.0] + assert production["pressure"] == [2.0, 3.0] + assert production["potential_energy"] == [-10.0, -11.0] + assert production["total_energy"] == [-9.0, -10.0] + assert production["total_volume"] == pytest.approx([125.0, 125.0]) + + def test_annealing_report_registered_and_builds_graph_table(tmp_path): task_dir = tmp_path / "task.000000" write_annealing_analysis_files(task_dir) @@ -511,6 +762,18 @@ def test_annealing_compute_lower_extracts_rdf_msd_and_volume_temperature(self): with tempfile.TemporaryDirectory() as tmp: test_annealing_compute_lower_extracts_rdf_msd_and_volume_temperature(Path(tmp)) + def test_annealing_compute_lower_extracts_vasp_trajectory(self): + with tempfile.TemporaryDirectory() as tmp: + test_annealing_compute_lower_extracts_vasp_trajectory(Path(tmp)) + + def test_annealing_dft_respects_disabled_rdf_and_msd(self): + with tempfile.TemporaryDirectory() as tmp: + test_annealing_dft_respects_disabled_rdf_and_msd(Path(tmp)) + + def test_annealing_vasp_inputs_and_abacus_rejection(self): + with tempfile.TemporaryDirectory() as tmp: + test_annealing_vasp_inputs_and_abacus_rejection(Path(tmp)) + def test_annealing_report_registered_and_builds_graph_table(self): with tempfile.TemporaryDirectory() as tmp: test_annealing_report_registered_and_builds_graph_table(Path(tmp)) diff --git a/tests/test_dpa4_benchmark.py b/tests/test_dpa4_benchmark.py new file mode 100644 index 00000000..ea46284d --- /dev/null +++ b/tests/test_dpa4_benchmark.py @@ -0,0 +1,91 @@ +import sys +import tempfile +import unittest +from pathlib import Path + + +BENCHMARK_DIR = ( + Path(__file__).resolve().parents[1] + / "apex" + / "skills" + / "apex-flow" + / "benchmarks" + / "dpa4-alloytongqi" +) +sys.path.insert(0, str(BENCHMARK_DIR)) + +from run_phonolammps_smoke import ( # noqa: E402 + _poscar_atom_count, + validate_force_constants, +) + + +def _force_constants_text(atom_count: int = 2) -> str: + lines = [str(atom_count)] + for first in range(1, atom_count + 1): + for second in range(1, atom_count + 1): + lines.extend( + ( + "", + f"{first} {second}", + "1.0 0.0 0.0", + "0.0 1.0 0.0", + "0.0 0.0 1.0", + ) + ) + return "\n".join(lines) + "\n" + + +class TestDPA4ForceConstantsValidation(unittest.TestCase): + def setUp(self): + self.tempdir = tempfile.TemporaryDirectory() + self.root = Path(self.tempdir.name) + + def tearDown(self): + self.tempdir.cleanup() + + def _write(self, text: str) -> Path: + path = self.root / "FORCE_CONSTANTS" + path.write_text(text, encoding="utf-8") + return path + + def test_complete_finite_force_constants_pass(self): + result = validate_force_constants(self._write(_force_constants_text()), 2) + self.assertEqual(result["status"], "passed") + self.assertEqual(result["atom_count"], 2) + self.assertEqual(result["matrix_blocks"], 4) + self.assertEqual(result["finite_values"], 36) + self.assertIsNone(result["error"]) + + def test_atom_count_mismatch_fails(self): + with self.assertRaisesRegex(ValueError, "atom count 2 != expected 3"): + validate_force_constants(self._write(_force_constants_text()), 3) + + def test_truncated_matrix_fails(self): + text = _force_constants_text().rsplit("0.0 0.0 1.0", 1)[0] + with self.assertRaisesRegex(ValueError, "truncated in matrix block"): + validate_force_constants(self._write(text), 2) + + def test_non_finite_matrix_fails(self): + text = _force_constants_text().replace("1.0 0.0 0.0", "nan 0.0 0.0", 1) + with self.assertRaisesRegex(ValueError, "non-finite value"): + validate_force_constants(self._write(text), 2) + + def test_trailing_data_fails(self): + with self.assertRaisesRegex(ValueError, "trailing non-empty data"): + validate_force_constants( + self._write(_force_constants_text() + "unexpected\n"), 2 + ) + + def test_poscar_atom_count_vasp5(self): + poscar = self.root / "POSCAR" + poscar.write_text( + "TiV\n1.0\n1 0 0\n0 1 0\n0 0 1\nTi V\n1 2\nDirect\n" + "0 0 0\n0.5 0.5 0.5\n0.25 0.25 0.25\n", + encoding="utf-8", + ) + self.assertEqual(_poscar_atom_count(poscar), 3) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_eos.py b/tests/test_eos.py index ce84a1e5..25b64664 100644 --- a/tests/test_eos.py +++ b/tests/test_eos.py @@ -2,11 +2,13 @@ import os import shutil import sys +import tempfile import unittest +from pathlib import Path import dpdata import numpy as np -from monty.serialization import loadfn +from monty.serialization import dumpfn, loadfn from pymatgen.io.vasp import Incar from apex.core.property.EOS import EOS @@ -107,3 +109,26 @@ def test_make_confs_1(self): ) with self.assertRaises(RuntimeError): self.eos.make_confs(self.target_path, self.equi_path) + + def test_compute_lower_nan_for_failed_task(self): + with tempfile.TemporaryDirectory() as tmp: + prop_dir = Path(tmp) / "eos_00" + task0 = prop_dir / "task.000000" + task1 = prop_dir / "task.000001" + task0.mkdir(parents=True) + task1.mkdir(parents=True) + dumpfn({"volume": 10.0}, task0 / "eos.json") + dumpfn({"volume": 12.0}, task1 / "eos.json") + dumpfn( + {"energies": [-2.0], "atom_numbs": [2]}, + task0 / "result_task.json", + ) + dumpfn({"failed": True}, task1 / "result_task.json") + + res_data, _ = self.eos._compute_lower( + str(prop_dir / "result.json"), + [str(task0), str(task1)], + [str(task0 / "result_task.json"), str(task1 / "result_task.json")], + ) + self.assertEqual(res_data[10.0], -1.0) + self.assertTrue(np.isnan(res_data[12.0])) diff --git a/tests/test_finitetlatt.py b/tests/test_finitetlatt.py index 2ce8ca2f..1fd10152 100644 --- a/tests/test_finitetlatt.py +++ b/tests/test_finitetlatt.py @@ -1,125 +1,428 @@ -import glob -import os -import shutil -import sys -import unittest - -from monty.serialization import loadfn - -sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) -__package__ = "tests" - -from apex.core.property.FiniteTlatt import FiniteTlatt -from apex.core.calculator.Lammps import Lammps - - -class TestFiniteTlatt(unittest.TestCase): - def setUp(self): - base = { - "structures": ["confs/hcp-Ti"], - "interaction": { - "type": "meam_spline", - "model": "lammps_input/Ti.meam.spline", - "type_map": {"Ti": 0} - }, - "properties": [ - { - "type": "finite_t_latt", - "supercell_size": [2, 2, 2], - "cal_setting":{ - "temperature": [400, 600], - "equi_step": 4000, - "N_every": 100, - "N_repeat": 5, - "N_freq": 1000, - "ave_step": 4000 - } - } - ], - } - - self.equi_path = "confs/hcp-Ti/relaxation/relax_task" - self.source_path = "equi/lammps" - self.target_path = "confs/hcp-Ti/FiniteTlatt_00" - - if not os.path.exists(self.equi_path): - os.makedirs(self.equi_path) - - self.confs = base["structures"] - self.inter_param = base["interaction"] - self.prop_param = base["properties"] - - self.finite = FiniteTlatt(self.prop_param[0]) - self.lammps = Lammps( - self.inter_param, os.path.join(self.source_path, "hcp-Ti-CONTCAR") - ) - - def tearDown(self): - if os.path.exists(os.path.abspath(os.path.join(self.equi_path, ".."))): - shutil.rmtree(os.path.abspath(os.path.join(self.equi_path, ".."))) - if os.path.exists(self.equi_path): - shutil.rmtree(self.equi_path) - if os.path.exists(self.target_path): - shutil.rmtree(self.target_path) - - def test_task_type(self): - self.assertEqual("finite_t_latt", self.finite.task_type()) - - def test_task_param(self): - self.assertEqual(self.prop_param[0], self.finite.task_param()) - - def test_make_potential_files(self): - cwd = os.getcwd() - abs_equi_path = os.path.abspath(self.equi_path) - self.lammps.make_potential_files(abs_equi_path) - self.assertTrue(os.path.islink(os.path.join(self.equi_path, "Ti.meam.spline"))) - self.assertTrue(os.path.isfile(os.path.join(self.equi_path, "inter.json"))) - ret = loadfn(os.path.join(self.equi_path, "inter.json")) - self.assertEqual(self.inter_param, ret) - os.chdir(cwd) - - def test_make_confs(self): - if not os.path.exists(os.path.join(self.equi_path, "CONTCAR")): - with self.assertRaises(RuntimeError): - self.finite.make_confs(self.target_path, self.equi_path) - shutil.copy( - os.path.join(self.source_path, "hcp-Ti-CONTCAR"), - os.path.join(self.equi_path, "CONTCAR"), - ) - - task_list = self.finite.make_confs(self.target_path, self.equi_path) - self.assertEqual(len(task_list), 2) - dfm_dirs = glob.glob(os.path.join(self.target_path, "task.*")) - num = 0 - dfm_dirs.sort() - - for ii in dfm_dirs: - self.assertTrue(os.path.isfile(os.path.join(ii, "POSCAR"))) - self.assertFalse(os.path.exists(os.path.join(ii, "POSCAR.tmp"))) - FiniteTlatt_json_file = os.path.join(ii, "FiniteTlatt.json") - self.assertTrue(os.path.isfile(FiniteTlatt_json_file)) - variable_FiniteTlatt_file = os.path.join(ii, "variable_FiniteTlatt.in") - self.assertTrue(os.path.isfile(variable_FiniteTlatt_file)) - with open(variable_FiniteTlatt_file, 'r') as file: - lines = file.readlines() - temp = lines[1].strip() - self.assertEqual(temp, "variable temperature equal %.2f" % self.prop_param[0]["cal_setting"]["temperature"][num]) - num += 1 - - def test_forward_common_files(self): - fc_files = ["in.lammps", "variable_FiniteTlatt.in", "Ti.meam.spline"] - self.assertEqual(self.lammps.forward_common_files(self.prop_param[0]["type"]), fc_files) - - def test_backward_files(self): - backward_files = [ - "log.lammps", - "outlog", - "apex_task_status.json", - ".debug.log", - ".debug.stdout", - ".debug.stderr", - "dump.relax", - "average_box.txt", - ] - self.assertEqual(self.lammps.backward_files(self.prop_param[0]["type"]), backward_files) - +import glob +import os +import shutil +import sys +import tempfile +import unittest + +from monty.serialization import dumpfn, loadfn +from pymatgen.io.vasp import Incar + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) +__package__ = "tests" + +from apex.core.property.FiniteTlatt import FiniteTlatt +from apex.core.calculator.Lammps import Lammps +from apex.core.calculator.VASP import VASP + + +class TestFiniteTlatt(unittest.TestCase): + def setUp(self): + base = { + "structures": ["confs/hcp-Ti"], + "interaction": { + "type": "meam_spline", + "model": "lammps_input/Ti.meam.spline", + "type_map": {"Ti": 0} + }, + "properties": [ + { + "type": "finite_t_latt", + "supercell_size": [2, 2, 2], + "cal_setting":{ + "temperature": [400, 600], + "equi_step": 4000, + "N_every": 100, + "N_repeat": 5, + "N_freq": 1000, + "ave_step": 4000 + } + } + ], + } + + self.equi_path = "confs/hcp-Ti/relaxation/relax_task" + self.source_path = "equi/lammps" + self.target_path = "confs/hcp-Ti/FiniteTlatt_00" + + if not os.path.exists(self.equi_path): + os.makedirs(self.equi_path) + + self.confs = base["structures"] + self.inter_param = base["interaction"] + self.prop_param = base["properties"] + + self.finite = FiniteTlatt(self.prop_param[0]) + self.lammps = Lammps( + self.inter_param, os.path.join(self.source_path, "hcp-Ti-CONTCAR") + ) + + def tearDown(self): + if os.path.exists(os.path.abspath(os.path.join(self.equi_path, ".."))): + shutil.rmtree(os.path.abspath(os.path.join(self.equi_path, ".."))) + if os.path.exists(self.equi_path): + shutil.rmtree(self.equi_path) + if os.path.exists(self.target_path): + shutil.rmtree(self.target_path) + + def test_task_type(self): + self.assertEqual("finite_t_latt", self.finite.task_type()) + + def test_task_param(self): + self.assertEqual(self.prop_param[0], self.finite.task_param()) + self.assertEqual("static", self.finite.task_param()["cal_type"]) + + def test_make_potential_files(self): + cwd = os.getcwd() + abs_equi_path = os.path.abspath(self.equi_path) + self.lammps.make_potential_files(abs_equi_path) + self.assertTrue(os.path.islink(os.path.join(self.equi_path, "Ti.meam.spline"))) + self.assertTrue(os.path.isfile(os.path.join(self.equi_path, "inter.json"))) + ret = loadfn(os.path.join(self.equi_path, "inter.json")) + self.assertEqual(self.inter_param, ret) + os.chdir(cwd) + + def test_make_confs(self): + if not os.path.exists(os.path.join(self.equi_path, "CONTCAR")): + with self.assertRaises(RuntimeError): + self.finite.make_confs(self.target_path, self.equi_path) + shutil.copy( + os.path.join(self.source_path, "hcp-Ti-CONTCAR"), + os.path.join(self.equi_path, "CONTCAR"), + ) + + task_list = self.finite.make_confs(self.target_path, self.equi_path) + self.assertEqual(len(task_list), 2) + dfm_dirs = glob.glob(os.path.join(self.target_path, "task.*")) + num = 0 + dfm_dirs.sort() + + for ii in dfm_dirs: + self.assertTrue(os.path.isfile(os.path.join(ii, "POSCAR"))) + self.assertFalse(os.path.exists(os.path.join(ii, "POSCAR.tmp"))) + FiniteTlatt_json_file = os.path.join(ii, "FiniteTlatt.json") + self.assertTrue(os.path.isfile(FiniteTlatt_json_file)) + variable_FiniteTlatt_file = os.path.join(ii, "variable_FiniteTlatt.in") + self.assertTrue(os.path.isfile(variable_FiniteTlatt_file)) + with open(variable_FiniteTlatt_file, 'r') as file: + lines = file.readlines() + temp = lines[1].strip() + self.assertEqual(temp, "variable temperature equal %.2f" % self.prop_param[0]["cal_setting"]["temperature"][num]) + num += 1 + + def test_forward_common_files(self): + fc_files = ["in.lammps", "variable_FiniteTlatt.in", "Ti.meam.spline"] + self.assertEqual(self.lammps.forward_common_files(self.prop_param[0]["type"]), fc_files) + + def test_backward_files(self): + backward_files = [ + "log.lammps", + "outlog", + "apex_task_status.json", + ".debug.log", + ".debug.stdout", + ".debug.stderr", + "dump.relax", + "average_box.txt", + ] + self.assertEqual(self.lammps.backward_files(self.prop_param[0]["type"]), backward_files) + + +class TestFiniteTlattDFT(unittest.TestCase): + POSCAR = """Si +1.0 +2.0 0.0 0.0 +0.0 2.0 0.0 +0.0 0.0 2.0 +Si +1 +Direct +0.0 0.0 0.0 +""" + + def test_vasp_cell_parser_and_defaults(self): + defaults = FiniteTlatt( + {"type": "finite_t_latt"}, {"type": "vasp"} + ).task_param()["cal_setting"] + self.assertEqual(5000, defaults["equi_step"]) + self.assertEqual(10000, defaults["ave_step"]) + self.assertEqual(1.0, defaults["timestep_fs"]) + self.assertEqual( + [300, 500, 700, 900, 1100, 1300, 1500], + defaults["temperature"], + ) + + with tempfile.TemporaryDirectory() as tmp: + outcar = os.path.join(tmp, "OUTCAR") + with open(outcar, "w") as fp: + fp.write( + " direct lattice vectors reciprocal lattice vectors\n" + " 4 0 0 0 0 0\n 0 4 0 0 0 0\n 0 0 4 0 0 0\n" + " direct lattice vectors reciprocal lattice vectors\n" + " 6 0 0 0 0 0\n 0 6 0 0 0 0\n 0 0 6 0 0 0\n" + ) + prop = FiniteTlatt( + {"type": "finite_t_latt"}, {"type": "vasp"} + ) + self.assertEqual((2.5, 2.5, 2.5), prop._average_box(tmp, [2, 2, 2])) + + def test_vasp_npt_inputs(self): + with tempfile.TemporaryDirectory() as tmp: + poscar = os.path.join(tmp, "POSCAR") + incar = os.path.join(tmp, "INCAR.base") + with open(poscar, "w") as fp: + fp.write(self.POSCAR) + with open(incar, "w") as fp: + fp.write("ENCUT=300\nKSPACING=0.5\nSMASS=0\n") + dumpfn( + {"temperature": 500, "supercell_size": [1, 1, 1]}, + os.path.join(tmp, "FiniteTlatt.json"), + ) + task_param = { + "type": "finite_t_latt", + "cal_type": "static", + "cal_setting": { + "equi_step": 10, + "ave_step": 20, + "timestep_fs": 1.0, + "langevin_gamma": 10.0, + }, + } + VASP( + { + "type": "vasp", + "incar": incar, + "potcars": {"Si": "Si"}, + }, + poscar, + ).make_input_file(tmp, "finite_t_latt", task_param) + with open(os.path.join(tmp, "INCAR.nvt")) as fp: + nvt = fp.read() + with open(os.path.join(tmp, "INCAR.equi")) as fp: + equi = fp.read() + with open(os.path.join(tmp, "INCAR")) as fp: + staged = fp.read() + self.assertIn("ISIF = 2", nvt) + self.assertIn("NSW = 0", nvt) + self.assertNotIn("LANGEVIN_GAMMA_L", nvt) + self.assertNotIn("PMASS", nvt) + self.assertIn("MDALGO = 3", equi) + self.assertEqual(equi, staged) + self.assertIn("ISIF = 3", equi) + self.assertIn("POTIM = 1.0", equi) + self.assertIn("LANGEVIN_GAMMA_L = 10.0", equi) + self.assertIn("PMASS = 1000.0", equi) + self.assertNotIn("SMASS", equi) + self.assertEqual( + ["OUTCAR", "outlog", "OSZICAR", "CONTCAR", "XDATCAR"], + VASP( + {"type": "vasp", "incar": incar, "potcars": {"Si": "Si"}}, + poscar, + ).backward_files("finite_t_latt"), + ) + + def test_vasp_segmented_nvt_temperature_schedule(self): + with tempfile.TemporaryDirectory() as tmp: + poscar = os.path.join(tmp, "POSCAR") + incar = os.path.join(tmp, "INCAR.base") + with open(poscar, "w") as fp: + fp.write(self.POSCAR) + with open(incar, "w") as fp: + fp.write("ENCUT=300\nKSPACING=0.5\n") + with open(os.path.join(tmp, "POTCAR"), "w") as fp: + fp.write("ENMAX = 200.0; ENMIN = 150.0\n") + dumpfn( + {"temperature": 900, "supercell_size": [1, 1, 1]}, + os.path.join(tmp, "FiniteTlatt.json"), + ) + task_param = { + "type": "finite_t_latt", + "cal_type": "static", + "cal_setting": { + "nvt_temperature_schedule": [300, 500, 700, 900], + "nvt_step": 50, + "equi_step": 100, + "ave_step": 300, + "timestep_fs": 1.0, + "encut": 300, + "langevin_gamma": 10.0, + }, + } + VASP( + { + "type": "vasp", + "incar": incar, + "potcars": {"Si": "Si"}, + }, + poscar, + ).make_input_file(tmp, "finite_t_latt", task_param) + + for index, temperatures in enumerate( + ((300, 500), (500, 700), (700, 900)) + ): + staged = Incar.from_file( + os.path.join(tmp, f"INCAR.nvt_{index:03d}") + ) + self.assertEqual(staged["NSW"], 50) + self.assertEqual(staged["ISIF"], 2) + self.assertEqual(staged["TEBEG"], temperatures[0]) + self.assertEqual(staged["TEEND"], temperatures[1]) + + plan = loadfn(os.path.join(tmp, "apex_vasp_stage_plan.json")) + self.assertEqual(plan["task_type"], "finite_t_latt") + self.assertEqual( + [stage["expected_ionic_steps"] for stage in plan["stages"]], + [50, 50, 50, 100, 300], + ) + self.assertEqual( + [stage["name"] for stage in plan["stages"][:3]], + [ + "nvt_000_300K_to_500K", + "nvt_001_500K_to_700K", + "nvt_002_700K_to_900K", + ], + ) + command = open(os.path.join(tmp, "run_command")).read() + self.assertIn("mv OUTCAR OUTCAR.nvt_000", command) + self.assertIn("mv OSZICAR OSZICAR.nvt_000", command) + self.assertIn("cp CONTCAR CONTCAR.nvt_000", command) + + def test_vasp_schedule_must_end_at_target_temperature(self): + with tempfile.TemporaryDirectory() as tmp: + poscar = os.path.join(tmp, "POSCAR") + incar = os.path.join(tmp, "INCAR.base") + with open(poscar, "w") as fp: + fp.write(self.POSCAR) + with open(incar, "w") as fp: + fp.write("ENCUT=300\nKSPACING=0.5\n") + with open(os.path.join(tmp, "POTCAR"), "w") as fp: + fp.write("ENMAX = 200.0; ENMIN = 150.0\n") + dumpfn( + {"temperature": 900, "supercell_size": [1, 1, 1]}, + os.path.join(tmp, "FiniteTlatt.json"), + ) + task_param = { + "type": "finite_t_latt", + "cal_type": "static", + "cal_setting": { + "nvt_temperature_schedule": [300, 600], + "nvt_step": 50, + "equi_step": 100, + "ave_step": 300, + "encut": 300, + }, + } + with self.assertRaisesRegex( + ValueError, "must match.*target temperature" + ): + VASP( + { + "type": "vasp", + "incar": incar, + "potcars": {"Si": "Si"}, + }, + poscar, + ).make_input_file(tmp, "finite_t_latt", task_param) + + def test_cell_statistics_preserve_legacy_result_shape(self): + with tempfile.TemporaryDirectory() as tmp: + task = os.path.join(tmp, "task.000000") + os.makedirs(task) + with open(os.path.join(task, "OUTCAR"), "w") as fp: + for length in (4.0, 6.0, 8.0, 10.0): + fp.write( + " direct lattice vectors reciprocal lattice vectors\n" + f" {length} 0 0 0 0 0\n" + f" 0 {length} 0 0 0 0\n" + f" 0 0 {length} 0 0 0\n" + ) + prop = FiniteTlatt( + { + "type": "finite_t_latt", + "supercell_size": [2, 2, 2], + "cal_setting": {"temperature": [500]}, + }, + {"type": "vasp"}, + ) + result, _ = prop._compute_lower( + os.path.join(tmp, "result.json"), [task], {} + ) + self.assertEqual([3.5, 3.5, 3.5, 500], result["500"]) + stats = result["_metadata"]["temperatures"]["500"] + self.assertEqual(4, stats["sample_count"]) + self.assertEqual([3.5, 3.5, 3.5], stats["lengths"]["mean"]) + self.assertEqual([90.0, 90.0, 90.0], stats["angles"]["mean"]) + self.assertEqual(4, stats["cell"]["sample_count"]) + self.assertGreater(stats["volume"]["std"], 0) + + def test_vasp_cell_statistics_drop_pre_ionic_initial_cell(self): + with tempfile.TemporaryDirectory() as tmp: + task = os.path.join(tmp, "task.000000") + os.makedirs(task) + with open(os.path.join(task, "OUTCAR"), "w") as fp: + for index, length in enumerate((4.0, 6.0, 8.0, 10.0)): + fp.write( + " direct lattice vectors reciprocal lattice vectors\n" + f" {length} 0 0 0 0 0\n" + f" 0 {length} 0 0 0 0\n" + f" 0 0 {length} 0 0 0\n" + ) + if index: + fp.write( + " POSITION TOTAL-FORCE (eV/Angst)\n" + ) + prop = FiniteTlatt( + { + "type": "finite_t_latt", + "supercell_size": [2, 2, 2], + "cal_setting": {"temperature": [500]}, + }, + {"type": "vasp"}, + ) + result, _ = prop._compute_lower( + os.path.join(tmp, "result.json"), [task], {} + ) + stats = result["_metadata"]["temperatures"]["500"] + self.assertEqual(3, stats["sample_count"]) + self.assertEqual([4.0, 4.0, 4.0], stats["lengths"]["mean"]) + + def test_vasp_md_defaults_and_potcar_encut_validation(self): + with tempfile.TemporaryDirectory() as tmp: + poscar = os.path.join(tmp, "POSCAR") + incar = os.path.join(tmp, "INCAR.base") + with open(poscar, "w") as fp: + fp.write(self.POSCAR) + with open(incar, "w") as fp: + fp.write("ENCUT=300\nKSPACING=0.5\n") + with open(os.path.join(tmp, "POTCAR"), "w") as fp: + fp.write("ENMAX = 250.0; ENMIN = 200.0\n") + dumpfn({"temperature": 500}, os.path.join(tmp, "FiniteTlatt.json")) + task_param = { + "type": "finite_t_latt", + "cal_type": "static", + "cal_setting": {"equi_step": 1, "ave_step": 1}, + } + calculator = VASP( + {"type": "vasp", "incar": incar, "potcars": {"Si": "Si"}}, + poscar, + ) + with self.assertRaisesRegex(ValueError, "1.3"): + calculator.make_input_file(tmp, "finite_t_latt", task_param) + task_param["cal_setting"]["encut"] = 325 + calculator.make_input_file(tmp, "finite_t_latt", task_param) + text = open(os.path.join(tmp, "INCAR.production")).read() + for expected in ( + "PREC = Accurate", + "EDIFF = 1e-06", + "ISMEAR = 1", + "SIGMA = 0.2", + "LASPH = True", + "LREAL = Auto", + "ALGO = Normal", + "NBLOCK = 1", + ): + self.assertIn(expected, text) + + def test_abacus_backend_is_rejected_early(self): + with self.assertRaisesRegex(NotImplementedError, "does not support.*ABACUS"): + FiniteTlatt({"type": "finite_t_latt"}, {"type": "abacus"}) diff --git a/tests/test_flow.py b/tests/test_flow.py index 104dd2de..4cb77484 100644 --- a/tests/test_flow.py +++ b/tests/test_flow.py @@ -284,6 +284,34 @@ def test_format_step_failure_includes_traceback_excerpt(tmp_path): assert "RuntimeError: phonopy failed" in formatted +def test_failure_log_excerpt_redacts_credentials(tmp_path): + log_path = tmp_path / "main.log" + log_path.write_text( + 'machine={"access_key": "secret-access", "password": "secret-password"}\n' + "BOHR_TICKET=secret-ticket python run.py\n" + "accessKey=secret-unquoted\n" + "ticket=secret-plain\n" + "Authorization: Bearer secret-bearer\n" + 'headers={"authorization": "Bearer secret-json-bearer"}\n' + "url=https://example.invalid/run?accessKey=secret-query&expiration=24\n" + "RuntimeError: submission failed\n", + encoding="utf-8", + ) + + excerpt = flow.FlowGenerator._failure_log_excerpt(str(log_path)) + + assert "secret-access" not in excerpt + assert "secret-password" not in excerpt + assert "secret-ticket" not in excerpt + assert "secret-unquoted" not in excerpt + assert "secret-plain" not in excerpt + assert "secret-bearer" not in excerpt + assert "secret-json-bearer" not in excerpt + assert "secret-query" not in excerpt + assert excerpt.count("[REDACTED]") == 8 + assert "RuntimeError: submission failed" in excerpt + + def test_format_step_failure_includes_lammps_diagnostic_excerpt(tmp_path): task_dir = tmp_path / "failed-artifacts" / "prop-key" / "RunLAMMPS" / "backward_dir" / "task.000000" task_dir.mkdir(parents=True) @@ -549,6 +577,10 @@ def test_format_step_failure_includes_traceback_excerpt(self): with tempfile.TemporaryDirectory() as tmp: test_format_step_failure_includes_traceback_excerpt(Path(tmp)) + def test_failure_log_excerpt_redacts_credentials(self): + with tempfile.TemporaryDirectory() as tmp: + test_failure_log_excerpt_redacts_credentials(Path(tmp)) + def test_format_step_failure_includes_lammps_diagnostic_excerpt(self): with tempfile.TemporaryDirectory() as tmp: test_format_step_failure_includes_lammps_diagnostic_excerpt(Path(tmp)) diff --git a/tests/test_gamma.py b/tests/test_gamma.py index e2203a41..2d5abc7f 100644 --- a/tests/test_gamma.py +++ b/tests/test_gamma.py @@ -4,7 +4,9 @@ import sys import unittest -from pymatgen.core.structure import Structure +import pytest +from monty.serialization import dumpfn +from pymatgen.core import Lattice, Structure from pymatgen.io.vasp import Incar from apex.core.property.Gamma import Gamma @@ -74,6 +76,84 @@ def test_task_type(self): def test_task_param(self): self.assertEqual(self.prop_param[0], self.gamma.task_param()) + def test_parent_lattice_hint_is_normalized(self): + gamma = Gamma( + { + "type": "gamma", + "parent_lattice": " BCC ", + "plane_miller": [0, 1, 1], + "slip_direction": [1, -1, 1], + } + ) + self.assertEqual(gamma.parent_lattice, "bcc") + self.assertEqual(gamma.task_param()["parent_lattice"], "bcc") + + def test_invalid_parent_lattice_hint_fails(self): + with self.assertRaisesRegex(ValueError, "bcc, fcc, hcp"): + Gamma({"type": "gamma", "parent_lattice": "b2"}) + + def test_displacement_points_require_zero_reference(self): + with self.assertRaisesRegex(ValueError, "must include 0"): + Gamma({"type": "gamma", "displacement_points": [0.25, 0.5, 1.0]}) + + def test_gamma_rejects_invalid_steps_and_vacuum(self): + for value in (0, -1, 1.5, True): + with self.subTest(n_steps=value): + with self.assertRaisesRegex(ValueError, "positive integer"): + Gamma({"type": "gamma", "n_steps": value}) + for value in (-1.0, float("nan"), float("inf"), True): + with self.subTest(vacuum_size=value): + with self.assertRaisesRegex(ValueError, "finite number"): + Gamma({"type": "gamma", "vacuum_size": value}) + + def test_gamma_orthogonalize_alias_is_strict_gate(self): + gamma = Gamma({"type": "gamma", "orthogonalize_cell": True}) + self.assertTrue(gamma.require_orthogonal_cell) + self.assertTrue(gamma.task_param()["require_orthogonal_cell"]) + with self.assertRaisesRegex(ValueError, "must be a boolean"): + Gamma({"type": "gamma", "orthogonalize_cell": "false"}) + with self.assertRaisesRegex(ValueError, "disagree"): + Gamma( + { + "type": "gamma", + "require_orthogonal_cell": True, + "orthogonalize_cell": False, + } + ) + + gamma = Gamma({"type": "gamma", "displacement_points": [0.5, 0.0]}) + self.assertEqual([0.0, 0.5], gamma.displacement_points) + + def test_nonrecommended_system_warns_and_uses_geometric_check(self): + self.gamma.structure_type = "fcc" + self.gamma.plane_miller = [0, 0, 1] + self.gamma.slip_direction = [1, 0, 0] + self.gamma.slip_length = None + structure = Structure( + Lattice.cubic(4.0), + ["Al"], + [[0.0, 0.0, 0.0]], + ) + + with self.assertLogs(level="WARNING") as captured: + plane, direction, slip_length, _ = ( + self.gamma._Gamma__convert_input_miller(structure) + ) + + self.assertEqual(plane, (0, 0, 1)) + self.assertEqual(direction, (1, 0, 0)) + self.assertEqual(slip_length, 1) + self.assertTrue( + any( + "falling back to a geometric construction" in message + for message in captured.output + ) + ) + + self.gamma.slip_direction = [1, 0, 1] + with self.assertRaisesRegex(RuntimeError, "is not on plane"): + self.gamma._Gamma__convert_input_miller(structure) + def test_refine_style_gamma_defaults_cal_type(self): gamma = Gamma({ "type": "gamma", @@ -97,6 +177,11 @@ def test_make_confs_bcc(self): task_list = self.gamma.make_confs(self.target_path, self.equi_path) dfm_dirs = glob.glob(os.path.join(self.target_path, "task.*")) self.assertEqual(len(dfm_dirs), self.gamma.n_steps + 1) + self.assertTrue( + os.path.isfile( + os.path.join(self.target_path, "slab_generation.json") + ) + ) incar0 = Incar.from_file(os.path.join("vasp_input", "INCAR.rlx")) incar0["ISIF"] = 4 @@ -156,3 +241,53 @@ def test_compute_lower(self): self.gamma._compute_lower(output_file, all_tasks, all_res) self.assertTrue(os.path.isfile(self.res_data)) + + +def test_gamma_compute_lower_uses_interface_count_and_fixed_area(tmp_path): + prop_dir = tmp_path / "conf" / "gamma_00" + task0 = prop_dir / "task.000000" + task1 = prop_dir / "task.000001" + equi_dir = tmp_path / "conf" / "relaxation" / "relax_task" + task0.mkdir(parents=True) + task1.mkdir(parents=True) + equi_dir.mkdir(parents=True) + dumpfn({"energies": [-2.0], "atom_numbs": [2]}, equi_dir / "result.json") + cell = [[2.0, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 10.0]] + for task, energy, displacement in ( + (task0, -10.0, 0.0), + (task1, -9.0, 0.5), + ): + dumpfn( + {"energies": [energy], "atom_numbs": [2], "cells": [cell]}, + task / "result_task.json", + ) + dumpfn([1, 1, 0], task / "miller.json") + dumpfn(1.0, task / "slip_length.json") + dumpfn(displacement, task / "normalized_displacement.json") + dumpfn( + {"slab_geometry": {"interface_count": 2}}, + task / "gamma_geometry.json", + ) + prop = Gamma( + { + "type": "gamma", + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1], + "vacuum_size": 0.0, + } + ) + result, _ = prop._compute_lower( + str(prop_dir / "result.json"), [str(task0), str(task1)], {} + ) + expected = 1.0 / (2.0 * 2.0) * 16.0217657 + assert result[0.5][1] == pytest.approx(expected) + + changed_cell = [[2.1, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 10.0]] + dumpfn( + {"energies": [-9.0], "atom_numbs": [2], "cells": [changed_cell]}, + task1 / "result_task.json", + ) + with pytest.raises(RuntimeError, match="in-plane area changed"): + prop._compute_lower( + str(prop_dir / "result.json"), [str(task0), str(task1)], {} + ) diff --git a/tests/test_gamma_slab.py b/tests/test_gamma_slab.py new file mode 100644 index 00000000..171c9c05 --- /dev/null +++ b/tests/test_gamma_slab.py @@ -0,0 +1,162 @@ +import math + +import numpy as np +import pytest +from pymatgen.core import Lattice, Structure +from pymatgen.core.surface import SlabGenerator + +from apex.core.property.gamma_slab import get_first_gamma_slab +from apex.core.property.gamma_slab import make_gamma_slab_generator +from apex.core.property.gamma_slab import validate_gamma_slab_settings +from apex.core.property.gamma_slab import validate_generated_gamma_slab + + +def rounding_regression_structure(): + return Structure( + Lattice( + [ + [4.48871691776465, 0.0, 0.0], + [0.9779202953637698, 2.861234792942396, 0.0], + [-0.6795759322843109, 0.22507920854606156, 2.1757680318455335], + ] + ), + ["Al"], + [[0.0, 0.0, 0.0]], + ) + + +def test_plane_units_prevent_floating_point_layer_promotion(): + structure = rounding_regression_structure() + miller = (1, 0, 1) + d_hkl = structure.lattice.d_hkl(miller) + + legacy = SlabGenerator( + structure, + miller_index=miller, + min_slab_size=d_hkl * 2, + min_vacuum_size=0, + center_slab=True, + in_unit_planes=False, + lll_reduce=True, + reorient_lattice=False, + primitive=False, + ) + assert d_hkl * 2 / legacy._proj_height == 2.0000000000000004 + assert math.ceil(d_hkl * 2 / legacy._proj_height) == 3 + assert len(legacy.get_slab(shift=0)) == 3 + + protected, metadata = make_gamma_slab_generator( + structure, miller, plane_target=2, min_slab_height=None + ) + slab = protected.get_slab(shift=0) + + assert len(slab) == 2 + assert metadata["oriented_cell_repeats"] == 2 + assert metadata["expected_base_atoms"] == 2 + + +def test_minimum_height_promotes_only_to_required_repeat(): + structure = rounding_regression_structure() + miller = (1, 0, 1) + initial, _ = make_gamma_slab_generator( + structure, miller, plane_target=1, min_slab_height=None + ) + target_height = initial._proj_height + 1.0e-6 + + protected, metadata = make_gamma_slab_generator( + structure, + miller, + plane_target=1, + min_slab_height=target_height, + ) + + assert metadata["oriented_cell_repeats"] == 2 + assert metadata["slab_height"] >= target_height + assert len(protected.get_slab(shift=0)) == 2 + + +def test_first_termination_matches_pymatgen_without_materializing_all(): + structure = Structure( + Lattice.cubic(4.0), + ["Al", "Al"], + [[0.0, 0.0, 0.0], [0.25, 0.25, 0.25]], + ) + generator = SlabGenerator( + structure, + miller_index=(1, 0, 0), + min_slab_size=2, + min_vacuum_size=0, + center_slab=True, + in_unit_planes=True, + lll_reduce=True, + reorient_lattice=False, + primitive=False, + ) + + expected = generator.get_slabs(ftol=0.001)[0] + actual = get_first_gamma_slab(generator, ftol=0.001) + + np.testing.assert_allclose(actual.lattice.matrix, expected.lattice.matrix) + np.testing.assert_allclose(actual.frac_coords, expected.frac_coords) + + +def test_atom_count_and_overlap_guards_fail_clearly(): + structure = Structure( + Lattice.cubic(3.0), + ["Al", "Al"], + [[0.0, 0.0, 0.0], [0.01, 0.0, 0.0]], + ) + metadata = { + "expected_base_atoms": 2, + "oriented_cell_repeats": 1, + } + + with pytest.raises(RuntimeError, match="exceeding max_atoms"): + validate_generated_gamma_slab( + structure, + metadata, + inplane_size=(1, 1), + max_atoms=1, + min_distance=0, + property_name="Gamma", + ) + + with pytest.raises(RuntimeError, match="overlapping atoms"): + validate_generated_gamma_slab( + structure, + metadata, + inplane_size=(1, 1), + max_atoms=None, + min_distance=0.2, + property_name="Gamma", + ) + + +@pytest.mark.parametrize( + "kwargs, message", + [ + ( + {"supercell_size": [1, 1], "min_slab_height": None, + "max_atoms": None, "min_distance": 0.2}, + "three values", + ), + ( + {"supercell_size": [1, 1, 2], "min_slab_height": -1, + "max_atoms": None, "min_distance": 0.2}, + "min_slab_height", + ), + ( + {"supercell_size": [1, 1, 2], "min_slab_height": None, + "max_atoms": 0, "min_distance": 0.2}, + "max_atoms", + ), + ( + {"supercell_size": [1, 1, 2], "min_slab_height": None, + "max_atoms": None, "min_distance": -0.1}, + "min_distance", + ), + ], +) +def test_invalid_gamma_slab_settings(kwargs, message): + with pytest.raises(ValueError, match=message): + validate_gamma_slab_settings(**kwargs) diff --git a/tests/test_gamma_surface.py b/tests/test_gamma_surface.py index 4dd345ad..a132bcfc 100644 --- a/tests/test_gamma_surface.py +++ b/tests/test_gamma_surface.py @@ -59,6 +59,55 @@ def test_task_type(self): def test_task_param(self): self.assertEqual(self.prop_param, self.gamma_surface.task_param()) + def test_parent_lattice_hint_is_normalized(self): + gamma_surface = GammaSurface( + { + "type": "gamma_surface", + "parent_lattice": " BCC ", + "plane_miller": [0, 1, 1], + "slip_direction": [1, -1, 1], + } + ) + self.assertEqual(gamma_surface.parent_lattice, "bcc") + self.assertEqual( + gamma_surface.task_param()["parent_lattice"], "bcc" + ) + + def test_invalid_parent_lattice_hint_fails(self): + with self.assertRaisesRegex(ValueError, "bcc, fcc, hcp"): + GammaSurface( + {"type": "gamma_surface", "parent_lattice": "b2"} + ) + + def test_nonrecommended_system_warns_and_uses_geometric_check(self): + self.gamma_surface.structure_type = "fcc" + structure = Structure( + Lattice.cubic(4.0), + ["Al"], + [[0.0, 0.0, 0.0]], + ) + + with self.assertLogs(level="WARNING") as captured: + plane, direction, slip_length, _ = ( + self.gamma_surface._GammaSurface__convert_input_miller( + structure + ) + ) + + self.assertEqual(plane, (0, 0, 1)) + self.assertEqual(direction, (1, 0, 0)) + self.assertEqual(slip_length, 1) + self.assertTrue( + any( + "falling back to a geometric construction" in message + for message in captured.output + ) + ) + + self.gamma_surface.slip_direction = [1, 0, 1] + with self.assertRaisesRegex(RuntimeError, "is not on plane"): + self.gamma_surface._GammaSurface__convert_input_miller(structure) + def test_make_confs_bcc(self): if not os.path.exists(os.path.join(self.equi_path, "CONTCAR")): with self.assertRaises(RuntimeError): @@ -78,6 +127,11 @@ def test_make_confs_bcc(self): dfm_dirs = glob.glob(os.path.join(self.target_path, "task.*")) self.assertEqual(len(dfm_dirs), (self.gamma_surface.n_steps_x + 1) * (self.gamma_surface.n_steps_y + 1)) self.assertEqual(len(task_list), len(dfm_dirs)) + self.assertTrue( + os.path.isfile( + os.path.join(self.target_path, "slab_generation.json") + ) + ) pairs = set() for ii in sorted(dfm_dirs): @@ -231,7 +285,7 @@ def test_gamma_surface_default_cal_setting_fills_missing_values(): "relax_vol": False, } assert prop.supercell_size == (1, 1, 5) - assert prop.vacuum_size == 0 + assert prop.vacuum_size == 20 assert prop.add_fix == ["true", "true", "false"] @@ -274,6 +328,27 @@ def test_gamma_surface_closed_loop_is_opt_in(): GammaSurface({"type": "gamma_surface", "closed_loop": "true"}) +def test_gamma_surface_rejects_invalid_vacuum_and_accepts_strict_alias(): + for value in (-1.0, float("nan"), float("inf"), True): + with pytest.raises(ValueError, match="finite number"): + GammaSurface({"type": "gamma_surface", "vacuum_size": value}) + prop = GammaSurface( + {"type": "gamma_surface", "orthogonalize_cell": True} + ) + assert prop.require_orthogonal_cell is True + assert prop.task_param()["require_orthogonal_cell"] is True + with pytest.raises(ValueError, match="must be a boolean"): + GammaSurface({"type": "gamma_surface", "orthogonalize_cell": "false"}) + with pytest.raises(ValueError, match="disagree"): + GammaSurface( + { + "type": "gamma_surface", + "require_orthogonal_cell": True, + "orthogonalize_cell": False, + } + ) + + def test_gamma_surface_finds_and_validates_oblique_periodic_vectors(): slab = Structure( lattice=Lattice( @@ -476,6 +551,69 @@ def test_gamma_surface_compute_lower_with_synthetic_results(tmp_path): assert (prop_dir / "result.json").is_file() +def test_gamma_surface_compute_lower_nan_for_failed_task(tmp_path): + prop_dir = tmp_path / "conf" / "gamma_surface_00" + task0 = prop_dir / "task.000000" + task1 = prop_dir / "task.000001" + equi_dir = tmp_path / "conf" / "relaxation" / "relax_task" + task0.mkdir(parents=True) + task1.mkdir(parents=True) + equi_dir.mkdir(parents=True) + + cell = np.eye(3).tolist() + dumpfn({"energies": [-2.0], "atom_numbs": [2]}, equi_dir / "result.json") + dumpfn( + {"energies": [-2.0], "atom_numbs": [2], "cells": [cell]}, + task0 / "result_task.json", + ) + dumpfn({"failed": True}, task1 / "result_task.json") + for task, frac_x, frac_y in [(task0, 0.0, 0.0), (task1, 0.5, 0.0)]: + dumpfn([0, 0, 1], task / "miller.json") + dumpfn({"frac_x": frac_x, "frac_y": frac_y}, task / "displacement.json") + dumpfn(2.0, task0 / "slip_length_x.json") + dumpfn(3.0, task0 / "slip_length_y.json") + + prop = GammaSurface( + { + "type": "gamma_surface", + "plane_miller": [0, 0, 1], + "slip_direction": [1, 0, 0], + } + ) + res_data, _ = prop._compute_lower( + str(prop_dir / "result.json"), + [str(task0), str(task1)], + {}, + ) + assert res_data["0.000000,0.000000"][2] == 0.0 + assert np.isnan(res_data["0.500000,0.000000"][2]) + assert np.isnan(res_data["0.500000,0.000000"][3]) + + +def test_gamma_surface_compute_lower_fails_if_reference_task_failed(tmp_path): + prop_dir = tmp_path / "conf" / "gamma_surface_00" + task0 = prop_dir / "task.000000" + equi_dir = tmp_path / "conf" / "relaxation" / "relax_task" + task0.mkdir(parents=True) + equi_dir.mkdir(parents=True) + dumpfn({"energies": [-2.0], "atom_numbs": [2]}, equi_dir / "result.json") + dumpfn({"failed": True}, task0 / "result_task.json") + dumpfn([0, 0, 1], task0 / "miller.json") + dumpfn({"frac_x": 0.0, "frac_y": 0.0}, task0 / "displacement.json") + dumpfn(2.0, task0 / "slip_length_x.json") + dumpfn(3.0, task0 / "slip_length_y.json") + + prop = GammaSurface( + { + "type": "gamma_surface", + "plane_miller": [0, 0, 1], + "slip_direction": [1, 0, 0], + } + ) + with pytest.raises(RuntimeError, match="reference task"): + prop._compute_lower(str(prop_dir / "result.json"), [str(task0)], {}) + + def test_gamma_surface_compute_lower_preserves_closed_loop_cartesian_data(tmp_path): prop_dir = tmp_path / "conf" / "gamma_surface_00" task = prop_dir / "task.000000" @@ -521,6 +659,60 @@ def test_gamma_surface_compute_lower_preserves_closed_loop_cartesian_data(tmp_pa assert entry[5]["slip_vector_y"] == [0.25, 3.0, 0.5] +def test_gamma_surface_compute_lower_uses_interface_count_and_fixed_area(tmp_path): + prop_dir = tmp_path / "conf" / "gamma_surface_00" + task0 = prop_dir / "task.000000" + task1 = prop_dir / "task.000001" + equi_dir = tmp_path / "conf" / "relaxation" / "relax_task" + task0.mkdir(parents=True) + task1.mkdir(parents=True) + equi_dir.mkdir(parents=True) + dumpfn({"energies": [-2.0], "atom_numbs": [2]}, equi_dir / "result.json") + cell = [[2.0, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 10.0]] + for task, energy, frac_x in ( + (task0, -10.0, 0.0), + (task1, -9.0, 0.5), + ): + dumpfn( + {"energies": [energy], "atom_numbs": [2], "cells": [cell]}, + task / "result_task.json", + ) + dumpfn([1, 1, 0], task / "miller.json") + dumpfn( + {"frac_x": frac_x, "frac_y": 0.0, "disp_cart": [frac_x, 0.0, 0.0]}, + task / "displacement.json", + ) + dumpfn(1.0, task / "slip_length_x.json") + dumpfn(1.0, task / "slip_length_y.json") + dumpfn( + {"slab_geometry": {"interface_count": 2}}, + task / "gamma_geometry.json", + ) + prop = GammaSurface( + { + "type": "gamma_surface", + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1], + "vacuum_size": 0.0, + } + ) + result, _ = prop._compute_lower( + str(prop_dir / "result.json"), [str(task0), str(task1)], {} + ) + expected = 1.0 / (2.0 * 2.0) * 16.0217657 + assert result["0.500000,0.000000"][2] == pytest.approx(expected) + + changed_cell = [[2.1, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 10.0]] + dumpfn( + {"energies": [-9.0], "atom_numbs": [2], "cells": [changed_cell]}, + task1 / "result_task.json", + ) + with pytest.raises(RuntimeError, match="in-plane area changed"): + prop._compute_lower( + str(prop_dir / "result.json"), [str(task0), str(task1)], {} + ) + + def test_gamma_surface_refine_inherits_metadata_and_constraints(tmp_path): init_dir = tmp_path / "gamma_surface_00" init_task = init_dir / "task.000000" @@ -660,6 +852,14 @@ def test_gamma_surface_compute_lower_with_synthetic_results(self): with tempfile.TemporaryDirectory() as tmp: test_gamma_surface_compute_lower_with_synthetic_results(Path(tmp)) + def test_gamma_surface_compute_lower_nan_for_failed_task(self): + with tempfile.TemporaryDirectory() as tmp: + test_gamma_surface_compute_lower_nan_for_failed_task(Path(tmp)) + + def test_gamma_surface_compute_lower_fails_if_reference_task_failed(self): + with tempfile.TemporaryDirectory() as tmp: + test_gamma_surface_compute_lower_fails_if_reference_task_failed(Path(tmp)) + def test_gamma_surface_compute_lower_preserves_closed_loop_cartesian_data(self): with tempfile.TemporaryDirectory() as tmp: test_gamma_surface_compute_lower_preserves_closed_loop_cartesian_data( diff --git a/tests/test_gruneisen.py b/tests/test_gruneisen.py index 44b0d3dd..fee89e96 100644 --- a/tests/test_gruneisen.py +++ b/tests/test_gruneisen.py @@ -535,7 +535,7 @@ def test_post_process_prepares_phonon_run_inputs_for_lammps(self): deepmd_gruneisen.post_process([str(task_dir)]) rewritten = (task_dir / "in.lammps").read_text() - self.assertIn("plugin load libdeepmd_lmp.so", rewritten) + self.assertNotIn("plugin load libdeepmd_lmp.so", rewritten) self.assertIn("pair_style deepmd frozen_model.pth", rewritten) self.assertNotIn("run 0", rewritten) self.assertTrue((task_dir / "in.relax.lammps").is_file()) diff --git a/tests/test_gui_submit_builder.py b/tests/test_gui_submit_builder.py index 7d271674..1c4cfced 100644 --- a/tests/test_gui_submit_builder.py +++ b/tests/test_gui_submit_builder.py @@ -263,6 +263,36 @@ def test_profile_templates_have_different_property_options(self): self.assertIn("gamma_surface", vasp) self.assertIn("gamma_surface", abacus) + def test_profile_gamma_surface_defaults_use_primary_slip_systems(self): + expected_top_level = { + "plane_miller": [1, 1, 1], + "slip_direction": [-1, 1, 0], + } + expected_overrides = { + "bcc": { + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1], + }, + "hcp": { + "plane_miller": [0, 0, 0, 1], + "slip_direction": [2, -1, -1, 0], + }, + } + + for profile in ("lammps", "vasp", "abacus"): + with self.subTest(profile=profile): + template = _load_profile_param_template(profile) + surface = next( + prop + for prop in template["properties"] + if prop["type"] == "gamma_surface" + ) + for key, value in expected_top_level.items(): + self.assertEqual(surface[key], value) + for structure_type, expected in expected_overrides.items(): + self.assertEqual(surface[structure_type], expected) + self.assertTrue(surface["closed_loop"]) + def test_lammps_interaction_types_exclude_vasp_abacus(self): options = [item["value"] for item in _interaction_type_options_for_profile("lammps", "meam")] self.assertNotIn("vasp", options) @@ -392,22 +422,27 @@ def test_account_overwrite_hides_password_in_summary(self): email="user@example.com", password="secret-password", program_id_text="1234", + access_key="secret-access-key", account_path=account_path, ) self.assertTrue(feedback["ok"]) self.assertEqual(account_state["email"], "user@example.com") self.assertEqual(account_state["program_id"], "1234") self.assertTrue(account_state["password_set"]) + self.assertTrue(account_state["access_key_set"]) summary = _render_account_summary(account_state) self.assertIn("Email: user@example.com", summary) self.assertIn("Program ID: 1234", summary) self.assertIn("Password: 已设置", summary) + self.assertIn("AccessKey: 已设置", summary) self.assertNotIn("secret-password", summary) + self.assertNotIn("secret-access-key", summary) with open(account_path, "r", encoding="utf-8") as f: on_disk = json.load(f) self.assertEqual(on_disk["password"], "secret-password") + self.assertEqual(on_disk["access_key"], "secret-access-key") def test_account_overwrite_requires_integer_program_id(self): with tempfile.TemporaryDirectory() as tmpdir: @@ -430,6 +465,7 @@ def test_load_account_state_from_missing_file(self): self.assertEqual(state["email"], "") self.assertEqual(state["program_id"], "") self.assertFalse(state["password_set"]) + self.assertFalse(state["access_key_set"]) def test_read_latest_workflow_id(self): with tempfile.TemporaryDirectory() as tmpdir: diff --git a/tests/test_lammps.py b/tests/test_lammps.py index e07917c4..d50f1ce0 100644 --- a/tests/test_lammps.py +++ b/tests/test_lammps.py @@ -6,6 +6,7 @@ import tempfile import unittest import warnings +from pathlib import Path import dpdata import numpy as np @@ -87,6 +88,119 @@ def test_set_model_param(self): } self.assertEqual(model_param, self.Lammps.model_param) + def test_local_dpa4_pt2_model_is_linked_and_forwarded(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp).resolve() + model = root / "alloytongqi.pt2" + model.write_bytes(b"pt2-test-model") + output_dir = root / "case" / "relaxation" / "relax_task" + output_dir.mkdir(parents=True) + calculator = Lammps( + { + "type": "deepmd", + "model": str(model), + "type_map": {"Al": 0}, + "deepmd_runtime": "dpa4_pt2", + "deepmd_version": "3.2.0b0", + }, + os.path.join(self.source_path, "Al-fcc.vasp"), + ) + + calculator.set_model_param() + calculator.make_potential_files(str(output_dir)) + + linked_model = output_dir / model.name + self.assertTrue(linked_model.is_symlink()) + self.assertEqual(linked_model.read_bytes(), b"pt2-test-model") + self.assertEqual(calculator.model_param["model_name"], [model.name]) + self.assertEqual( + calculator.forward_files(), ["conf.lmp", "in.lammps", model.name] + ) + self.assertEqual( + calculator.forward_common_files(), ["in.lammps", model.name] + ) + + def test_image_resident_dpa4_pt2_model_keeps_absolute_path(self): + with tempfile.TemporaryDirectory() as tmp: + output_dir = Path(tmp).resolve() / "case" / "relaxation" / "relax_task" + output_dir.mkdir(parents=True) + model = "/opt/dpa4-runtime/models/DPA4-alloytongqi/alloytongqi.t4-sm75.pt2" + interaction = { + "type": "deepmd", + "model": model, + "model_in_image": True, + "type_map": {"Al": 0}, + "deepmd_runtime": "dpa4_pt2", + "deepmd_version": "3.2.0b0", + } + calculator = Lammps( + interaction, + os.path.join(self.source_path, "Al-fcc.vasp"), + ) + + calculator.set_model_param() + calculator.make_potential_files(str(output_dir)) + + self.assertEqual(calculator.model_param["model_name"], [model]) + self.assertIn(model, inter_deepmd(calculator.model_param)) + self.assertFalse((output_dir / Path(model).name).exists()) + self.assertEqual(loadfn(output_dir / "inter.json"), interaction) + self.assertEqual(calculator.forward_files(), ["conf.lmp", "in.lammps"]) + self.assertEqual(calculator.forward_common_files(), ["in.lammps"]) + + def test_image_resident_model_validation_fails_closed(self): + base = { + "type": "deepmd", + "model": "/opt/dpa4-runtime/model.pt2", + "model_in_image": True, + "type_map": {"Al": 0}, + "deepmd_runtime": "dpa4_pt2", + } + invalid_cases = [ + ({**base, "deepmd_runtime": "legacy"}, "only supported"), + ({**base, "model": "relative/model.pt2"}, "absolute .pt2"), + ({**base, "model": "/opt/dpa4-runtime/model.pb"}, "absolute .pt2"), + ({**base, "model_in_image": "true"}, "must be a boolean"), + ] + + for interaction, message in invalid_cases: + with self.subTest(interaction=interaction), self.assertRaisesRegex( + ValueError, message + ): + Lammps(interaction, os.path.join(self.source_path, "Al-fcc.vasp")) + + def test_custom_dpa4_pt2_input_rejects_legacy_plugin_load(self): + with tempfile.TemporaryDirectory(dir=".") as tmp: + root = Path(tmp).resolve() + output_dir = root / "relaxation" / "relax_task" + output_dir.mkdir(parents=True) + shutil.copy( + os.path.join(self.source_path, "Al-fcc.vasp"), + output_dir / "POSCAR", + ) + custom_input = root / "custom.in.lammps" + custom_input.write_text( + "clear\natom_style atomic\natom_modify map yes\n" + "read_data conf.lmp\nplugin load libdeepmd_lmp.so\n" + "pair_style deepmd local.pt2\npair_coeff * * Al\n" + ) + calculator = Lammps( + { + "type": "deepmd", + "model": str(root / "local.pt2"), + "in_lammps": str(custom_input), + "type_map": {"Al": 0}, + "deepmd_runtime": "dpa4_pt2", + "deepmd_version": "3.2.0b0", + }, + os.path.join(self.source_path, "Al-fcc.vasp"), + ) + + with self.assertRaisesRegex(ValueError, "LAMMPS_PLUGIN_PATH auto-loading"): + calculator.make_input_file( + str(output_dir), "relaxation", self.relax_param + ) + def test_make_potential_files(self): cwd = os.getcwd() abs_equi_path = os.path.abspath(self.equi_path) @@ -141,3 +255,29 @@ def test_compute_skips_annealing_without_relax_dump_warning(self): self.assertFalse( any("dump.relax" in str(item.message) for item in caught) ) + + def test_compute_reports_failed_runtime_status_before_parsing_dump(self): + with tempfile.TemporaryDirectory() as tmpdir: + dumpfn({"cal_type": "relaxation"}, os.path.join(tmpdir, "task.json")) + dumpfn( + { + "state": "failed", + "reason": "nonzero_lammps_error", + "exit_code": 1, + "message": "Command exited with non-zero code 1.", + }, + os.path.join(tmpdir, "apex_task_status.json"), + ) + with self.assertRaisesRegex( + RuntimeError, + "LAMMPS task failed before post-processing.*exit_code=1", + ): + self.Lammps.compute(tmpdir) + + def test_parse_dump_file_rejects_empty_dump(self): + with tempfile.TemporaryDirectory() as tmpdir: + dump_path = os.path.join(tmpdir, "dump.relax") + with open(dump_path, "w", encoding="utf-8"): + pass + with self.assertRaisesRegex(RuntimeError, "no TIMESTEP frames"): + self.Lammps._parse_dump_file(dump_path, [], [], [], []) diff --git a/tests/test_lammps_utils.py b/tests/test_lammps_utils.py index d62f7d5f..ab708dd0 100644 --- a/tests/test_lammps_utils.py +++ b/tests/test_lammps_utils.py @@ -10,6 +10,11 @@ TYPE_MAP = {"Al": 0} PARAM = {"type": "deepmd"} +DPA4_PT2_PARAM = { + "type": "deepmd", + "deepmd_runtime": "dpa4_pt2", + "deepmd_version": "3.2.0b0", +} def dummy_interaction(param): @@ -22,6 +27,7 @@ def assert_common_lammps_setup(script): assert "dimension" in script assert "boundary" in script assert "atom_style" in script + assert "atom_modify map yes" in script assert "box tilt large" in script assert "read_data conf.lmp" in script assert "mass 1 26.982" in script @@ -37,6 +43,105 @@ def assert_common_lammps_setup(script): assert 'print "Final Stress (xx yy zz xy xz yz) = ${Pxx} ${Pyy} ${Pzz} ${Pxy} ${Pxz} ${Pyz}"' in script +def assert_atom_map_before_every_box_load(script): + lines = script.splitlines() + box_load_lines = [ + index + for index, line in enumerate(lines) + if line.split()[:1] in (["read_data"], ["read_restart"]) + ] + assert box_load_lines + for index in box_load_lines: + assert lines[index - 1].strip() == "atom_modify map yes" + + +def dpa4_pt2_builders(): + return [ + lambda: lammps_utils.make_lammps_eval( + "conf.lmp", TYPE_MAP, dummy_interaction, DPA4_PT2_PARAM + ), + lambda: lammps_utils.make_lammps_equi( + "conf.lmp", + TYPE_MAP, + dummy_interaction, + DPA4_PT2_PARAM, + prop_type="relaxation", + ), + lambda: lammps_utils.make_lammps_elastic( + "conf.lmp", TYPE_MAP, dummy_interaction, DPA4_PT2_PARAM + ), + lambda: lammps_utils.make_lammps_press_relax( + "conf.lmp", TYPE_MAP, 0.95, dummy_interaction, DPA4_PT2_PARAM + ), + lambda: lammps_utils.make_lammps_FiniteTlatt( + "conf.lmp", TYPE_MAP, dummy_interaction, DPA4_PT2_PARAM + ), + lambda: lammps_utils.make_lammps_annealing( + "conf.lmp", TYPE_MAP, dummy_interaction, DPA4_PT2_PARAM, {} + ), + ] + + +@pytest.mark.parametrize("builder", dpa4_pt2_builders()) +def test_dpa4_pt2_builders_enable_atom_map_before_read_data(builder): + script = builder() + + assert_atom_map_before_every_box_load(script) + + +def test_dpa4_pt2_finite_t_elastic_enables_atom_map(tmp_path): + with open(tmp_path / "FiniteTelastic.json", "w") as fp: + json.dump({"role": "reference"}, fp) + + script = lammps_utils.make_lammps_FiniteTelastic( + "conf.lmp", + TYPE_MAP, + dummy_interaction, + DPA4_PT2_PARAM, + tmp_path, + ) + + assert_atom_map_before_every_box_load(script) + + +def test_atom_map_rewriter_is_per_clear_block_and_legacy_is_unchanged(): + source = "clear\nread_data first.lmp\nclear\nread_restart second.restart\n" + + rewritten = lammps_utils.ensure_atom_map_before_read_data( + source, DPA4_PT2_PARAM + ) + + assert rewritten.count("atom_modify map yes") == 2 + assert_atom_map_before_every_box_load(rewritten) + assert lammps_utils.ensure_atom_map_before_read_data(source, PARAM) == source + + +@pytest.mark.parametrize("map_style", ["yes", "array", "hash"]) +def test_atom_map_rewriter_accepts_existing_map_styles_idempotently(map_style): + source = ( + "clear\n" + f"atom_modify map {map_style} # already enables an atom map\n" + "read_data conf.lmp\n" + ) + + assert ( + lammps_utils.ensure_atom_map_before_read_data(source, DPA4_PT2_PARAM) + == source + ) + + +def test_dpa4_pt2_rejects_explicit_legacy_plugin_load(): + source = ( + "clear\n" + "atom_modify map yes\n" + "read_data conf.lmp\n" + "plugin load /legacy/libdeepmd_lmp.so\n" + ) + + with pytest.raises(ValueError, match="LAMMPS_PLUGIN_PATH auto-loading"): + lammps_utils.ensure_atom_map_before_read_data(source, DPA4_PT2_PARAM) + + def test_element_list_orders_by_lammps_type_id(): assert lammps_utils.element_list({"O": 1, "Al": 0, "Ti": 2}) == [ "Al", @@ -228,6 +333,7 @@ def test_make_lammps_finite_t_latt_default_cal_setting(): ) assert "include variable_FiniteTlatt.in" in script + assert "atom_modify map yes" in script assert "read_data conf.lmp" in script assert "replicate ${nx} ${ny} ${nz}" in script assert "pair_style dummy" in script @@ -306,6 +412,7 @@ def test_make_lammps_finite_t_elastic_equi_role(tmp_path): script = make_finite_t_elastic_input(tmp_path, "equi") assert "clear\ninclude variable_FiniteTelastic.in" in script + assert script.count("atom_modify map yes") == 1 assert "read_data conf.lmp" in script assert "replicate ${nx} ${ny} ${nz}" in script assert "pair_style dummy" in script @@ -343,6 +450,7 @@ def test_make_lammps_finite_t_elastic_response_roles(tmp_path, role): script = make_finite_t_elastic_input(tmp_path, role) assert script.count("clear\ninclude variable_FiniteTelastic.in") == 2 + assert script.count("atom_modify map yes") == 2 assert "read_data conf.lmp" in script assert "read_restart ${restart_source}" in script assert "write_restart ${equi_restart}" in script @@ -393,6 +501,7 @@ def make_annealing_input(cal_setting): def assert_common_annealing_script(script): assert "include variable_Annealing.in" in script + assert "atom_modify map yes" in script assert "read_data conf.lmp" in script assert "replicate ${nx} ${ny} ${nz}" in script assert "pair_style dummy" in script @@ -480,6 +589,28 @@ def test_make_lammps_annealing_langevin_nve(): class TestLammpsUtils(unittest.TestCase): + def test_dpa4_pt2_builders_enable_atom_map_before_read_data(self): + for builder in dpa4_pt2_builders(): + with self.subTest(builder=builder): + test_dpa4_pt2_builders_enable_atom_map_before_read_data(builder) + + def test_dpa4_pt2_finite_t_elastic_enables_atom_map(self): + with tempfile.TemporaryDirectory() as tmp: + test_dpa4_pt2_finite_t_elastic_enables_atom_map(Path(tmp)) + + def test_atom_map_rewriter_is_per_clear_block_and_legacy_is_unchanged(self): + test_atom_map_rewriter_is_per_clear_block_and_legacy_is_unchanged() + + def test_atom_map_rewriter_accepts_existing_map_styles_idempotently(self): + for map_style in ["yes", "array", "hash"]: + with self.subTest(map_style=map_style): + test_atom_map_rewriter_accepts_existing_map_styles_idempotently( + map_style + ) + + def test_dpa4_pt2_rejects_explicit_legacy_plugin_load(self): + test_dpa4_pt2_rejects_explicit_legacy_plugin_load() + def test_element_list_orders_by_lammps_type_id(self): test_element_list_orders_by_lammps_type_id() diff --git a/tests/test_melting_point.py b/tests/test_melting_point.py new file mode 100644 index 00000000..624de8fa --- /dev/null +++ b/tests/test_melting_point.py @@ -0,0 +1,376 @@ +import json +import os +import shutil +import tempfile +import unittest + +from monty.serialization import loadfn + +from apex.core.common_prop import make_property_instance +from apex.core.property.MeltingPoint import ( + MeltingPoint, + _aggregate_temperatures, + _infer_bracket, + _snapshot_projection, + render_melting_point_lammps_input, +) +from apex.core.calculator.lib import lammps_utils +from apex.core.calculator.Lammps import Lammps +from apex.reporter.DashReportApp import return_prop_class, return_prop_type +from apex.reporter.property_report import MeltingPointReport + + +POSCAR = """Ti +1.0 +3.0 0.0 0.0 +0.0 3.0 0.0 +0.0 0.0 6.0 +Ti +4 +Direct +0.0 0.0 0.10 +0.5 0.5 0.35 +0.0 0.5 0.65 +0.5 0.0 0.85 +""" + + +class TestMeltingPointProperty(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.equi = os.path.join(self.tmp.name, "relaxation", "relax_task") + os.makedirs(self.equi) + with open(os.path.join(self.equi, "CONTCAR"), "w") as fp: + fp.write(POSCAR) + self.work = os.path.join(self.tmp.name, "melting_point_00") + self.param = { + "type": "melting_point", + "method": "two_phase", + "supercell_size": [1, 1, 2], + "cal_setting": { + "temperature": [1600, 1700], + "replicas": 2, + "premelt_steps": 2, + "conditioning_steps": 2, + "production_steps": 10, + "dump_step": 1, + "thermo_step": 1, + "restart_interval": 2, + "timestep": 1.0, + }, + } + + def tearDown(self): + self.tmp.cleanup() + + def test_registered_and_generates_temperature_replica_matrix(self): + prop = make_property_instance(self.param, {"type": "deepmd"}) + self.assertIsInstance(prop, MeltingPoint) + tasks = prop.make_confs(self.work, self.equi) + self.assertEqual(4, len(tasks)) + metadata = [loadfn(os.path.join(task, "MeltingPoint.json")) for task in tasks] + self.assertEqual([1600.0, 1600.0, 1700.0, 1700.0], [row["temperature_K"] for row in metadata]) + self.assertNotEqual(metadata[0]["velocity_seeds"], metadata[1]["velocity_seeds"]) + self.assertEqual(4, metadata[0]["release_step"]) + self.assertEqual(2, metadata[0]["restart_interval"]) + with open(os.path.join(tasks[0], "variable_MeltingPoint.in")) as fp: + self.assertIn("variable restart_interval equal 2", fp.read()) + + def test_renderer_matches_validated_two_phase_protocol(self): + prop = MeltingPoint(self.param, {"type": "deepmd"}) + text = render_melting_point_lammps_input( + "conf.lmp", + {"Ti": 0}, + lammps_utils.inter_deepmd, + { + "type": "deepmd", + "model_name": ["model.pb"], + "param_type": {"Ti": 0}, + "deepmd_version": "2.1.1", + }, + prop.task_param(), + ) + self.assertIn("compute q6 all orientorder/atom", text) + self.assertIn("fix pin_solid solid_seed setforce", text) + self.assertIn("fix melt_liquid liquid_seed nvt", text) + self.assertIn("unfix pin_solid", text) + self.assertIn("fix condition_all all nvt", text) + self.assertNotIn("fix condition_liquid liquid_seed nvt", text) + self.assertIn("fix coexistence all npt", text) + self.assertIn("iso ${target_pressure} ${target_pressure} ${pdamp}", text) + self.assertIn("dump.melting id type xs ys zs c_q6[1]", text) + self.assertIn( + "restart ${restart_interval} restart.melting.1 restart.melting.2", + text, + ) + self.assertIn("write_restart restart.melting.final", text) + + def test_lammps_calculator_writes_property_input_and_manifests(self): + model = os.path.join(self.tmp.name, "model.pb") + with open(model, "wb") as fp: + fp.write(b"test model placeholder") + prop = MeltingPoint(self.param, {"type": "deepmd"}) + task = prop.make_confs(self.work, self.equi)[0] + calculator = Lammps( + { + "type": "deepmd", + "model": model, + "type_map": {"Ti": 0}, + }, + os.path.join(task, "POSCAR"), + ) + calculator.make_input_file(task, "melting_point", prop.task_param()) + with open(os.path.join(task, "in.lammps")) as fp: + text = fp.read() + self.assertIn("APEX_MELTING_STAGE coexistence_release", text) + self.assertEqual( + ["in.lammps", "variable_MeltingPoint.in", "MeltingPoint.json", "model.pb"], + calculator.forward_files("melting_point"), + ) + self.assertIn("dump.melting", calculator.backward_files("melting_point")) + self.assertIn( + "restart.melting.*", calculator.backward_files("melting_point") + ) + + def test_restart_files_are_copied_per_temperature_and_forwarded(self): + restart_dir = os.path.join(self.tmp.name, "restart_inputs") + os.makedirs(restart_dir) + restart_files = [] + for temperature in self.param["cal_setting"]["temperature"]: + path = os.path.join(restart_dir, f"restart.{temperature}") + with open(path, "wb") as fp: + fp.write(str(temperature).encode("ascii")) + restart_files.append(path) + + parameter = json.loads(json.dumps(self.param)) + parameter["cal_setting"]["restart_files"] = restart_files + prop = MeltingPoint(parameter, {"type": "deepmd"}) + tasks = prop.make_confs(self.work, self.equi) + + for index, task in enumerate(tasks): + temperature_index = index // parameter["cal_setting"]["replicas"] + with open(os.path.join(task, "restart.coexistence.start"), "rb") as fp: + self.assertEqual( + str(parameter["cal_setting"]["temperature"][temperature_index]).encode("ascii"), + fp.read(), + ) + + model = os.path.join(self.tmp.name, "restart-model.pb") + with open(model, "wb") as fp: + fp.write(b"test model placeholder") + calculator = Lammps( + {"type": "deepmd", "model": model, "type_map": {"Ti": 0}}, + os.path.join(tasks[0], "POSCAR"), + ) + self.assertIn( + "restart.coexistence.start", + calculator.forward_files("melting_point", prop.task_param()), + ) + self.assertNotIn( + "restart.coexistence.start", + calculator.forward_files("finite_t_latt", prop.task_param()), + ) + + text = render_melting_point_lammps_input( + "conf.lmp", + {"Ti": 0}, + lammps_utils.inter_deepmd, + { + "type": "deepmd", + "model_name": ["model.pb"], + "param_type": {"Ti": 0}, + "deepmd_version": "2.1.1", + }, + prop.task_param(), + ) + self.assertIn("read_restart restart.coexistence.start", text) + self.assertIn("reset_timestep 0", text) + self.assertIn("run 0", text) + self.assertNotIn("read_data conf.lmp", text) + self.assertNotIn("run ${premelt_steps}", text) + metadata = loadfn(os.path.join(tasks[0], "MeltingPoint.json")) + self.assertTrue(metadata["restart_mode"]) + self.assertEqual(0, metadata["reference_step"]) + self.assertEqual(0, metadata["release_step"]) + + def test_restart_files_require_one_existing_file_per_temperature(self): + parameter = json.loads(json.dumps(self.param)) + parameter["cal_setting"]["restart_files"] = ["only-one-restart"] + with self.assertRaisesRegex(ValueError, "one entry per temperature"): + MeltingPoint(parameter, {"type": "deepmd"}).make_confs( + self.work, self.equi + ) + + def test_rejects_dft_and_invalid_axis(self): + with self.assertRaises(NotImplementedError): + MeltingPoint(self.param, {"type": "vasp"}) + bad = json.loads(json.dumps(self.param)) + bad["cal_setting"]["interface_axis"] = "w" + with self.assertRaises(ValueError): + MeltingPoint(bad, {"type": "deepmd"}) + + def test_rejects_non_positive_velocity_seeds(self): + for invalid_seed in (0, -1): + bad = json.loads(json.dumps(self.param)) + bad["cal_setting"]["velocity_seeds"] = { + "premelt": invalid_seed, + "condition": 2, + "release": 3, + } + with self.subTest(seed=invalid_seed), self.assertRaisesRegex( + ValueError, "positive integers" + ): + MeltingPoint(bad, {"type": "deepmd"}) + + def test_rejects_fractional_integer_settings(self): + for field, value in ( + ("supercell_size", [1.5, 1, 2]), + ("production_steps", 10.5), + ("dump_step", True), + ): + bad = json.loads(json.dumps(self.param)) + if field == "supercell_size": + bad[field] = value + else: + bad["cal_setting"][field] = value + with self.subTest(field=field), self.assertRaises(ValueError): + MeltingPoint(bad, {"type": "deepmd"}) + + bad = json.loads(json.dumps(self.param)) + bad["cal_setting"]["velocity_seeds"] = { + "premelt": 1.5, + "condition": 2, + "release": 3, + } + with self.assertRaisesRegex(ValueError, "positive integers"): + MeltingPoint(bad, {"type": "deepmd"}) + + def test_snapshot_projection_contains_requested_interface_axis(self): + self.assertEqual((0, 2, "fractional x", "fractional z"), _snapshot_projection("z")) + self.assertEqual((0, 1, "fractional x", "fractional y"), _snapshot_projection("y")) + self.assertEqual((1, 0, "fractional y", "fractional x"), _snapshot_projection("x")) + + +class TestMeltingPointAnalysis(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.task = self.tmp.name + self.meta = { + "schema": "apex.melting_point.task/v1", + "property": "melting_point", + "method": "two_phase", + "temperature_K": 1600.0, + "replica": 1, + "interface_axis": "z", + "liquid_fraction": 0.5, + "premelt_steps": 2, + "conditioning_steps": 2, + "release_step": 4, + "production_steps": 10, + "timestep_ps": 1.0, + "analysis_stride": 1, + "analysis_block_ps": 2.0, + "minimum_q6_gap": 0.03, + "minimum_directional_change": 0.02, + "spatial_bins": 4, + "velocity_seeds": {"premelt": 1, "condition": 2, "release": 3}, + } + with open(os.path.join(self.task, "MeltingPoint.json"), "w") as fp: + json.dump(self.meta, fp) + + def tearDown(self): + self.tmp.cleanup() + + @staticmethod + def _write_frame(fp, step, q6): + coords = [(0.1, 0.1, 0.10), (0.6, 0.6, 0.35), (0.1, 0.6, 0.65), (0.6, 0.1, 0.85)] + fp.write("ITEM: TIMESTEP\n%d\n" % step) + fp.write("ITEM: NUMBER OF ATOMS\n4\n") + fp.write("ITEM: BOX BOUNDS pp pp pp\n0 3\n0 3\n0 6\n") + fp.write("ITEM: ATOMS id type xs ys zs c_q6[1]\n") + for index, ((x, y, z), value) in enumerate(zip(coords, q6), 1): + fp.write(f"{index} 1 {x} {y} {z} {value}\n") + + def test_q6_interface_motion_and_bracket(self): + with open(os.path.join(self.task, "dump.melting"), "w") as fp: + self._write_frame(fp, 0, [0.5, 0.5, 0.5, 0.5]) + self._write_frame(fp, 2, [0.5, 0.5, 0.1, 0.1]) + for step in range(4, 15): + liquid_q6 = 0.1 + 0.035 * (step - 4) + self._write_frame(fp, step, [0.5, 0.5, liquid_q6, liquid_q6]) + with open(os.path.join(self.task, "log.lammps"), "w") as fp: + fp.write("Step Temp Press PotEng KinEng TotEng\n") + for step in range(4, 15): + fp.write(f"{step} 1600 0 -10 2 -8\n") + prop = MeltingPoint( + { + "type": "melting_point", + "cal_setting": { + "temperature": [1600], + "premelt_steps": 2, + "conditioning_steps": 2, + "production_steps": 10, + "dump_step": 1, + "thermo_step": 1, + "timestep": 1.0, + }, + }, + {"type": "deepmd"}, + ) + result, _ = prop._compute_lower( + os.path.join(self.tmp.name, "result.json"), [self.task], [] + ) + self.assertEqual([], result["failed_tasks"]) + point = result["points"][0] + self.assertEqual("solid_growth", point["interface_motion"]["outcome"]) + self.assertGreater(point["interface_motion"]["interface_velocity_A_per_ps"], 0) + self.assertAlmostEqual(0.4, point["reference_q6_gap"]) + bracket = _infer_bracket([ + {"temperature_K": 1600, "consensus_outcome": "solid_growth"}, + {"temperature_K": 1700, "consensus_outcome": "liquid_growth"}, + ]) + self.assertEqual("bracketed", bracket["status"]) + self.assertEqual(1650, bracket["estimated_melting_temperature_K"]) + + def test_incomplete_replicas_cannot_form_a_bracket(self): + points = [ + { + "temperature_K": 1600, + "replica": 1, + "interface_motion": { + "outcome": "solid_growth", + "interface_velocity_A_per_ps": 0.1, + }, + }, + { + "temperature_K": 1700, + "replica": 1, + "interface_motion": { + "outcome": "liquid_growth", + "interface_velocity_A_per_ps": -0.1, + }, + }, + ] + rows = _aggregate_temperatures( + points, + expected_replicas=2, + expected_temperatures=[1600, 1700, 1800], + ) + + self.assertEqual(["inconclusive"] * 3, [ + row["consensus_outcome"] for row in rows + ]) + self.assertFalse(any(row["replicas_complete"] for row in rows)) + self.assertEqual(0, rows[-1]["replica_count"]) + self.assertIsNone(rows[-1]["interface_velocity_mean_A_per_ps"]) + self.assertEqual("inconclusive_or_unbracketed", _infer_bracket(rows)["status"]) + + +class TestMeltingPointReporter(unittest.TestCase): + def test_reporter_registration(self): + self.assertEqual("melting_point", return_prop_type("melting_point_00")) + self.assertIs(MeltingPointReport, return_prop_class("melting_point")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_ops.py b/tests/test_ops.py index d1630f97..73cc1e2e 100644 --- a/tests/test_ops.py +++ b/tests/test_ops.py @@ -3,6 +3,7 @@ import os import glob import shutil +import subprocess import tempfile from types import SimpleNamespace from unittest.mock import patch @@ -14,21 +15,37 @@ Artifact, TransientError, ) -from monty.serialization import loadfn +from monty.serialization import dumpfn, loadfn from apex.op.relaxation_ops import RelaxMake, _check_relaxation_outputs -from apex.op.property_ops import PropsMake, PropsPost, PropsRepairStatusCheck, _is_failed_task_status +from apex.op.property_ops import ( + PropsMake, + PropsPost, + PropsRepairStatusCheck, + TASK_FAILURE_TOLERANT_TYPES, + _is_failed_task_status, +) from apex.op.RunLAMMPS import RunLAMMPS +from apex.op.RunVASP import RunVASP +from apex.core.calculator import lammps_model_files_for_cleanup from apex.superop.SimplePropertySteps import SimplePropertySteps +from apex.core.lib.vasp_runtime import build_kpoint_aware_vasp_command +from apex.core.lib import dispatcher as dispatcher_module from apex.task_failure import ( REMOTE_LAMMPS_STARTUP_FAILURE, + TRANSIENT_LAMMPS_RETRY_REASON, classify_apex_task_status, classify_lammps_exit_code, is_header_only_lammps_failure, is_lammps_header_only_log, load_and_classify_task_status, ) -from apex.utils import apex_task_succeeded, all_apex_task_status_succeeded +from apex.utils import ( + all_apex_task_status_succeeded, + apex_task_succeeded, + get_task_type, +) +from apex.core.property.Property import is_failed_task_result try: from context import write_poscar except ModuleNotFoundError: @@ -39,6 +56,58 @@ class TestTaskStatusHelpers(unittest.TestCase): + def test_image_resident_model_is_excluded_from_cleanup(self): + image_model = "/opt/dpa4-runtime/models/alloytongqi.pt2" + self.assertEqual( + lammps_model_files_for_cleanup( + {"model": image_model, "model_in_image": True} + ), + [], + ) + self.assertEqual( + lammps_model_files_for_cleanup({"model": "local.pt2"}), + ["local.pt2"], + ) + + def test_is_failed_task_result_accepts_dict_and_mapping_like(self): + self.assertTrue(is_failed_task_result(None)) + self.assertTrue(is_failed_task_result({"failed": True, "energies": [1.0]})) + self.assertTrue(is_failed_task_result({"atom_numbs": [1]})) + self.assertFalse(is_failed_task_result({"energies": [-1.0], "atom_numbs": [1]})) + + class MappingLike: + def __getitem__(self, key): + if key == "energies": + return [-1.0] + raise KeyError(key) + + self.assertFalse(is_failed_task_result(MappingLike())) + + # Calculator compute() returns dpdata as_dict with nested data.energies + as_dict = { + "@module": "dpdata.system", + "@class": "LabeledSystem", + "data": { + "atom_numbs": [2], + "energies": { + "@module": "numpy", + "@class": "array", + "dtype": "float64", + "data": [-1.0], + }, + }, + } + self.assertFalse(is_failed_task_result(as_dict)) + self.assertTrue( + is_failed_task_result( + { + "@module": "dpdata.system", + "@class": "LabeledSystem", + "data": {"atom_numbs": [2]}, + } + ) + ) + def test_task_failure_helpers_cover_error_branches(self): with tempfile.TemporaryDirectory() as tmpdir: task_dir = Path(tmpdir) @@ -176,7 +245,7 @@ def test_props_post_reports_lammps_status_failures_before_compute(self): root = Path(tmpdir) input_all = root / "all" input_post = root / "post" - prop_dir = input_post / "confs" / "std-bcc" / "eos_00" + prop_dir = input_post / "confs" / "std-bcc" / "elastic_00" task_dir = prop_dir / "task.000000" (input_all / "confs").mkdir(parents=True) task_dir.mkdir(parents=True) @@ -186,6 +255,44 @@ def test_props_post_reports_lammps_status_failures_before_compute(self): try: with self.assertRaisesRegex(RuntimeError, "LAMMPS failed for property task"): + PropsPost().execute(OPIO({ + "input_post": input_post, + "input_all": input_all, + "prop_param": {"type": "elastic"}, + "inter_param": {"type": "deepmd", "model": "model.pb"}, + "task_names": ["confs/std-bcc/elastic_00/task.000000"], + "path_to_prop": "confs/std-bcc/elastic_00", + })) + finally: + os.chdir(cwd) + + def test_props_post_tolerates_failed_tasks_for_eos(self): + with tempfile.TemporaryDirectory() as tmpdir: + cwd = os.getcwd() + root = Path(tmpdir) + input_all = root / "all" + input_post = root / "post" + prop_dir = input_post / "confs" / "std-bcc" / "eos_00" + task_dir = prop_dir / "task.000000" + (input_all / "confs").mkdir(parents=True) + task_dir.mkdir(parents=True) + (task_dir / "apex_task_status.json").write_text( + '{"state": "failed", "reason": "nonzero_exit", "exit_code": 7}' + ) + + class FakeProp: + parameter = {"type": "eos"} + + def compute(self, output_file, print_file, path_to_work): + dumpfn({10.0: float("nan")}, output_file) + Path(print_file).write_text("ok\n") + + self.assertIn("eos", TASK_FAILURE_TOLERANT_TYPES) + try: + with patch( + "apex.core.common_prop.make_property_instance", + return_value=FakeProp(), + ): PropsPost().execute(OPIO({ "input_post": input_post, "input_all": input_all, @@ -194,11 +301,92 @@ def test_props_post_reports_lammps_status_failures_before_compute(self): "task_names": ["confs/std-bcc/eos_00/task.000000"], "path_to_prop": "confs/std-bcc/eos_00", })) + candidates = list(Path(tmpdir).rglob("failed_lammps_tasks.json")) + self.assertTrue(candidates, "failed_lammps_tasks.json was not written") + payload = loadfn(candidates[0]) + self.assertEqual(len(payload["failed_tasks"]), 1) finally: os.chdir(cwd) class TestSimplePropertySteps(unittest.TestCase): + def test_dispatcher_applies_kpoint_selector_to_vasp_tasks(self): + captured = {} + + def fake_task(**kwargs): + captured.update(kwargs) + return SimpleNamespace(**kwargs) + + with tempfile.TemporaryDirectory() as tmpdir: + task_dir = Path(tmpdir) / "task.000000" + task_dir.mkdir() + (task_dir / "KPOINTS").write_text( + "Automatic mesh\n0\nGamma\n1 1 1\n0 0 0\n" + ) + with patch.object( + dispatcher_module.Machine, + "load_from_dict", + return_value=object(), + ), patch.object( + dispatcher_module.Resources, + "load_from_dict", + return_value=object(), + ), patch.object( + dispatcher_module, "Task", fake_task + ), patch.object( + dispatcher_module, + "Submission", + side_effect=lambda **kwargs: SimpleNamespace(**kwargs), + ): + dispatcher_module.make_submission( + mdata_machine={}, + mdata_resources={}, + commands=["mpirun -n 4 /opt/vasp/bin/vasp_std"], + work_path=tmpdir, + run_tasks=["task.000000"], + group_size=1, + forward_common_files=[], + forward_files=[], + backward_files=[], + outlog="outlog", + errlog="errlog", + ) + self.assertIn("vasp_gam", captured["command"]) + self.assertIn("vasp_std", captured["command"]) + self.assertIn("KPOINTS", captured["command"]) + + def test_vasp_runtime_selects_executable_from_actual_kpoints(self): + command = build_kpoint_aware_vasp_command( + "printf vasp_std > selected" + ) + with tempfile.TemporaryDirectory() as tmpdir: + task_dir = Path(tmpdir) + gamma = ( + "Automatic mesh\n0\nGamma\n1 1 1\n0 0 0\n" + ) + (task_dir / "KPOINTS").write_text(gamma) + subprocess.run( + ["bash", "-c", command], cwd=task_dir, check=True + ) + self.assertEqual( + (task_dir / "selected").read_text(), "vasp_gam" + ) + + non_gamma = ( + "Automatic mesh\n0\nGamma\n2 1 1\n0 0 0\n" + ) + (task_dir / "KPOINTS").write_text(non_gamma) + subprocess.run( + ["bash", "-c", command], cwd=task_dir, check=True + ) + self.assertEqual( + (task_dir / "selected").read_text(), "vasp_std" + ) + + def test_vasp_runtime_requires_switchable_executable(self): + with self.assertRaisesRegex(ValueError, "vasp_std or vasp_gam"): + build_kpoint_aware_vasp_command("mpirun vasp_ncl") + def test_lammps_repair_step_feeds_checked_post_to_post_step(self): import apex.superop.SimplePropertySteps as simple_steps @@ -230,6 +418,7 @@ def __init__(self, name, template=None, artifacts=None, parameters=None, parameters={ "task_names": f"{name}-task_names", "njobs": f"{name}-njobs", + "backward_list": f"{name}-backward_list", }, ) @@ -292,8 +481,685 @@ def fake_add(self, step): "Props-post-retrieve_path", ) + added_steps.clear() + with patch.object(simple_steps, "Step", FakeStep), \ + patch.object(simple_steps, "PythonOPTemplate", FakeTemplate), \ + patch.object(simple_steps, "Slices", lambda *args, **kwargs: ("slices", args, kwargs)), \ + patch.object(simple_steps, "argo_range", lambda value: f"range:{value}"), \ + patch.object(simple_steps, "argo_len", lambda value: f"len:{value}"), \ + patch.object(SimplePropertySteps, "add", fake_add): + obj._build( + "step", + make_op=object(), + run_op=object(), + post_op=object(), + make_image="make-image", + run_image="run-image", + post_image="post-image", + run_command="mpirun vasp_std", + calculator="vasp", + upload_python_packages=[], + ) + + vasp_run = next(step for step in added_steps if step.name == "PropsVASP-Cal") + self.assertEqual( + vasp_run.parameters["backward_list"], + "Props-make-backward_list", + ) + self.assertEqual(vasp_run.parameters["log_name"], "outlog") + self.assertIn( + "APEX_RUN_COMMAND=", + vasp_run.parameters["run_image_config"]["command"], + ) + self.assertIn( + "vasp_gam", + vasp_run.parameters["run_image_config"]["command"], + ) + self.assertIn( + "vasp_std", + vasp_run.parameters["run_image_config"]["command"], + ) + + +class TestRunVASP(unittest.TestCase): + @staticmethod + def _write_common_inputs(task_dir): + (task_dir / "POSCAR").write_text("original-poscar\n") + (task_dir / "INCAR").write_text("NSW = 1\n") + (task_dir / "POTCAR").write_text("potcar\n") + (task_dir / "KPOINTS").write_text( + "Automatic mesh\n0\nGamma\n1 1 1\n0 0 0\n" + ) + (task_dir / "fake_vasp.py").write_text( + "from pathlib import Path\n" + "import re\n" + "incar = Path('INCAR').read_text()\n" + "match = re.search(r'NSW\\s*=\\s*(\\d+)', incar)\n" + "nsw = match.group(1) if match else 'unset'\n" + "step_count = int(nsw) if nsw != 'unset' else 0\n" + "with Path('calls.txt').open('a') as stream:\n" + " stream.write(nsw + '\\n')\n" + "Path('OUTCAR').write_text(\n" + " ''.join(\n" + " ' POSITION " + "TOTAL-FORCE (eV/Angst)\\n'\n" + " for _ in range(step_count)\n" + " )\n" + " + 'General timing and accounting informations for this job:\\n'\n" + " + 'Total CPU time used (sec): 1.0\\n'\n" + " + 'Elapsed time (sec): 1.0\\n'\n" + " + 'Voluntary context switches: 1\\n'\n" + " + 'extra wrapper line after the VASP footer\\n'\n" + ")\n" + "Path('OSZICAR').write_text(\n" + " ''.join(f'{step} T= 300 E= 0\\n' " + "for step in range(1, step_count + 1))\n" + ")\n" + "Path('CONTCAR').write_text('contcar NSW=' + nsw + '\\n')\n" + "Path('XDATCAR').write_text('xdatcar NSW=' + nsw + '\\n')\n" + ) + + @staticmethod + def _op_input(task_dir, task_name, command, backward_list=None): + return OPIO({ + "task_name": task_name, + "task_path": task_dir, + "backward_list": backward_list or [ + "OUTCAR", "CONTCAR", "XDATCAR" + ], + "log_name": "outlog", + "backward_dir_name": "backward_dir", + "run_image_config": {"command": command}, + "optional_artifact": None, + "optional_input": {}, + }) + + def test_vasp_backend_is_wired_to_apex_run_op(self): + task_type, run_op = get_task_type({"interaction": {"type": "vasp"}}) + self.assertEqual(task_type, "vasp") + self.assertIs(run_op, RunVASP) + + def test_single_stage_vasp_still_runs(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + cwd = os.getcwd() + try: + os.chdir(root) + result = RunVASP().execute(self._op_input( + task_dir, "single", "python fake_vasp.py" + )) + finally: + os.chdir(cwd) + + backward = root / result["backward_dir"] + self.assertTrue((backward / "OUTCAR").is_file()) + status = loadfn(backward / "apex_vasp_stage_status.json") + self.assertEqual(status["state"], "succeeded") + self.assertEqual(status["task_type"], "single_stage") + self.assertTrue(status["stages"][0]["footer_complete"]) + self.assertEqual( + (root / "single" / "calls.txt").read_text().splitlines(), + ["1"], + ) + + def test_single_stage_recovers_mismatched_runtime_log_alias(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + op_input = self._op_input( + task_dir, + "single-log-alias", + "python fake_vasp.py", + ["OUTCAR", "outlog", "CONTCAR", "OSZICAR", "XDATCAR"], + ) + op_input["log_name"] = "runner.log" + cwd = os.getcwd() + try: + os.chdir(root) + result = RunVASP().execute(op_input) + finally: + os.chdir(cwd) + + backward = root / result["backward_dir"] + self.assertTrue((backward / "runner.log").is_file()) + self.assertTrue((backward / "outlog").is_file()) + self.assertEqual( + (backward / "runner.log").read_bytes(), + (backward / "outlog").read_bytes(), + ) + self.assertEqual( + loadfn(backward / "apex_vasp_stage_status.json")["state"], + "succeeded", + ) + + def test_finite_t_latt_runs_both_writable_incar_stages(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + original_poscar = (task_dir / "POSCAR").read_text() + (task_dir / "task.json").write_text( + '{"type": "finite_t_latt"}\n' + ) + (task_dir / "INCAR.equi").write_text("NSW = 100\n") + (task_dir / "INCAR.production").write_text("NSW = 300\n") + (task_dir / "run_command").write_text( + "set -e\n" + "cp INCAR.equi INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "mv OUTCAR OUTCAR.equi\n" + "[ ! -f XDATCAR ] || mv XDATCAR XDATCAR.equi\n" + "cp CONTCAR POSCAR\n" + "cp INCAR.production INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + ) + cwd = os.getcwd() + try: + os.chdir(root) + result = RunVASP().execute(self._op_input( + task_dir, + "finite", + "APEX_RUN_COMMAND='python fake_vasp.py' bash run_command", + )) + finally: + os.chdir(cwd) + + backward = root / result["backward_dir"] + self.assertEqual( + (root / "finite" / "calls.txt").read_text().splitlines(), + ["100", "300"], + ) + self.assertTrue((backward / "OUTCAR.equi").is_file()) + self.assertTrue((backward / "XDATCAR.equi").is_file()) + status = loadfn(backward / "apex_vasp_stage_status.json") + self.assertEqual(status["state"], "succeeded") + self.assertEqual( + [stage["name"] for stage in status["stages"]], + ["equi", "production"], + ) + self.assertEqual( + [stage["observed_ionic_steps"] for stage in status["stages"]], + [100, 300], + ) + self.assertTrue(all( + stage["footer_complete"] for stage in status["stages"] + )) + # Stage switching must mutate only the OP working copy. + self.assertEqual((task_dir / "INCAR").read_text(), "NSW = 1\n") + self.assertEqual((task_dir / "POSCAR").read_text(), original_poscar) + + def test_finite_t_latt_runs_nvt_before_npt_stages(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "task.json").write_text( + '{"type": "finite_t_latt"}\n' + ) + (task_dir / "INCAR.nvt").write_text("NSW = 50\nISIF = 2\n") + (task_dir / "INCAR.equi").write_text("NSW = 100\nISIF = 3\n") + (task_dir / "INCAR.production").write_text( + "NSW = 300\nISIF = 3\n" + ) + (task_dir / "run_command").write_text( + "set -e\n" + "cp INCAR.nvt INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "mv OUTCAR OUTCAR.nvt\n" + "[ ! -f XDATCAR ] || mv XDATCAR XDATCAR.nvt\n" + "cp CONTCAR POSCAR\n" + "cp INCAR.equi INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "mv OUTCAR OUTCAR.equi\n" + "[ ! -f XDATCAR ] || mv XDATCAR XDATCAR.equi\n" + "cp CONTCAR POSCAR\n" + "cp INCAR.production INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + ) + cwd = os.getcwd() + try: + os.chdir(root) + result = RunVASP().execute(self._op_input( + task_dir, + "finite-three-stage", + "APEX_RUN_COMMAND='python fake_vasp.py' bash run_command", + )) + finally: + os.chdir(cwd) + + backward = root / result["backward_dir"] + self.assertEqual( + ( + root / "finite-three-stage" / "calls.txt" + ).read_text().splitlines(), + ["50", "100", "300"], + ) + self.assertTrue((backward / "OUTCAR.nvt").is_file()) + self.assertTrue((backward / "OUTCAR.equi").is_file()) + status = loadfn(backward / "apex_vasp_stage_status.json") + self.assertEqual(status["state"], "succeeded") + self.assertEqual( + [stage["name"] for stage in status["stages"]], + ["nvt", "equi", "production"], + ) + self.assertEqual( + [stage["expected_ionic_steps"] for stage in status["stages"]], + [50, 100, 300], + ) + self.assertEqual( + [stage["observed_ionic_steps"] for stage in status["stages"]], + [50, 100, 300], + ) + + def test_failed_vasp_preserves_current_stage_evidence(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "task.json").write_text( + '{"type": "finite_t_latt"}\n' + ) + (task_dir / "INCAR.nvt").write_text("NSW = 50\nISIF = 2\n") + (task_dir / "INCAR.equi").write_text("NSW = 100\nISIF = 3\n") + (task_dir / "INCAR.production").write_text( + "NSW = 300\nISIF = 3\n" + ) + (task_dir / "fake_fail.py").write_text( + "from pathlib import Path\n" + "import re\n" + "incar = Path('INCAR').read_text()\n" + "nsw = int(re.search(r'NSW\\s*=\\s*(\\d+)', incar).group(1))\n" + "if nsw == 50:\n" + " Path('OUTCAR').write_text(\n" + " ''.join(' POSITION TOTAL-FORCE (eV/Angst)\\n' " + "for _ in range(nsw))\n" + " + 'General timing and accounting informations\\n'\n" + " + 'Total CPU time used (sec): 1\\n'\n" + " + 'Elapsed time (sec): 1\\n'\n" + " )\n" + " Path('OSZICAR').write_text(\n" + " ''.join(f'{step} T= 300 E= 0\\n' " + "for step in range(1, nsw + 1))\n" + " )\n" + " Path('CONTCAR').write_text('completed nvt structure\\n')\n" + " Path('XDATCAR').write_text('completed nvt trajectory\\n')\n" + "else:\n" + " Path('OUTCAR').write_text('partial ionic step\\n')\n" + " Path('OSZICAR').write_text('DAV: 1\\n')\n" + " Path('CONTCAR').write_text('partial structure\\n')\n" + " raise SystemExit(7)\n" + ) + (task_dir / "run_command").write_text( + "set -e\n" + "cp INCAR.nvt INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "mv OUTCAR OUTCAR.nvt\n" + "mv OSZICAR OSZICAR.nvt\n" + "cp CONTCAR CONTCAR.nvt\n" + "mv XDATCAR XDATCAR.nvt\n" + "cp CONTCAR POSCAR\n" + "cp INCAR.equi INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "mv OUTCAR OUTCAR.equi\n" + "cp CONTCAR POSCAR\n" + "cp INCAR.production INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + ) + dflow_tmp = root / "dflow-tmp" + (dflow_tmp / "inputs" / "artifacts").mkdir(parents=True) + (dflow_tmp / "outputs" / "artifacts").mkdir(parents=True) + op = RunVASP() + op.tmp_root = str(dflow_tmp) + cwd = os.getcwd() + try: + os.chdir(root) + with self.assertRaises(TransientError): + op.execute(self._op_input( + task_dir, + "finite-failed", + ( + "APEX_RUN_COMMAND='python fake_fail.py' " + "bash run_command" + ), + )) + finally: + os.chdir(cwd) + + local_evidence = ( + root / "finite-failed" / "backward_dir" + ) + packed_evidence = ( + dflow_tmp / "outputs" / "artifacts" / "backward_dir" + ) + for evidence in (local_evidence, packed_evidence): + self.assertTrue((evidence / "OUTCAR").is_file()) + self.assertTrue((evidence / "OSZICAR").is_file()) + self.assertTrue((evidence / "INCAR").is_file()) + self.assertTrue((evidence / "outlog").is_file()) + self.assertTrue((evidence / "OUTCAR.nvt").is_file()) + self.assertTrue((evidence / "OSZICAR.nvt").is_file()) + self.assertTrue((evidence / "CONTCAR.nvt").is_file()) + self.assertTrue((evidence / "XDATCAR.nvt").is_file()) + failure = loadfn(evidence / "apex_vasp_failure.json") + self.assertEqual( + failure["current_or_next_stage"], "equi" + ) + self.assertEqual(failure["error_type"], "TransientError") + status = loadfn( + evidence / "apex_vasp_stage_status.json" + ) + self.assertEqual(status["state"], "failed") + self.assertEqual( + status["missing_or_incomplete_stages"], + ["equi", "production"], + ) + + def test_finite_t_latt_rejects_short_stage_with_complete_footer(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "task.json").write_text( + '{"type": "finite_t_latt"}\n' + ) + (task_dir / "INCAR.equi").write_text("NSW = 3\n") + (task_dir / "INCAR.production").write_text("NSW = 5\n") + (task_dir / "fake_short.py").write_text( + "from pathlib import Path\n" + "Path('OUTCAR').write_text(\n" + " ' POSITION TOTAL-FORCE (eV/Angst)\\n'\n" + " 'General timing and accounting informations\\n'\n" + " 'Total CPU time used (sec): 1\\n'\n" + " 'Elapsed time (sec): 1\\n'\n" + ")\n" + "Path('OSZICAR').write_text('1 T= 300 E= 0\\n')\n" + "Path('CONTCAR').write_text('partial\\n')\n" + "Path('XDATCAR').write_text('partial\\n')\n" + ) + (task_dir / "run_command").write_text( + "set -e\n" + "cp INCAR.equi INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "mv OUTCAR OUTCAR.equi\n" + "mv OSZICAR OSZICAR.equi\n" + "cp CONTCAR POSCAR\n" + "cp INCAR.production INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + ) + cwd = os.getcwd() + try: + os.chdir(root) + with self.assertRaises(TransientError): + RunVASP().execute(self._op_input( + task_dir, + "finite-short", + "APEX_RUN_COMMAND='python fake_short.py' bash run_command", + )) + finally: + os.chdir(cwd) + + status = loadfn( + root / "finite-short" / "backward_dir" + / "apex_vasp_stage_status.json" + ) + self.assertEqual(status["state"], "failed") + self.assertEqual( + status["stages"][0]["observed_ionic_steps"], 1 + ) + self.assertIn( + "ionic_step_count_mismatch", + status["stages"][0]["failure_reasons"][0], + ) + + def test_single_stage_allows_early_convergence_with_normal_footer(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "INCAR").write_text("NSW = 20\n") + (task_dir / "fake_vasp.py").write_text( + (task_dir / "fake_vasp.py").read_text().replace( + "range(step_count)", "range(2)" + ) + ) + cwd = os.getcwd() + try: + os.chdir(root) + result = RunVASP().execute(self._op_input( + task_dir, "relax-early", "python fake_vasp.py" + )) + finally: + os.chdir(cwd) + + status = loadfn( + root / result["backward_dir"] + / "apex_vasp_stage_status.json" + ) + self.assertEqual(status["state"], "succeeded") + self.assertEqual( + status["stages"][0]["expected_ionic_steps"], 20 + ) + self.assertEqual( + status["stages"][0]["observed_ionic_steps"], 2 + ) + + def test_single_stage_rejects_missing_normal_footer(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "fake_no_footer.py").write_text( + "from pathlib import Path\n" + "Path('OUTCAR').write_text(' POSITION TOTAL-FORCE\\n')\n" + "Path('CONTCAR').write_text('partial\\n')\n" + "Path('XDATCAR').write_text('partial\\n')\n" + ) + cwd = os.getcwd() + try: + os.chdir(root) + with self.assertRaises(TransientError): + RunVASP().execute(self._op_input( + task_dir, "no-footer", "python fake_no_footer.py" + )) + finally: + os.chdir(cwd) + + status = loadfn( + root / "no-footer" / "backward_dir" + / "apex_vasp_stage_status.json" + ) + self.assertFalse(status["stages"][0]["footer_complete"]) + self.assertTrue( + status["stages"][0]["failure_reasons"][0].startswith( + "missing_footer_markers:" + ) + ) + + def test_single_stage_rejects_footer_outside_tail_region(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "fake_stale_footer.py").write_text( + "from pathlib import Path\n" + "footer = (\n" + " 'General timing and accounting informations\\n'\n" + " 'Total CPU time used (sec): 1\\n'\n" + " 'Elapsed time (sec): 1\\n'\n" + ")\n" + "Path('OUTCAR').write_text(\n" + " footer + ''.join(f'incomplete tail {i}\\n' " + "for i in range(300))\n" + ")\n" + "Path('OSZICAR').write_text('1 T= 300 E= 0\\n')\n" + "Path('CONTCAR').write_text('partial\\n')\n" + "Path('XDATCAR').write_text('partial\\n')\n" + ) + cwd = os.getcwd() + try: + os.chdir(root) + with self.assertRaises(TransientError): + RunVASP().execute(self._op_input( + task_dir, + "stale-footer", + "python fake_stale_footer.py", + )) + finally: + os.chdir(cwd) + + status = loadfn( + root / "stale-footer" / "backward_dir" + / "apex_vasp_stage_status.json" + ) + self.assertFalse(status["stages"][0]["footer_complete"]) + + def test_failure_destination_discovers_real_pythonop_tmp_ancestor(self): + with tempfile.TemporaryDirectory() as tmpdir: + dflow_tmp = Path(tmpdir) / "tmp" + (dflow_tmp / "inputs" / "artifacts").mkdir(parents=True) + (dflow_tmp / "outputs" / "artifacts").mkdir(parents=True) + work_dir = ( + dflow_tmp + / "confs" + / "hcp_Ti_36" + / "finite_t_latt_00" + / "task.000000" + ) + work_dir.mkdir(parents=True) + cwd = os.getcwd() + try: + os.chdir(work_dir) + destinations = RunVASP()._failure_destinations( + "backward_dir" + ) + finally: + os.chdir(cwd) + + self.assertIn( + ( + dflow_tmp + / "outputs" + / "artifacts" + / "backward_dir" + ).resolve(), + [path.resolve() for path in destinations], + ) + + def test_single_stage_cannot_pass_finite_t_latt_validation(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "task.json").write_text( + '{"type": "finite_t_latt"}\n' + ) + cwd = os.getcwd() + try: + os.chdir(root) + with self.assertRaisesRegex( + TransientError, "equi" + ): + RunVASP().execute(self._op_input( + task_dir, "incomplete", "python fake_vasp.py" + )) + finally: + os.chdir(cwd) + + status = loadfn( + root / "incomplete" / "backward_dir" + / "apex_vasp_stage_status.json" + ) + self.assertEqual(status["state"], "failed") + self.assertEqual( + status["missing_or_incomplete_stages"], + ["equi", "production"], + ) + + def test_annealing_validates_every_outcar_stage(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "input" + task_dir.mkdir() + self._write_common_inputs(task_dir) + (task_dir / "task.json").write_text('{"type": "annealing"}\n') + (task_dir / "INCAR.eq").write_text("NSW = 10\n") + (task_dir / "INCAR.production").write_text("NSW = 20\n") + (task_dir / "run_command").write_text( + "set -e\n" + "rm -f OUTCAR.apex XDATCAR.apex\n" + "cp INCAR.eq INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "printf '\\nAPEX_STAGE eq\\n' >> OUTCAR.apex\n" + "cat OUTCAR >> OUTCAR.apex\n" + "cp CONTCAR POSCAR\n" + "cp INCAR.production INCAR\n" + 'eval "$APEX_RUN_COMMAND"\n' + "printf '\\nAPEX_STAGE production\\n' >> OUTCAR.apex\n" + "cat OUTCAR >> OUTCAR.apex\n" + "mv OUTCAR.apex OUTCAR\n" + ) + cwd = os.getcwd() + try: + os.chdir(root) + result = RunVASP().execute(self._op_input( + task_dir, + "annealing", + "APEX_RUN_COMMAND='python fake_vasp.py' bash run_command", + )) + finally: + os.chdir(cwd) + + backward = root / result["backward_dir"] + status = loadfn(backward / "apex_vasp_stage_status.json") + self.assertEqual(status["state"], "succeeded") + self.assertEqual( + [stage["name"] for stage in status["stages"]], + ["eq", "production"], + ) + self.assertEqual( + [stage["observed_ionic_steps"] for stage in status["stages"]], + [10, 20], + ) + self.assertEqual( + (root / "annealing" / "calls.txt").read_text().splitlines(), + ["10", "20"], + ) + class TestRunLAMMPSDebug(unittest.TestCase): + def test_cleanup_preserves_image_resident_model_path(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + task_dir = root / "task" + task_dir.mkdir() + image_model = root / "image-model.pt2" + image_model.symlink_to(root / "missing-image-target.pt2") + dumpfn( + { + "type": "deepmd", + "model": str(image_model), + "model_in_image": True, + }, + task_dir / "inter.json", + ) + + RunLAMMPS._cleanup_model_links(task_dir) + + self.assertTrue(image_model.is_symlink()) + def test_run_lammps_writes_debug_log_on_success(self): with tempfile.TemporaryDirectory() as tmpdir: task_dir = Path(tmpdir) @@ -418,6 +1284,40 @@ def test_run_lammps_classifies_persistent_header_only_failure(self): self.assertEqual(status["reason"], REMOTE_LAMMPS_STARTUP_FAILURE) self.assertEqual(status["attempts"], 2) + def test_run_lammps_retries_transient_sigkill_failure(self): + with tempfile.TemporaryDirectory() as tmpdir: + task_dir = Path(tmpdir) + script = task_dir / "sigkill_once.py" + script.write_text( + "from pathlib import Path\n" + "count_file = Path('count.txt')\n" + "count = int(count_file.read_text()) if count_file.exists() else 0\n" + "count_file.write_text(str(count + 1))\n" + "Path('log.lammps').write_text(f'attempt {count + 1}\\n')\n" + "Path('dump.melting').write_text(f'attempt {count + 1}\\n')\n" + "raise SystemExit(137 if count == 0 else 0)\n" + ) + op = RunLAMMPS() + with patch.dict(os.environ, {"APEX_LAMMPS_HEADER_RETRY_DELAY": "0"}): + op.execute(OPIO({ + "input_lammps": task_dir, + "run_command": ( + "APEX_LAMMPS_TRANSIENT_RETRY=1 " + f"{sys.executable} {script.name}" + ), + })) + + self.assertEqual((task_dir / "count.txt").read_text(), "2") + self.assertTrue((task_dir / "log.lammps.attempt1").is_file()) + self.assertTrue((task_dir / "dump.melting.attempt1").is_file()) + status = loadfn(task_dir / "apex_task_status.json") + self.assertEqual(status["state"], "succeeded") + self.assertEqual(status["attempts"], 2) + self.assertEqual( + status["retry_reason"], + TRANSIENT_LAMMPS_RETRY_REASON, + ) + class TestMakeRelaxOPs(unittest.TestCase): def setUp(self) -> None: diff --git a/tests/test_parent_lattice_mapping.py b/tests/test_parent_lattice_mapping.py new file mode 100644 index 00000000..a691cbd5 --- /dev/null +++ b/tests/test_parent_lattice_mapping.py @@ -0,0 +1,366 @@ +import numpy as np +import pytest +import tempfile +import unittest +from pathlib import Path +from monty.serialization import dumpfn, loadfn +from pymatgen.core import Structure + +from apex.core.lib.parent_lattice_mapping import ( + resolve_parent_slip_geometry, + resolve_parent_supercell, +) +from apex.core.calculator.lib.lammps_utils import cvt_lammps_conf +from apex.core.property.gamma_geometry import build_parent_gamma_slab +from apex.core.property.gamma_slab import minimum_pair_distance +from apex.core.property.Gamma import Gamma +from apex.core.property.GammaSurface import GammaSurface + + +def _make_sheared_bcc_5x5x2(): + parent_lattice = np.array( + [ + [3.02, 0.00, 0.00], + [-0.25, 3.01, 0.00], + [-0.26, -0.28, 2.99], + ] + ) + structure = Structure( + parent_lattice, + ["V", "V"], + [[0.0, 0.0, 0.0], [0.5, 0.5, 0.5]], + ) + structure.make_supercell(np.diag([5, 5, 2])) + for index in range(5): + structure.replace(index, "Ti") + # Deterministic local alloy-scale offsets exercise tolerant anonymous site + # matching without changing the parent-supercell topology. + for index in range(len(structure)): + offset = 0.01 * np.array( + [ + np.sin(index), + np.cos(2 * index), + np.sin(3 * index), + ] + ) + structure.translate_sites( + [index], offset, frac_coords=False, to_unit_cell=True + ) + return structure + + +def test_resolve_bcc_parent_mapping_and_indices(): + structure = _make_sheared_bcc_5x5x2() + mapping = resolve_parent_supercell(structure, "bcc") + assert np.array_equal(mapping.supercell_matrix, np.diag([5, 5, 2])) + assert mapping.determinant == 50 + assert mapping.site_max < 0.1 + + geometry = resolve_parent_slip_geometry( + structure, + mapping, + plane_miller=[1, 1, 0], + slip_direction=[-1, 1, 1], + ) + assert np.allclose(geometry.mapped_plane_full, [5, 5, 0]) + assert np.array_equal(geometry.slab_miller, [1, 1, 0]) + assert np.allclose(geometry.mapped_direction, [-0.2, 0.2, 0.5]) + assert geometry.burgers_fraction == 0.5 + assert abs(np.dot(geometry.plane_normal_cart, geometry.direction_cart)) < 1e-8 + + +def test_resolve_non_diagonal_hnf_supercell(): + parent = Structure( + [[3.0, 0.0, 0.0], [0.05, 3.02, 0.0], [0.02, -0.03, 2.98]], + ["V", "V"], + [[0.0, 0.0, 0.0], [0.5, 0.5, 0.5]], + ) + expected = np.array([[2, 1, 0], [0, 5, 1], [0, 0, 5]]) + parent.make_supercell(expected) + mapping = resolve_parent_supercell(parent, "bcc") + assert np.array_equal(mapping.supercell_matrix, expected) + assert mapping.source == "automatic_hnf" + assert mapping.site_max < 1e-8 + + +def test_parent_gamma_slab_is_200_atoms_and_layer_gap_split(): + structure = _make_sheared_bcc_5x5x2() + mapping = resolve_parent_supercell(structure, "bcc") + geometry = resolve_parent_slip_geometry( + structure, mapping, [1, 1, 0], [-1, 1, 1] + ) + built = build_parent_gamma_slab( + structure, + geometry, + supercell_size=[2, 1, 1], + plane_target=1, + min_slab_height=10.0, + vacuum_size=20.0, + ) + assert len(built.slab) == 200 + assert len(built.lower_indices) + len(built.upper_indices) == 200 + assert len(built.lower_indices) > 0 + assert len(built.upper_indices) > 0 + assert built.metadata["moved_count"] == len(built.upper_indices) + assert 9.0 < built.metadata["material_height_angstrom"] < 12.0 + assert built.metadata["added_vacuum_angstrom"] == 20.0 + assert built.metadata["interface_count"] == 1 + assert built.metadata["parent_translation_topology_closed"] is True + assert np.allclose(built.slab.lattice.matrix[2, :2], 0.0, atol=1e-10) + assert np.allclose(built.slab.lattice.matrix[:2, 2], 0.0, atol=1e-10) + # No atomic layer may be split between the two vacuum-facing boundaries. + z = np.sort(built.slab.cart_coords[:, 2]) + layer_cuts = np.where(np.diff(z) > 0.5)[0] + 1 + layer_sizes = [len(group) for group in np.split(z, layer_cuts)] + assert layer_sizes == [40, 40, 40, 40, 40] + + midpoint = built.slab.copy() + local_burgers = geometry.local_frame @ geometry.burgers_vector_cart + midpoint.translate_sites( + built.upper_indices, + 0.5 * local_burgers, + frac_coords=False, + to_unit_cell=True, + ) + assert minimum_pair_distance(built.slab) > 2.0 + assert minimum_pair_distance(midpoint) > 2.0 + + +def test_gamma_make_confs_uses_parent_indices_without_extra_parameters(tmp_path): + equi = tmp_path / "relaxation" / "relax_task" + work = tmp_path / "gamma_00" + equi.mkdir(parents=True) + structure = _make_sheared_bcc_5x5x2() + structure.to(equi / "CONTCAR", "POSCAR") + dumpfn( + { + "energies": [-100.0], + "atom_numbs": [5, 95], + "cells": [structure.lattice.matrix.tolist()], + }, + equi / "result.json", + ) + prop = Gamma( + { + "type": "gamma", + "parent_lattice": "bcc", + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1], + "supercell_size": [2, 1, 1], + "min_slab_height": 10.0, + "max_atoms": 220, + "min_distance": 1.7, + "vacuum_size": 20.0, + "displacement_points": [0.0, 0.5], + }, + {"type": "deepmd"}, + ) + tasks = prop.make_confs(work, equi) + assert len(tasks) == 2 + manifest = loadfn(work / "gamma_geometry.json") + assert manifest["parent_mapping"]["supercell_matrix"] == [ + [5, 0, 0], + [0, 5, 0], + [0, 0, 2], + ] + assert manifest["slip_geometry"]["burgers_fraction"] == 0.5 + assert manifest["slab_geometry"]["upper_count"] > 0 + assert manifest["slab_geometry"]["lower_count"] > 0 + assert ( + manifest["slab_geometry"]["upper_count"] + + manifest["slab_geometry"]["lower_count"] + == 200 + ) + assert Structure.from_file(work / "task.000000" / "POSCAR").num_sites == 200 + assert Structure.from_file(work / "task.000001" / "POSCAR").num_sites == 200 + + +def test_lammps_conversion_preserves_local_surface_normal(tmp_path): + structure = _make_sheared_bcc_5x5x2() + mapping = resolve_parent_supercell(structure, "bcc") + geometry = resolve_parent_slip_geometry( + structure, mapping, [1, 1, 0], [-1, 1, 1] + ) + built = build_parent_gamma_slab( + structure, + geometry, + supercell_size=[2, 1, 1], + plane_target=1, + min_slab_height=10.0, + vacuum_size=20.0, + ) + poscar = tmp_path / "POSCAR" + lammps_data = tmp_path / "conf.lmp" + built.slab.to(poscar, "POSCAR") + cvt_lammps_conf(str(poscar), str(lammps_data), ["Ti", "V"]) + + # dpdata converts the already surface-aligned POSCAR to LAMMPS's restricted + # triclinic convention. It may rotate within the surface plane, but both + # in-plane lattice vectors must remain perpendicular to local z. Thus the + # existing setforce 0 0 NULL constraint still means normal-only relaxation. + import dpdata + + converted = dpdata.System( + str(lammps_data), fmt="lammps/lmp", type_map=["Ti", "V"] + ) + cell = np.asarray(converted.data["cells"][0], dtype=float) + assert np.allclose(cell[:2, 2], 0.0, atol=1e-10) + assert cell[2, 2] > 0.0 + + +def test_parent_gamma_surface_matches_gamma_line_x_section(tmp_path): + equi = tmp_path / "relaxation" / "relax_task" + equi.mkdir(parents=True) + structure = _make_sheared_bcc_5x5x2() + structure.to(equi / "CONTCAR", "POSCAR") + dumpfn( + { + "energies": [-100.0], + "atom_numbs": [5, 95], + "cells": [structure.lattice.matrix.tolist()], + }, + equi / "result.json", + ) + common = { + "parent_lattice": "bcc", + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1], + "supercell_size": [2, 1, 1], + "min_slab_height": 10.0, + "max_atoms": 220, + "min_distance": 1.7, + "vacuum_size": 20.0, + } + line = Gamma( + { + "type": "gamma", + **common, + "displacement_points": [0.0, 0.5], + }, + {"type": "deepmd"}, + ) + surface = GammaSurface( + { + "type": "gamma_surface", + **common, + "closed_loop": False, + "n_steps_x": 2, + "n_steps_y": 1, + }, + {"type": "deepmd"}, + ) + line_tasks = line.make_confs(tmp_path / "gamma_00", equi) + surface_tasks = surface.make_confs(tmp_path / "gamma_surface_00", equi) + assert len(line_tasks) == 2 + assert len(surface_tasks) == 6 + + manifest = loadfn(tmp_path / "gamma_surface_00" / "gamma_geometry.json") + assert manifest["parent_mapping"]["supercell_matrix"] == [ + [5, 0, 0], + [0, 5, 0], + [0, 0, 2], + ] + assert manifest["slab_geometry"]["interface_count"] == 1 + assert manifest["slab_geometry"]["upper_count"] > 0 + assert manifest["slab_geometry"]["lower_count"] > 0 + + for line_task, surface_task in zip( + line_tasks, + [surface_tasks[0], surface_tasks[2]], + ): + line_structure = Structure.from_file(line_task + "/POSCAR") + surface_structure = Structure.from_file(surface_task + "/POSCAR") + np.testing.assert_allclose( + line_structure.lattice.matrix, + surface_structure.lattice.matrix, + atol=1.0e-10, + ) + assert [site.specie.symbol for site in line_structure] == [ + site.specie.symbol for site in surface_structure + ] + np.testing.assert_allclose( + line_structure.frac_coords, + surface_structure.frac_coords, + atol=1.0e-10, + ) + + +def test_parent_gamma_strict_orthogonal_gate_is_fail_closed(tmp_path): + equi = tmp_path / "relaxation" / "relax_task" + equi.mkdir(parents=True) + structure = _make_sheared_bcc_5x5x2() + structure.to(equi / "CONTCAR", "POSCAR") + dumpfn( + {"energies": [-100.0], "atom_numbs": [5, 95]}, + equi / "result.json", + ) + prop = GammaSurface( + { + "type": "gamma_surface", + "parent_lattice": "bcc", + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1], + "supercell_size": [2, 1, 1], + "min_slab_height": 10.0, + "max_atoms": 220, + "min_distance": 1.7, + "vacuum_size": 20.0, + "require_orthogonal_cell": True, + "n_steps_x": 1, + "n_steps_y": 1, + }, + {"type": "deepmd"}, + ) + with pytest.raises(RuntimeError, match="will not Gram-Schmidt"): + prop.make_confs(tmp_path / "gamma_surface_00", equi) + + +def test_parent_gamma_strict_orthogonal_gate_accepts_pure_bcc(): + structure = Structure( + np.eye(3) * 3.0945974428563945, + ["V", "V"], + [[0.0, 0.0, 0.0], [0.5, 0.5, 0.5]], + ) + structure.make_supercell(np.diag([5, 5, 2])) + mapping = resolve_parent_supercell(structure, "bcc") + geometry = resolve_parent_slip_geometry( + structure, mapping, [1, 1, 0], [-1, 1, 1] + ) + built = build_parent_gamma_slab( + structure, + geometry, + supercell_size=[2, 1, 1], + plane_target=1, + min_slab_height=10.0, + vacuum_size=20.0, + require_orthogonal_cell=True, + ) + assert built.metadata["cell_geometry"]["zero_tilt"] is True + assert built.metadata["cell_geometry"]["orthogonalization_applied"] is False + + +def load_tests(loader, tests, pattern): + """Expose the pytest-style regression functions to unittest CI.""" + del loader, tests, pattern + suite = unittest.TestSuite() + functions = ( + test_resolve_bcc_parent_mapping_and_indices, + test_resolve_non_diagonal_hnf_supercell, + test_parent_gamma_slab_is_200_atoms_and_layer_gap_split, + test_gamma_make_confs_uses_parent_indices_without_extra_parameters, + test_lammps_conversion_preserves_local_surface_normal, + test_parent_gamma_surface_matches_gamma_line_x_section, + test_parent_gamma_strict_orthogonal_gate_is_fail_closed, + test_parent_gamma_strict_orthogonal_gate_accepts_pure_bcc, + ) + for function in functions: + if "tmp_path" in function.__code__.co_varnames: + def run_with_tmp_path(function=function): + with tempfile.TemporaryDirectory() as directory: + function(Path(directory)) + + test = unittest.FunctionTestCase(run_with_tmp_path) + else: + test = unittest.FunctionTestCase(function) + suite.addTest(test) + return suite diff --git a/tests/test_phonon.py b/tests/test_phonon.py index 04fa439d..23015452 100644 --- a/tests/test_phonon.py +++ b/tests/test_phonon.py @@ -286,7 +286,7 @@ def test_phonolammps_command_includes_matching_primitive_axes(self): phonon = Phonon({"type": "phonon", "supercell_size": [2, 2, 2]}) self.assertEqual( phonon._build_phonolammps_run_command(), - "phonolammps in.lammps -c POSCAR --dim 2 2 2 " + "phonolammps in.lammps -c POSCAR --dim 2 2 2 --logshow " "-pa 1 0 0 0 1 0 0 0 1", ) @@ -642,7 +642,7 @@ def test_make_phonon_conf(self): with open(os.path.join(self.target_path, "task.000000/band.conf")) as fp: self.assertIn("PRIMITIVE_AXES = P", fp.read()) - def test_post_process_injects_deepmd_plugin_for_phonon(self): + def test_post_process_uses_integrated_deepmd_for_phonon(self): deepmd_phonon = Phonon( {"type": "phonon", "supercell_size": [2, 2, 2]}, inter_param={"type": "deepmd"}, @@ -651,14 +651,43 @@ def test_post_process_injects_deepmd_plugin_for_phonon(self): shutil.rmtree(task_dir.parent, ignore_errors=True) task_dir.mkdir(parents=True, exist_ok=True) (task_dir / "in.lammps").write_text( - "clear\npair_style deepmd frozen_model.pth\npair_coeff * * Cu O\nrun 0\n" + "clear\nplugin load libdeepmd_lmp.so\n" + "pair_style deepmd frozen_model.pth\npair_coeff * * Cu O\nrun 0\n" ) try: deepmd_phonon.post_process([str(task_dir)]) rewritten = (task_dir / "in.lammps").read_text() - self.assertIn("plugin load libdeepmd_lmp.so", rewritten) + self.assertNotIn("plugin load libdeepmd_lmp.so", rewritten) self.assertIn("pair_style deepmd frozen_model.pth", rewritten) self.assertNotIn("run 0", rewritten) + self.assertIn("--logshow", (task_dir / "run_command").read_text()) + finally: + shutil.rmtree(task_dir.parent, ignore_errors=True) + + def test_post_process_uses_runtime_autoload_for_dpa4_pt2(self): + dpa4_phonon = Phonon( + {"type": "phonon", "supercell_size": [2, 2, 2]}, + inter_param={ + "type": "deepmd", + "deepmd_runtime": "dpa4_pt2", + }, + ) + task_dir = Path("output/phonon_dpa4_pt2_post/task.000000") + shutil.rmtree(task_dir.parent, ignore_errors=True) + task_dir.mkdir(parents=True, exist_ok=True) + (task_dir / "in.lammps").write_text( + "clear\npair_style deepmd /opt/dpa4-runtime/model.pt2\n" + "pair_coeff * * Cu O\nrun 0\n" + ) + + try: + dpa4_phonon.post_process([str(task_dir)]) + rewritten = (task_dir / "in.lammps").read_text() + self.assertNotIn("plugin load", rewritten) + self.assertIn( + "pair_style deepmd /opt/dpa4-runtime/model.pt2", rewritten + ) + self.assertNotIn("run 0", rewritten) finally: shutil.rmtree(task_dir.parent, ignore_errors=True) diff --git a/tests/test_preview.py b/tests/test_preview.py new file mode 100644 index 00000000..00e8ef37 --- /dev/null +++ b/tests/test_preview.py @@ -0,0 +1,409 @@ +import io +import os +import sys +import tempfile +import unittest +from contextlib import redirect_stderr +from pathlib import Path +from types import SimpleNamespace +from unittest import mock + +import numpy as np +from ase import Atoms +from monty.serialization import dumpfn +from pymatgen.core import Lattice, Structure + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) +__package__ = "tests" + +from apex import preview as preview_mod # noqa: E402 +from apex.preview import ( # noqa: E402 + _gamma_frames_for_view, + _min_pair_distance, + _prepare_equilibrium_dir, + _requested_gif_views, + _resolved_gamma_view_context, + _slip_plane_transform, + _structure_bounds, + _view_output_gif_path, + _warn_gamma_surface_overlaps, +) + + +class TestPreviewHelpers(unittest.TestCase): + def test_requested_gif_views(self): + self.assertEqual( + _requested_gif_views("both"), + ["slip-plane", "parent-bc"], + ) + self.assertEqual(_requested_gif_views("default"), ["default"]) + self.assertEqual( + _requested_gif_views("auto", "gamma"), + ["slip-plane", "parent-bc"], + ) + self.assertEqual( + _requested_gif_views("auto", "gamma_surface"), + ["slip-plane", "parent-bc"], + ) + self.assertEqual(_requested_gif_views("auto", "vacancy"), ["default"]) + with self.assertRaisesRegex(ValueError, "Unknown GIF view"): + _requested_gif_views("invalid") + + def test_preview_parser_defaults_gamma_views_to_auto(self): + args = preview_mod.build_parser().parse_args(["param.json"]) + self.assertEqual(args.gif_view, "auto") + + from apex.main import parse_args + + with mock.patch.object( + sys, + "argv", + ["apex", "preview", "param.json"], + ): + _, main_args = parse_args() + self.assertEqual(main_args.gif_view, "auto") + + def test_main_preview_prints_generated_paths(self): + import apex.main as main_mod + + with mock.patch.object( + sys, + "argv", + ["apex", "preview", "param.json"], + ), mock.patch.object( + preview_mod, + "preview_from_args", + return_value=["first.gif", "second.gif"], + ), mock.patch("builtins.print") as print_mock: + main_mod.main() + + print_mock.assert_has_calls( + [mock.call("first.gif"), mock.call("second.gif")] + ) + + def test_structure_bounds_include_projected_unit_cell(self): + atoms = Atoms( + "H", + positions=[[2.0, 12.0, 0.0]], + cell=[[4.0, 0.0, 0.0], [0.0, 0.0, 4.0], [0.0, 24.0, 0.0]], + pbc=True, + ) + + x_min, x_max, y_min, y_max = _structure_bounds(atoms, 0.35) + + self.assertLessEqual(x_min, 0.0) + self.assertGreaterEqual(x_max, 4.0) + self.assertLessEqual(y_min, 0.0) + self.assertGreaterEqual(y_max, 24.0) + + def test_view_output_paths_preserve_legacy_default(self): + base = Path("/tmp/example.gif") + self.assertEqual(_view_output_gif_path(base, "default"), base) + self.assertEqual( + _view_output_gif_path(base, "slip-plane"), + Path("/tmp/example_slip_plane.gif"), + ) + self.assertEqual( + _view_output_gif_path(base, "parent-bc"), + Path("/tmp/example_parent_bc.gif"), + ) + + def test_prepare_equilibrium_dir_adds_preview_result(self): + with tempfile.TemporaryDirectory() as tmp: + source = Path(tmp) / "source" + source.mkdir() + Structure( + Lattice.cubic(4.0), + ["H"], + [[0.0, 0.0, 0.0]], + ).to(filename=source / "POSCAR", fmt="poscar") + prepared = Path( + _prepare_equilibrium_dir( + str(source), + Path(tmp) / "preview", + "test", + ) + ) + self.assertTrue((prepared / "CONTCAR").is_file()) + self.assertEqual( + (prepared / "result.json").read_text(), + '{"preview_only": true}\n', + ) + + def test_slip_plane_view_looks_along_generated_normal(self): + cell = np.diag([10.0, 10.0, 10.0]) + first = Atoms( + "HH", + positions=[[1.0, 1.0, 1.0], [5.0, 5.0, 5.0]], + cell=cell, + pbc=True, + ) + second = first.copy() + second.positions[1, 0] += 0.5 + transform = _slip_plane_transform([first, second]) + np.testing.assert_allclose(transform[0], [1.0, 0.0, 0.0]) + np.testing.assert_allclose(transform[2], [0.0, 0.0, 1.0]) + + def test_slip_plane_view_handles_full_period_endpoint(self): + first = Atoms( + "HH", + scaled_positions=[[0.1, 0.1, 0.1], [0.5, 0.5, 0.5]], + cell=np.diag([10.0, 10.0, 10.0]), + pbc=True, + ) + second = first.copy() + second.set_scaled_positions( + [[0.1, 0.1, 0.1], [1.5, 0.5, 0.5]] + ) + + transform = _slip_plane_transform([first, second]) + + np.testing.assert_allclose(transform[0], [1.0, 0.0, 0.0]) + + def test_resolved_gamma_view_context_uses_crystal_override(self): + parent = Structure( + Lattice.cubic(4.0), + ["Mo", "Mo"], + [[0.0, 0.0, 0.0], [0.5, 0.5, 0.5]], + ) + prop_obj = SimpleNamespace( + conv_std_structure=parent, + structure_type="bcc", + plane_miller=[1, 1, 0], + slip_direction=[-1, 1, 1], + ) + + resolved_parent, plane, direction = _resolved_gamma_view_context( + prop_obj + ) + + self.assertIs(resolved_parent, parent) + np.testing.assert_allclose(plane, [1.0, 1.0, 0.0]) + np.testing.assert_allclose(direction, [-1.0, 1.0, 1.0]) + + def test_resolved_gamma_view_context_converts_hcp_indices(self): + parent = Structure( + Lattice.hexagonal(3.0, 4.8), + ["Ti", "Ti"], + [[0.0, 0.0, 0.0], [1 / 3, 2 / 3, 0.5]], + ) + prop_obj = SimpleNamespace( + conv_std_structure=parent, + structure_type="hcp", + plane_miller=[0, 0, 0, 1], + slip_direction=[2, -1, -1, 0], + ) + + _, plane, direction = _resolved_gamma_view_context(prop_obj) + + np.testing.assert_allclose(plane, [0.0, 0.0, 1.0]) + np.testing.assert_allclose(direction, [3.0, 0.0, 0.0]) + + def test_parent_bc_view_looks_along_parent_bc_normal(self): + with tempfile.TemporaryDirectory() as tmp: + parent_path = Path(tmp) / "POSCAR" + Structure( + Lattice.cubic(10.0), + ["H", "H"], + [[0.1, 0.1, 0.1], [0.5, 0.5, 0.5]], + ).to(filename=parent_path, fmt="poscar") + first = Atoms( + "HH", + positions=[[1.0, 1.0, 1.0], [5.0, 5.0, 5.0]], + cell=np.diag([10.0, 10.0, 10.0]), + pbc=True, + ) + second = first.copy() + second.positions[1, 0] += 0.5 + transformed = _gamma_frames_for_view( + [first, second], + "parent-bc", + parent_structure=Structure.from_file(parent_path), + plane_miller=[0, 1, 0], + slip_direction=[1, 0, 0], + ) + original_a = np.array([10.0, 0.0, 0.0]) + transformed_a = transformed[0].cell.array[0] + self.assertAlmostEqual(transformed_a[0], 0.0) + self.assertAlmostEqual(transformed_a[1], 0.0) + self.assertAlmostEqual(abs(transformed_a[2]), np.linalg.norm(original_a)) + + def test_gamma_surface_view_uses_primary_cell_axis(self): + first = Atoms( + "HH", + positions=[[1.0, 1.0, 1.0], [5.0, 5.0, 5.0]], + cell=np.diag([10.0, 10.0, 30.0]), + pbc=True, + ) + second = first.copy() + # A two-dimensional surface traversal can move along the secondary + # direction first; the primary view must remain tied to slab a. + second.positions[1, 1] += 0.5 + + transformed = _gamma_frames_for_view( + [first, second], + "slip-plane", + parent_structure=None, + plane_miller=None, + slip_direction=None, + use_cell_axis=True, + ) + + np.testing.assert_allclose(transformed[0].cell.array[0], [10.0, 0.0, 0.0]) + + def test_gamma_family_preview_defaults_to_two_views_with_20a_vacuum(self): + def normal_height(atoms): + cell = np.asarray(atoms.cell.array, dtype=float) + normal = np.cross(cell[0], cell[1]) + normal /= np.linalg.norm(normal) + return abs(float(np.dot(cell[2], normal))) + + for property_type in ("gamma", "gamma_surface"): + with self.subTest( + property_type=property_type + ), tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + structure_dir = root / "bcc" + structure_dir.mkdir() + Structure( + Lattice.cubic(3.2), + ["Mo", "Mo"], + [[0.0, 0.0, 0.0], [0.5, 0.5, 0.5]], + ).to(filename=structure_dir / "POSCAR", fmt="poscar") + + prop = { + "type": property_type, + "req_calc": True, + "plane_miller": [1, 1, 0], + "slip_direction": [1, -1, -1], + "supercell_size": [1, 1, 2], + } + if property_type == "gamma": + prop["n_steps"] = 1 + else: + prop.update( + { + "closed_loop": False, + "n_steps_x": 1, + "n_steps_y": 1, + } + ) + payload = { + "structures": ["bcc"], + "interaction": {"type": "vasp"}, + "properties": [prop], + } + parameter_path = root / "param.json" + dumpfn(payload, parameter_path) + + rendered = [] + + def capture(frames, output_gif, **_kwargs): + rendered.append((Path(output_gif), frames)) + + with mock.patch.object( + preview_mod, + "_write_gif", + side_effect=capture, + ): + outputs = preview_mod.preview_parameter_file( + str(parameter_path) + ) + + self.assertEqual( + [Path(path).name for path in outputs], + ["param_slip_plane.gif", "param_parent_bc.gif"], + ) + self.assertEqual(len(rendered), 2) + default_vacuum_height = normal_height(rendered[0][1][0]) + + prop["vacuum_size"] = 0 + dumpfn(payload, parameter_path) + zero_vacuum_rendered = [] + + def capture_zero(frames, output_gif, **_kwargs): + zero_vacuum_rendered.append((Path(output_gif), frames)) + + with mock.patch.object( + preview_mod, + "_write_gif", + side_effect=capture_zero, + ): + preview_mod.preview_parameter_file(str(parameter_path)) + + zero_vacuum_height = normal_height( + zero_vacuum_rendered[0][1][0] + ) + self.assertAlmostEqual( + default_vacuum_height - zero_vacuum_height, + 20.0, + places=6, + ) + + def test_min_pair_distance_detects_overlap(self): + overlapping = Structure( + Lattice.cubic(10.0), + ["H", "H"], + [[0.0, 0.0, 0.0], [0.01, 0.0, 0.0]], + coords_are_cartesian=True, + ) + separated = Structure( + Lattice.cubic(10.0), + ["H", "H"], + [[0.0, 0.0, 0.0], [2.0, 0.0, 0.0]], + coords_are_cartesian=True, + ) + self.assertLess(_min_pair_distance(overlapping), 0.2) + self.assertGreater(_min_pair_distance(separated), 0.2) + + def test_warn_gamma_surface_overlaps_prints_once(self): + with tempfile.TemporaryDirectory() as tmp: + task0 = os.path.join(tmp, "task.000000") + task1 = os.path.join(tmp, "task.000001") + os.makedirs(task0) + os.makedirs(task1) + Structure( + Lattice.cubic(10.0), + ["H", "H"], + [[0.0, 0.0, 0.0], [0.01, 0.0, 0.0]], + coords_are_cartesian=True, + ).to(filename=os.path.join(task0, "POSCAR"), fmt="poscar") + Structure( + Lattice.cubic(10.0), + ["H", "H"], + [[0.0, 0.0, 0.0], [2.0, 0.0, 0.0]], + coords_are_cartesian=True, + ).to(filename=os.path.join(task1, "POSCAR"), fmt="poscar") + + buf = io.StringIO() + with redirect_stderr(buf): + _warn_gamma_surface_overlaps([task0, task1]) + self.assertIn( + "Generated Gamma surface contains overlapping atoms.", + buf.getvalue(), + ) + + def test_warn_gamma_surface_overlaps_silent_when_ok(self): + with tempfile.TemporaryDirectory() as tmp: + task0 = os.path.join(tmp, "task.000000") + os.makedirs(task0) + Structure( + Lattice.cubic(10.0), + ["H", "H"], + [[0.0, 0.0, 0.0], [2.0, 0.0, 0.0]], + coords_are_cartesian=True, + ).to(filename=os.path.join(task0, "POSCAR"), fmt="poscar") + buf = io.StringIO() + with redirect_stderr(buf): + _warn_gamma_surface_overlaps([task0]) + self.assertEqual(buf.getvalue(), "") + + def test_preview_source_skips_post_process(self): + with open(preview_mod.__file__, encoding="utf-8") as fp: + source = fp.read() + self.assertNotIn("prop_obj.post_process(task_list)", source) + self.assertIn("_warn_gamma_surface_overlaps(task_list)", source) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_skill.py b/tests/test_skill.py index 24b53b3e..696bf3c0 100644 --- a/tests/test_skill.py +++ b/tests/test_skill.py @@ -1,4 +1,5 @@ import importlib.util +import hashlib import io import os import sys @@ -29,19 +30,51 @@ def test_bundled_skill_exists(self): self.assertIn(f"name: {SKILL_NAME}", text) self.assertIn("BOHRIUM_PROJECT_ID", text) self.assertIn("validate_apex_combo.py", text) - self.assertIn("models/DPA-3.2-5M", text) - self.assertIn("DPA-3.2-5M-OMat24.pth", text) + self.assertIn("models/DPA4-alloytongqi/model.pt", text) + self.assertIn("dpa4-phonolammps:0.0.2", text) + self.assertIn("Unknown model type: dpa4", text) + self.assertNotIn("DPA-3.2", text) + self.assertIn("15-property", text) + self.assertIn("displacement_points", text) + self.assertIn("restart.coexistence.start", text) + self.assertIn("finite_t_latt` never receives this file", text) + self.assertIn("--gif-view", text) root = get_skill_root() self.assertTrue((root / "models" / "README.md").is_file()) - self.assertTrue( - ( - root - / "models" - / "DPA-3.2-5M" - / "DPA-3.2-5M-OMat24.pth" - ).is_file() + profile = root / "data" / "dpa4_alloytongqi_t4_profile.json" + self.assertTrue(profile.is_file()) + profile_text = profile.read_text(encoding="utf-8") + self.assertIn("__DPA4_IMAGE_REF__", profile_text) + self.assertIn("pre_snapshot_only", profile_text) + self.assertIn("c4_m15_1 * NVIDIA T4", profile_text) + self.assertNotIn("4c-nano", profile_text) + benchmark = root / "benchmarks" / "dpa4-alloytongqi" + for name in ( + "README.md", + "generate_cases.py", + "build_manifest.py", + "verify_manifest.py", + "manifest.schema.json", + ): + self.assertTrue((benchmark / name).is_file()) + model = root / "models" / "DPA4-alloytongqi" / "model.pt" + self.assertTrue(model.is_file()) + self.assertEqual(model.stat().st_size, 30_403_297) + self.assertEqual( + hashlib.sha256(model.read_bytes()).hexdigest(), + "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad", ) - self.assertTrue((root / "scripts" / "fetch_models.py").is_file()) + self.assertFalse((root / "models" / "DPA-3.2-5M").exists()) + self.assertFalse((root / "scripts" / "fetch_models.py").exists()) + local = root / "variants" / "local" + self.assertTrue((local / "SKILL.md").is_file()) + self.assertTrue((local / "reference" / "submission.md").is_file()) + for profile in ( + "bohrium-direct.md", + "local-debug.md", + "local-cluster.md", + ): + self.assertTrue((local / "profiles" / profile).is_file()) def test_no_hardcoded_project_id_in_skill_docs(self): root = get_skill_root() @@ -67,7 +100,7 @@ def test_skill_documents_safe_submission_defaults(self): self.assertIn("never refresh it in `run.sh`", skill) self.assertIn("Do not read BOHRIUM_ACCESS_KEY", submission) self.assertNotIn("Recommended run.sh Ticket Refresh Template", submission) - self.assertIn("use the model bundled with this skill", skill) + self.assertIn("as source-checkpoint\n provenance", skill) self.assertIn('"type_map": "auto"', skill) self.assertIn("already a supercell", structure) self.assertIn("`supercell` / `supercell_size` to `[1,1,1]`", structure) @@ -91,10 +124,51 @@ def test_skill_zip_flag(self): self.assertTrue( any(n.startswith(f"{SKILL_NAME}/scripts/") for n in names) ) - self.assertTrue( - any(n.endswith("DPA-3.2-5M-OMat24.pth") for n in names) + self.assertEqual( + [n for n in names if n.endswith(".pt")], + [f"{SKILL_NAME}/models/DPA4-alloytongqi/model.pt"], + ) + self.assertFalse(any(n.endswith(".pth") for n in names)) + self.assertFalse(any("DPA-3.2" in n for n in names)) + self.assertFalse(any("/variants/" in n for n in names)) + self.assertFalse(any("global_local_" in n for n in names)) + self.assertFalse(any("global_bohrium_direct.json" in n for n in names)) + self.assertIn( + f"{SKILL_NAME}/data/dpa4_alloytongqi_t4_profile.json", names + ) + self.assertIn( + f"{SKILL_NAME}/scripts/dpa4_profile.py", names + ) + self.assertIn( + f"{SKILL_NAME}/benchmarks/dpa4-alloytongqi/" + "manifest.schema.json", + names, + ) + self.assertIn( + f"{SKILL_NAME}/benchmarks/dpa4-alloytongqi/" + "verify_manifest.py", + names, + ) + with zipfile.ZipFile(out) as zf: + cloud_skill = zf.read( + f"{SKILL_NAME}/SKILL.md" + ).decode("utf-8") + text_payload = b"\n".join( + zf.read(name) + for name in names + if not name.endswith(".pt") + ) + self.assertIn("outer Bohrium job", cloud_skill) + self.assertIn("bohrium_config.ticket", cloud_skill) + self.assertIn( + b"registry.dp.tech/dptech/dp/native/prod-397637/" + b"deepmd-kit-phonolammps:3.1.3", + text_payload, + ) + self.assertNotIn( + b"registry.dp.tech/dptech/dpa-calculator:dpa-mLip-452b0667", + text_payload, ) - self.assertFalse(any(n.endswith(".pt") for n in names)) def test_skill_agent_prompt(self): buf = io.StringIO() @@ -105,6 +179,15 @@ def test_skill_agent_prompt(self): self.assertIn(SKILL_NAME, out) self.assertIn("apex skill --zip", out) self.assertIn("MatMaster", out) + self.assertIn("Bohrium cloud", out) + self.assertIn("local", out) + self.assertIn("local cluster", out) + self.assertIn("bohrium-direct.md", out) + self.assertIn("local-debug.md", out) + self.assertIn("local-cluster.md", out) + self.assertIn("variants/local", out) + self.assertIn("get_skill_root", out) + self.assertNotIn(str(get_skill_root()), out) class TestValidateApexCombo(unittest.TestCase): @@ -160,8 +243,22 @@ def test_ok_combo(self): def test_recommend_lammps_gpu(self): rec = self.combo.recommend("lammps", "gpu") - self.assertIn("3.1.3", rec["image"]) - self.assertIn("T4", rec["scass_type"]) + self.assertIn("dpa4-phonolammps:0.0.2", rec["image"]) + self.assertIn("L20", rec["scass_type"]) + + def test_recommend_lammps_cpu_uses_apex_flow(self): + rec = self.combo.recommend("lammps", "cpu") + self.assertIn("apex-flow:1.3.0.post", rec["image"]) + self.assertIn("_cpu", rec["scass_type"]) + + def test_dpa4_image_is_blocked_on_cpu(self): + ok, errors = self.combo.check_combo( + "registry.dp.tech/dptech/dp/native/prod-16664/" + "dpa4-phonolammps:0.0.2", + "c8_m32_cpu", + ) + self.assertFalse(ok) + self.assertTrue(any("DPA4 image" in error for error in errors)) def test_cli_check_exit_codes(self): self.assertEqual( @@ -218,7 +315,7 @@ def test_lammps_interaction_defaults_to_auto_type_map(self): interaction = self.gen.build_interaction( backend="lammps", potential="deepmd", - model="DPA-3.2-5M-OMat24.pth", + model="model.pt", ) self.assertEqual(interaction["type_map"], "auto") @@ -228,11 +325,18 @@ def test_generate_config_has_no_type_map_cli_option(self): ).read_text(encoding="utf-8") self.assertNotIn('add_argument("--type-map"', source) - def test_lammps_default_uses_validated_phonolammps_image(self): + def test_lammps_gpu_default_uses_validated_phonolammps_image(self): + self.assertEqual( + self.gen.LAMMPS_GPU_IMAGE, + "registry.dp.tech/dptech/dp/native/prod-16664/" + "dpa4-phonolammps:0.0.2", + ) + + def test_lammps_cpu_default_uses_apex_flow_image(self): self.assertEqual( - self.gen.LAMMPS_IMAGE, + self.gen.LAMMPS_CPU_IMAGE, "registry.dp.tech/dptech/dp/native/prod-397637/" - "deepmd-kit-phonolammps:3.1.3", + "apex-flow:1.3.0.post", ) def test_combo_validator_has_no_property_cli(self): @@ -249,10 +353,16 @@ def setUpClass(cls): cls.gen = _load_script("generate_config.py") cls.validator = _load_script("validate_inputs.py") - def test_generated_gamma_surface_preserves_legacy_default(self): + def test_generated_gamma_surface_uses_periodic_recommended_default(self): self.assertIs( self.gen.PROPERTY_DEFAULTS["gamma_surface"]["closed_loop"], - False, + True, + ) + self.assertEqual( + self.gen.PROPERTY_DEFAULTS["gamma"]["vacuum_size"], 20 + ) + self.assertEqual( + self.gen.PROPERTY_DEFAULTS["gamma_surface"]["vacuum_size"], 20 ) def test_closed_loop_requires_boolean_and_no_custom_lengths(self): diff --git a/tests/test_skill_scripts.py b/tests/test_skill_scripts.py index d3d537a8..05c54265 100644 --- a/tests/test_skill_scripts.py +++ b/tests/test_skill_scripts.py @@ -1,4 +1,5 @@ import importlib.util +import hashlib import io import json import os @@ -6,6 +7,7 @@ import sys import tempfile import unittest +from argparse import Namespace from pathlib import Path from unittest.mock import patch from urllib.error import URLError @@ -38,6 +40,58 @@ def read(self): return json.dumps(self.payload).encode() +class TestFiniteTemperatureTemplates(unittest.TestCase): + def test_backend_truth_table_and_vasp_defaults(self): + skill_root = get_skill_root() + templates = json.loads( + (skill_root / "data" / "default_templates.json").read_text() + ) + properties = templates["properties"] + self.assertEqual( + {"lammps": True, "abacus": False, "vasp": True}, + properties["finite_t_latt"], + ) + self.assertEqual( + {"lammps": True, "abacus": False, "vasp": True}, + properties["annealing"], + ) + self.assertEqual( + {"lammps": True, "abacus": False, "vasp": False}, + properties["finite_t_elastic"], + ) + self.assertEqual( + {"lammps": True, "abacus": False, "vasp": False}, + properties["melting_point"], + ) + + vasp_props = json.loads( + ( + skill_root.parents[2] + / "apex" + / "default_config" + / "vasp" + / "param_props.json" + ).read_text() + )["properties"] + finite_latt = next( + prop for prop in vasp_props if prop["type"] == "finite_t_latt" + ) + self.assertEqual( + [300, 500, 700, 900, 1100, 1300, 1500], + finite_latt["cal_setting"]["temperature"], + ) + self.assertEqual(5000, finite_latt["cal_setting"]["equi_step"]) + self.assertEqual(10000, finite_latt["cal_setting"]["ave_step"]) + coexistence = next( + prop + for prop in vasp_props + if prop["type"] == "annealing" + and prop.get("protocol") == "coexistence" + ) + self.assertEqual(5000, coexistence["cal_setting"]["equi_step"]) + self.assertEqual(10000, coexistence["cal_setting"]["production_step"]) + + class TestGenerateConfigHelpers(unittest.TestCase): @classmethod def setUpClass(cls): @@ -53,6 +107,9 @@ def test_get_bohrium_ticket_success(self): self.assertEqual(self.gen.get_bohrium_ticket("secret"), ticket) request = mocked.call_args.args[0] self.assertIn("accessKey=secret", request.full_url) + self.assertIn( + f"expiration={self.gen.TICKET_EXPIRE_HOURS}", request.full_url + ) def test_get_bohrium_ticket_rejects_transport_and_api_errors(self): with patch.object(self.gen, "urlopen", side_effect=URLError("offline")): @@ -105,6 +162,15 @@ def test_build_global_json_selects_backend_resources(self): self.assertEqual(config["program_id"], 42) self.assertEqual(config["scass_type"], expected_scass) self.assertIn(command, config["lammps_run_command"]) + if backend == "lammps": + expected_image = ( + self.gen.LAMMPS_GPU_IMAGE + if potential in self.gen.GPU_POTENTIALS + else self.gen.LAMMPS_CPU_IMAGE + ) + self.assertEqual( + config["lammps_image_name"], expected_image + ) self.assertEqual(validate.call_count, len(cases)) def test_build_global_json_backend_fields_and_overrides(self): @@ -140,6 +206,69 @@ def test_build_global_json_backend_fields_and_overrides(self): "registry.example/private/vasp:licensed", ) + def test_build_global_json_sandbox_routes_lammps_images(self): + gpu = self.gen.build_global_json_sandbox( + "lammps", "deepmd", access_key="key", project_id=42 + ) + cpu = self.gen.build_global_json_sandbox( + "lammps", "eam_alloy", access_key="key", project_id=42 + ) + + self.assertEqual(gpu["lammps_image_name"], self.gen.LAMMPS_GPU_IMAGE) + self.assertEqual(gpu["image_address"], self.gen.LAMMPS_GPU_IMAGE) + self.assertEqual( + gpu["machine_type"], + self.gen.SANDBOX_MACHINE_TYPES["lammps_gpu"], + ) + self.assertEqual(cpu["lammps_image_name"], self.gen.LAMMPS_CPU_IMAGE) + self.assertEqual(cpu["image_address"], self.gen.LAMMPS_CPU_IMAGE) + self.assertEqual( + cpu["machine_type"], + self.gen.SANDBOX_MACHINE_TYPES["lammps_cpu"], + ) + + def test_explicit_lammps_image_supports_cpu_deepmd(self): + image = ( + "registry.dp.tech/dptech/dp/native/prod-397637/" + "deepmd-kit-phonolammps:3.1.3" + ) + config = self.gen.build_global_json_sandbox( + "lammps", + "deepmd", + access_key="key", + project_id=42, + machine_type="c8_m32_cpu", + lammps_image=image, + ) + self.assertEqual(config["lammps_image_name"], image) + self.assertEqual(config["image_address"], image) + + def test_load_confirmed_property_configs_filters_in_requested_order(self): + payload = { + "properties": [ + {"type": "eos", "vol_step": 0.02}, + {"type": "phonon", "supercell_size": [3, 3, 3]}, + ] + } + with tempfile.TemporaryDirectory() as tmpdir: + path = Path(tmpdir) / "confirmed.json" + path.write_text(json.dumps(payload), encoding="utf-8") + configs = self.gen.load_confirmed_property_configs( + str(path), ["phonon", "eos"] + ) + self.assertEqual([item["type"] for item in configs], ["phonon", "eos"]) + self.assertEqual(configs[0]["supercell_size"], [3, 3, 3]) + + def test_build_param_json_uses_confirmed_property_configs_exactly(self): + confirmed = [{"type": "eos", "vol_step": 0.02}] + param = self.gen.build_param_json( + "confs/input", + {"type": "deepmd", "model": "model.pth"}, + ["eos"], + property_configs=confirmed, + ) + self.assertEqual(param["properties"], confirmed) + def test_build_global_json_requires_access_key_and_propagates_combo_error(self): with patch.dict(os.environ, {}, clear=True): with self.assertRaisesRegex(RuntimeError, "BOHRIUM_ACCESS_KEY"): @@ -157,6 +286,93 @@ def test_build_global_json_requires_access_key_and_propagates_combo_error(self): "lammps", "deepmd", access_key="key", project_id=1 ) + def test_unpublished_dpa4_profile_fails_closed_before_ticket(self): + with patch.object(self.gen, "get_bohrium_ticket") as ticket: + with self.assertRaisesRegex(RuntimeError, "not published"): + self.gen.build_global_json( + "lammps", + "deepmd", + access_key="key", + project_id=42, + runtime_profile=self.gen.DPA4_RUNTIME_PROFILE, + ) + ticket.assert_not_called() + + def test_published_dpa4_profile_builds_exact_contract(self): + profile = json.loads( + ( + get_skill_root() + / "data" + / "dpa4_alloytongqi_t4_profile.json" + ).read_text(encoding="utf-8") + ) + profile["qualification_status"] = "post_snapshot_passed" + profile["image"] = { + "ref": "registry.example/dpa4:qualified", + "digest": "sha256:" + "a" * 64, + } + with patch.object( + self.gen, "get_bohrium_ticket", return_value="t" * 36 + ), patch.object(self.gen, "_validate_image_scass") as validate: + global_config = self.gen.build_global_json( + "lammps", + "deepmd", + access_key="key", + project_id=42, + runtime_profile=profile, + ) + self.assertEqual( + global_config["lammps_image_name"], + "registry.example/dpa4:qualified@sha256:" + "a" * 64, + ) + self.assertEqual( + global_config["scass_type"], "c4_m15_1 * NVIDIA T4" + ) + self.assertEqual( + global_config["lammps_run_command"], + "/usr/local/bin/dpa4-lmp -in in.lammps", + ) + self.assertEqual( + global_config["phonolammps_run_command"], + "/usr/local/bin/dpa4-phonolammps {input_file} " + "-c {poscar} --dim {dim} {primitive_axes}", + ) + validate.assert_called_once_with( + global_config["lammps_image_name"], + "c4_m15_1 * NVIDIA T4", + runtime_profile=self.gen.DPA4_RUNTIME_PROFILE, + ) + + interaction = self.gen.build_interaction( + "lammps", "deepmd", runtime_profile=profile + ) + self.assertEqual(interaction, { + "type": "deepmd", + "deepmd_runtime": "dpa4_pt2", + "model_in_image": True, + "model": ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/" + "alloytongqi.t4-sm75.pt2" + ), + "runtime_model_sha256": ( + "2614db9463f5864d80a78fec037aeae26930df2004bb9f1148a69b83c25b3daf" + ), + "source_checkpoint": ( + "/opt/dpa4-runtime/models/DPA4-alloytongqi/model.pt" + ), + "source_checkpoint_sha256": ( + "c84b268cc6191afc72bd2d5c001cbe526a0d2e04ebf6dbd7df021306e9abe9ad" + ), + "type_map": "auto", + }) + with self.assertRaisesRegex(ValueError, "do not pass --model"): + self.gen.build_interaction( + "lammps", + "deepmd", + model="model.pt", + runtime_profile=profile, + ) + def test_validate_image_scass_accepts_and_rejects_combos(self): self.gen._validate_image_scass( "deepmd-kit:3.1.3", "c32_m64_cpu" @@ -189,6 +405,19 @@ def test_build_interaction_all_backends(self): with self.assertRaisesRegex(ValueError, "Unknown backend"): self.gen.build_interaction("unknown") + def test_stage_lammps_model_copies_and_returns_basename(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + source_dir = root / "source" + output_dir = root / "job" + source_dir.mkdir() + output_dir.mkdir() + source = source_dir / "model.pt2" + source.write_bytes(b"dpa4") + staged = self.gen.stage_lammps_model(str(source), output_dir) + self.assertEqual(staged, "model.pt2") + self.assertEqual((output_dir / staged).read_bytes(), b"dpa4") + def test_build_param_json_flow_types_and_dft_overrides(self): lammps = {"type": "deepmd"} joint = self.gen.build_param_json( @@ -219,6 +448,162 @@ def test_build_param_json_flow_types_and_dft_overrides(self): self.assertNotIn("relaxation", dft) self.assertEqual(dft["properties"][0]["BAND_POINTS"], 21) + def test_gamma_overrides_preserve_defaults_and_route_by_property(self): + interaction = {"type": "deepmd"} + baseline = self.gen.build_param_json( + "confs/input", + interaction, + ["gamma", "gamma_surface"], + flow_type="props", + ) + self.assertEqual( + baseline["properties"][0], + self.gen.PROPERTY_DEFAULTS["gamma"], + ) + self.assertEqual( + baseline["properties"][1], + self.gen.PROPERTY_DEFAULTS["gamma_surface"], + ) + + args = Namespace( + gamma_parent_lattice="bcc", + gamma_plane_miller=[1.0, 1.0, 0.0], + gamma_slip_direction=[1.0, -1.0, 0.0], + gamma_supercell_size=[2.0, 3.0, 4.5], + gamma_vacuum_size=20.0, + gamma_require_orthogonal_cell=True, + gamma_min_slab_height=7.5, + gamma_max_atoms=80, + gamma_min_distance=0.4, + gamma_displacement_points=[0.5, 0.0], + gamma_n_steps=5, + gamma_n_steps_x=3, + gamma_n_steps_y=4, + gamma_closed_loop=True, + ) + overrides = self.gen.gamma_overrides_from_args(args) + self.assertEqual( + self.gen.validate_gamma_cli_options( + ["gamma", "gamma_surface"], overrides + ), + [], + ) + configured = self.gen.build_param_json( + "confs/input", + interaction, + ["gamma", "gamma_surface"], + flow_type="props", + gamma_overrides=overrides, + ) + line, surface = configured["properties"] + self.assertEqual(line["parent_lattice"], "bcc") + self.assertEqual(line["supercell_size"], [2, 3, 4.5]) + self.assertEqual(line["vacuum_size"], 20.0) + self.assertTrue(line["require_orthogonal_cell"]) + self.assertEqual(line["displacement_points"], [0.0, 0.5]) + self.assertEqual(line["n_steps"], 5) + self.assertNotIn("n_steps_x", line) + self.assertEqual(surface["parent_lattice"], "bcc") + self.assertTrue(surface["require_orthogonal_cell"]) + self.assertNotIn("displacement_points", surface) + self.assertEqual(surface["n_steps_x"], 3) + self.assertEqual(surface["n_steps_y"], 4) + self.assertTrue(surface["closed_loop"]) + self.assertNotIn("n_steps", surface) + + def test_gamma_cli_options_reject_invalid_values_and_scope(self): + cases = ( + (["elastic"], {"max_atoms": 10}, "--gamma-*"), + (["gamma_surface"], {"n_steps": 2}, "--gamma-n-steps"), + (["gamma"], {"n_steps_x": 2}, "--gamma-n-steps-x"), + (["gamma"], {"plane_miller": [1, 1]}, "3 or 4"), + (["gamma"], {"slip_direction": [1, 1, 1, 1, 1]}, "3 or 4"), + ( + ["gamma"], + { + "plane_miller": [1, 1, 1], + "slip_direction": [1, 0, 0, 0], + }, + "dimensions differ", + ), + (["gamma"], {"supercell_size": [0, 1, 2]}, "component 1"), + (["gamma"], {"supercell_size": [1, 1.5, 2]}, "component 2"), + (["gamma"], {"supercell_size": [1, 1, 0]}, "plane count"), + (["gamma"], {"min_slab_height": 0}, "positive"), + (["gamma"], {"max_atoms": 0}, "positive integer"), + (["gamma"], {"min_distance": -0.1}, "non-negative"), + (["gamma"], {"n_steps": 0}, "positive integer"), + (["gamma_surface"], {"n_steps_x": 0}, "positive integer"), + (["gamma_surface"], {"n_steps_y": 0}, "positive integer"), + (["gamma"], {"closed_loop": True}, "gamma_surface"), + (["gamma"], {"parent_lattice": "b2"}, "bcc, fcc, or hcp"), + (["gamma"], {"vacuum_size": -1}, "non-negative"), + ( + ["gamma"], + {"displacement_points": [0.5, 1.0]}, + "must include 0", + ), + ( + ["gamma_surface"], + {"displacement_points": [0.0, 0.5]}, + "requires --properties gamma", + ), + ) + for properties, overrides, message in cases: + with self.subTest(properties=properties, overrides=overrides): + errors = self.gen.validate_gamma_cli_options( + properties, overrides + ) + self.assertTrue( + any(message in error for error in errors), errors + ) + + def test_melting_overrides_and_restart_validation(self): + with tempfile.TemporaryDirectory() as tmp: + restart_a = Path(tmp) / "restart.1600" + restart_b = Path(tmp) / "restart.1700" + restart_a.write_bytes(b"a") + restart_b.write_bytes(b"b") + args = Namespace( + melting_temperatures=[1600.0, 1700.0], + melting_replicas=2, + melting_restart_files=[str(restart_a), str(restart_b)], + ) + overrides = self.gen.melting_overrides_from_args(args) + self.assertEqual( + [], + self.gen.validate_melting_cli_options( + ["melting_point"], overrides + ), + ) + configured = self.gen.build_param_json( + "confs/input", + {"type": "deepmd"}, + ["melting_point"], + flow_type="props", + melting_overrides=overrides, + ) + cal = configured["properties"][0]["cal_setting"] + self.assertEqual([1600, 1700], cal["temperature"]) + self.assertEqual(2, cal["replicas"]) + self.assertEqual( + [str(restart_a), str(restart_b)], cal["restart_files"] + ) + self.assertTrue( + self.gen.validate_melting_cli_options( + ["elastic"], overrides + ) + ) + bad = dict(overrides, restart_files=[str(restart_a)]) + self.assertTrue( + any( + "one file per" in error + for error in self.gen.validate_melting_cli_options( + ["melting_point"], bad + ) + ) + ) + def test_validate_config_and_parse_str_map(self): self.assertTrue( self.gen.validate_config("vasp", None, ["finite_t_elastic"]) @@ -283,6 +668,80 @@ def test_main_generates_complete_job_directory(self): self.assertTrue(submit.stat().st_mode & stat.S_IXUSR) self.assertIn("Workflow name sanitized", stdout.getvalue()) + def test_main_dpa4_runtime_profile_is_locked_before_ticket(self): + with tempfile.TemporaryDirectory() as tmp: + structure = Path(tmp) / "Ti.vasp" + structure.write_text("structure", encoding="utf-8") + argv = [ + "generate_config.py", + "create", + "--structure", str(structure), + "--backend", "lammps", + "--runtime-profile", "dpa4-alloytongqi-t4", + "--properties", "elastic", + "--project-id", "7", + "--access-key", "key", + ] + stderr = io.StringIO() + with patch.object(sys, "argv", argv), patch.object( + self.gen, "get_bohrium_ticket" + ) as ticket, patch("sys.stderr", stderr), patch( + "sys.stdout", io.StringIO() + ): + with self.assertRaisesRegex(SystemExit, "1"): + self.gen.main() + ticket.assert_not_called() + self.assertIn("not published", stderr.getvalue()) + + def test_main_stages_melting_restarts_per_temperature(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + structure = root / "Ti.vasp" + model = root / "model.pb" + restart_a = root / "restart.1600" + restart_b = root / "restart.1700" + output = root / "job" + structure.write_text("structure", encoding="utf-8") + model.write_text("model", encoding="utf-8") + restart_a.write_bytes(b"restart-a") + restart_b.write_bytes(b"restart-b") + argv = [ + "generate_config.py", + "create", + "--structure", str(structure), + "--backend", "lammps", + "--potential", "deepmd", + "--model", str(model), + "--properties", "melting_point", + "--melting-temperatures", "1600", "1700", + "--melting-replicas", "2", + "--melting-restart-files", str(restart_a), str(restart_b), + "--output-dir", str(output), + "--project-id", "7", + "--access-key", "key", + ] + with patch.object(sys, "argv", argv), patch.object( + self.gen, "get_bohrium_ticket", return_value="t" * 36 + ), patch("sys.stdout", io.StringIO()): + self.gen.main() + + param = json.loads( + (output / "param.json").read_text(encoding="utf-8") + ) + cal = param["properties"][0]["cal_setting"] + self.assertEqual([1600, 1700], cal["temperature"]) + self.assertEqual(2, cal["replicas"]) + self.assertEqual( + ["melting_restart_000000.1600", "melting_restart_000001.1700"], + cal["restart_files"], + ) + self.assertEqual( + b"restart-a", (output / cal["restart_files"][0]).read_bytes() + ) + self.assertEqual( + b"restart-b", (output / cal["restart_files"][1]).read_bytes() + ) + def test_main_rejects_invalid_config_before_ticket_request(self): argv = [ "generate_config.py", @@ -399,22 +858,334 @@ class TestValidateInputs(unittest.TestCase): @classmethod def setUpClass(cls): cls.validator = _load_script("validate_inputs.py") + cls.gen = _load_script("generate_config.py") + + def _dpa4_global(self, exact_image): + return { + "batch_type": "Bohrium", + "context_type": "Bohrium", + "lammps_image_name": exact_image, + "lammps_run_command": self.validator.DPA4_LAMMPS_RUN_COMMAND, + "phonolammps_run_command": ( + self.validator.DPA4_PHONOLAMMPS_RUN_COMMAND + ), + "scass_type": self.validator.DPA4_SCASS_TYPE, + "group_size": self.validator.DPA4_GROUP_SIZE, + "pool_size": self.validator.DPA4_POOL_SIZE, + } + + @staticmethod + def _published_dpa4_profile(image_ref, image_digest): + return { + "published": True, + "image": {"ref": image_ref, "digest": image_digest}, + } + + def test_bundled_dpa4_source_checkpoint_is_never_a_lammps_runtime(self): + with tempfile.TemporaryDirectory() as tmpdir: + root = Path(tmpdir) + model = root / "model.pt" + model.write_bytes(b"unit-test-dpa4") + digest = hashlib.sha256(model.read_bytes()).hexdigest() + param = { + "interaction": { + "type": "deepmd", + "model": "model.pt", + "type_map": "auto", + }, + "properties": [{"type": "eos"}], + } + with patch.object(self.validator, "BUNDLED_DPA4_SHA256", digest): + errors, warnings = self.validator.validate_bundled_dpa4_runtime( + param, + {"lammps_image_name": "legacy/default"}, + root, + ) + self.assertTrue(any("never model.pt" in e for e in errors)) + self.assertFalse(warnings) + + errors, warnings = self.validator.validate_bundled_dpa4_runtime( + param, + {"lammps_image_name": "registry.example/compatible:dpa4"}, + root, + ) + self.assertTrue(any("never model.pt" in e for e in errors)) + self.assertFalse(warnings) + + def test_exact_dpa4_skill_contract_and_phonon_exception(self): + image_ref = "registry.example/apex/dpa4-runtime:tested" + image_digest = "sha256:" + "b" * 64 + exact_image = f"{image_ref}@{image_digest}" + interaction = { + "type": "deepmd", + "deepmd_runtime": self.validator.DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": self.validator.DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": self.validator.DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint": self.validator.DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": self.validator.BUNDLED_DPA4_SHA256, + "type_map": "auto", + } + param = { + "interaction": interaction, + "properties": [{"type": "phonon"}], + } + with tempfile.TemporaryDirectory() as tmpdir, patch.object( + self.validator, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile( + image_ref, + image_digest, + ), + ): + errors, warnings = self.validator.validate_bundled_dpa4_runtime( + param, self._dpa4_global(exact_image), Path(tmpdir) + ) + self.assertEqual(errors, []) + self.assertEqual(warnings, []) + + errors, _ = self.validator.validate_bundled_dpa4_runtime( + param, + self._dpa4_global("legacy/default"), + Path(tmpdir), + ) + self.assertTrue(any("lammps_image_name" in error for error in errors)) + + def test_dpa4_skill_contract_accepts_expanded_type_map(self): + image_ref = "registry.example/apex/dpa4-runtime:tested" + image_digest = "sha256:" + "e" * 64 + exact_image = f"{image_ref}@{image_digest}" + interaction = { + "type": "deepmd", + "deepmd_runtime": self.validator.DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": self.validator.DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": self.validator.DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint": self.validator.DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": self.validator.BUNDLED_DPA4_SHA256, + "type_map": {"Ti": 0, "V": 1}, + } + param = { + "interaction": interaction, + "properties": [{"type": "eos"}], + } + with tempfile.TemporaryDirectory() as tmpdir, patch.object( + self.validator, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile( + image_ref, + image_digest, + ), + ): + errors, warnings = self.validator.validate_bundled_dpa4_runtime( + param, + self._dpa4_global(exact_image), + Path(tmpdir), + ) + self.assertEqual(errors, []) + self.assertEqual(warnings, []) + + interaction["type_map"] = {"Ti": 1, "V": 3} + errors, _ = self.validator.validate_bundled_dpa4_runtime( + param, + self._dpa4_global(exact_image), + Path(tmpdir), + ) + self.assertTrue(any("contiguous" in error for error in errors)) + + def test_dpa4_skill_contract_rejects_unverified_execution_combos(self): + image_ref = "registry.example/apex/dpa4-runtime:tested" + image_digest = "sha256:" + "c" * 64 + exact_image = f"{image_ref}@{image_digest}" + interaction = { + "type": "deepmd", + "deepmd_runtime": self.validator.DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": self.validator.DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": self.validator.DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint": self.validator.DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": self.validator.BUNDLED_DPA4_SHA256, + "type_map": "auto", + } + param = { + "interaction": interaction, + "properties": [{"type": "phonon"}], + } + invalid_globals = { + "other T4 SKU": {"scass_type": "c8_m31_1 * NVIDIA T4"}, + "wrong job type": {"job_type": "not-container"}, + "wrong platform": {"platform": "not-ali"}, + "bare LAMMPS": {"lammps_run_command": "lmp -in in.lammps"}, + "bare phonoLAMMPS": { + "phonolammps_run_command": "phonolammps" + }, + "multi task": {"group_size": 2}, + "remote multi-rank wrapper": { + "dispatcher_remote_command": [ + "mpirun", + "-n", + "2", + "python3", + ] + }, + "dispatcher multi-rank wrapper": { + "dispatcher_config": { + "command": ["mpirun", "-n", "2", "python3"] + } + }, + "dispatcher remote multi-rank wrapper": { + "dispatcher_config": { + "remote_command": ["mpirun", "-n", "2", "python3"] + } + }, + "dispatcher JSON injection": { + "dispatcher_config": {"json_file": "attacker.json"} + }, + "resource override": { + "resources": {"number_node": 2, "gpu_per_node": 1} + }, + "task override": {"task": {"command": "mpirun -n 2 dpa4-lmp"}}, + "nested image override": { + "machine": { + "remote_profile": { + "input_data": { + "image_name": "registry.example/wrong:latest" + } + } + } + }, + "dispatcher nested image override": { + "dispatcher_config": { + "machine_dict": { + "context_type": "Bohrium", + "batch_type": "Bohrium", + "remote_profile": { + "input_data": { + "scass_type": self.validator.DPA4_SCASS_TYPE, + "image_name": "registry.example/wrong:latest", + } + }, + } + } + }, + "local": {"context_type": "LocalContext", "batch_type": "Shell"}, + } + with tempfile.TemporaryDirectory() as tmpdir, patch.object( + self.validator, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile( + image_ref, + image_digest, + ), + ): + for label, override in invalid_globals.items(): + global_config = self._dpa4_global(exact_image) + global_config.update(override) + with self.subTest(label=label): + errors, _ = self.validator.validate_bundled_dpa4_runtime( + param, global_config, Path(tmpdir) + ) + self.assertTrue(errors) + + errors, _ = self.validator.validate_bundled_dpa4_runtime( + param, None, Path(tmpdir) + ) + self.assertTrue(any("requires global.json" in error for error in errors)) + + def test_dpa4_skill_contract_rejects_non_lammps_base_overwrite(self): + image_ref = "registry.example/apex/dpa4-runtime:tested" + image_digest = "sha256:" + "d" * 64 + exact_image = f"{image_ref}@{image_digest}" + dpa4 = { + "type": "deepmd", + "deepmd_runtime": self.validator.DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": self.validator.DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": self.validator.DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint": self.validator.DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": self.validator.BUNDLED_DPA4_SHA256, + "type_map": "auto", + } + param = { + "interaction": { + "type": "vasp", + "potcars": {"Ti": "Ti"}, + "potcar_prefix": ".", + }, + "properties": [ + { + "type": "eos", + "cal_setting": {"overwrite_interaction": dpa4}, + } + ], + } + with tempfile.TemporaryDirectory() as tmpdir, patch.object( + self.validator, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile( + image_ref, + image_digest, + ), + ): + errors, _ = self.validator.validate_bundled_dpa4_runtime( + param, self._dpa4_global(exact_image), Path(tmpdir) + ) + self.assertTrue( + any("requires a LAMMPS base interaction" in error for error in errors) + ) + + def test_dpa4_skill_contract_rejects_placeholder_wrong_hash_and_mix(self): + interaction = { + "type": "deepmd", + "deepmd_runtime": self.validator.DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": self.validator.DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": "0" * 64, + "source_checkpoint": self.validator.DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": self.validator.BUNDLED_DPA4_SHA256, + "type_map": "auto", + } + param = { + "interaction": interaction, + "properties": [ + {"type": "eos"}, + { + "type": "elastic", + "cal_setting": { + "overwrite_interaction": { + "type": "deepmd", + "model": "legacy.pb", + "type_map": "auto", + } + }, + }, + ], + } + with tempfile.TemporaryDirectory() as tmpdir: + errors, _ = self.validator.validate_bundled_dpa4_runtime( + param, {"lammps_image_name": "anything"}, Path(tmpdir) + ) + self.assertTrue(any("runtime_model_sha256" in error for error in errors)) + self.assertTrue(any("cannot mix legacy" in error for error in errors)) + self.assertTrue(any("not finalized" in error for error in errors)) def test_validate_global(self): errors, warnings = self.validator.validate_global({}) self.assertTrue( - any("Missing 'machine' section" in error for error in errors) + any("Missing local execution configuration" in error for error in errors) ) self.assertTrue( any("Missing 'run_command'" in error for error in errors) ) - self.assertTrue(warnings) + self.assertFalse(warnings) errors, warnings = self.validator.validate_global( {"machine": {}, "run_command": "run"} ) - self.assertIn("Missing 'machine.batch_type'", errors) - self.assertTrue(warnings) + self.assertTrue( + any("machine.batch_type" in error for error in errors) + ) + self.assertFalse(warnings) self.assertEqual( self.validator.validate_global( @@ -427,6 +1198,106 @@ def test_validate_global(self): ([], []), ) + def test_validate_global_execution_profiles(self): + data_root = get_skill_root() / "data" + bohrium_direct = json.loads( + (data_root / "global_bohrium_direct.json").read_text() + ) + errors, warnings = self.validator.validate_global(bohrium_direct) + self.assertFalse(errors) + self.assertTrue(any("apex account --show" in warning for warning in warnings)) + + local_debug = json.loads( + (data_root / "global_local_debug.json").read_text() + ) + self.assertEqual( + self.validator.validate_global(local_debug), + ([], []), + ) + + cluster_template = json.loads( + (data_root / "global_local_cluster_slurm.json").read_text() + ) + errors, _ = self.validator.validate_global(cluster_template) + self.assertTrue(any("placeholders" in error for error in errors)) + + local_cluster = { + "context_type": "Local", + "run_command": "srun lmp -in in.lammps", + "machine": { + "batch_type": "Slurm", + "context_type": "Local", + }, + "resources": { + "number_node": 1, + "custom_flags": ["#SBATCH --partition=compute"], + }, + } + self.assertEqual( + self.validator.validate_global(local_cluster), + ([], []), + ) + + pbs_cluster = { + "context_type": "Local", + "run_command": "mpirun lmp -in in.lammps", + "machine": { + "batch_type": "PBS", + "context_type": "Local", + }, + "resources": {"number_node": 1}, + } + self.assertEqual( + self.validator.validate_global(pbs_cluster), + ([], []), + ) + + local_cluster["run_command"] = "" + local_cluster["resources"]["custom_flags"] = [ + "#SBATCH --partition=" + ] + errors, _ = self.validator.validate_global(local_cluster) + self.assertTrue(any("placeholders" in error for error in errors)) + + def test_validate_global_openapi_sandbox(self): + config = { + "dflow_host": "https://lbg-workflow-dflow.dp.tech", + "k8s_api_server": "https://lbg-workflow-dflow.dp.tech", + "dflow_config": { + "host": "https://lbg-workflow-dflow.dp.tech", + "k8s_api_server": "https://lbg-workflow-dflow.dp.tech", + "namespace": "dflow", + "token": "", + }, + "batch_type": "OpenAPI", + "context_type": "OpenAPI", + "access_key": "test-key", + "project_id": 42, + "machine_type": "c8_m32_1 * NVIDIA 4090", + "image_address": self.gen.LAMMPS_IMAGE, + "dispatcher_image": "dispatcher:image", + "bohrium_config": { + "access_key": "test-key", + "project_id": 42, + "app_key": "agent", + }, + "lammps_image_name": self.gen.LAMMPS_IMAGE, + "lammps_run_command": "lmp -in in.lammps", + } + self.assertEqual(self.validator.validate_global(config), ([], [])) + config["bohrium_config"]["project_id"] = "42" + errors, _ = self.validator.validate_global(config) + self.assertTrue(any("bohrium_config.project_id" in e for e in errors)) + + def test_validate_melting_point_property(self): + prop = self.gen.PROPERTY_DEFAULTS["melting_point"] + self.assertEqual( + self.validator.validate_properties([prop], "deepmd"), + ([], []), + ) + errors, _ = self.validator.validate_properties([prop], "vasp") + self.assertTrue(any("LAMMPS-only" in e for e in errors)) + def test_validate_current_global_requires_integer_matching_project_ids(self): config = { "dflow_host": "https://workflows.deepmodeling.com", @@ -504,6 +1375,268 @@ def test_validate_dft_kspacing_requires_vasp_kspacing(self): self.assertEqual(errors, []) self.assertEqual(warnings, []) + def test_validate_gamma_guard_settings(self): + base = { + "type": "gamma", + "plane_miller": [1, 1, 1], + "slip_direction": [-1, 1, 0], + } + errors, warnings = self.validator.validate_properties( + [base], "lammps" + ) + self.assertFalse(errors) + self.assertTrue(any("min_slab_height" in item for item in warnings)) + self.assertTrue(any("max_atoms" in item for item in warnings)) + + invalid_cases = ( + ({"supercell_size": [0, 1, 2]}, "supercell_size[0]"), + ({"supercell_size": [1, 1.5, 2]}, "supercell_size[1]"), + ({"supercell_size": [1, 1, float("inf")]}, "positive finite"), + ({"min_slab_height": 0}, "min_slab_height"), + ({"max_atoms": True}, "max_atoms"), + ({"min_distance": -0.1}, "min_distance"), + ({"n_steps": 0}, "n_steps"), + ) + for update, message in invalid_cases: + with self.subTest(update=update): + prop = dict(base) + prop.update(update) + errors, _ = self.validator.validate_properties( + [prop], "lammps" + ) + self.assertTrue( + any(message in item for item in errors), errors + ) + + def test_gamma_preflight_reports_slab_and_enforces_atom_limit(self): + from pymatgen.core import Lattice, Structure + + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + conf = root / "conf" + conf.mkdir() + structure = Structure( + Lattice.cubic(4.0), + ["Al", "Al", "Al", "Al"], + [ + [0, 0, 0], + [0, 0.5, 0.5], + [0.5, 0, 0.5], + [0.5, 0.5, 0], + ], + ) + structure.to(filename=conf / "POSCAR", fmt="poscar") + (root / "INCAR").write_text( + "KSPACING = 100\nKGAMMA = .TRUE.\n" + "NCORE = 2\nKPAR = 1\n", + encoding="utf-8", + ) + prop = { + "type": "gamma", + "plane_miller": [1, 1, 1], + "slip_direction": [-1, 1, 0], + "supercell_size": [1, 1, 2], + "min_slab_height": 1.0, + "max_atoms": 100, + "min_distance": 0.1, + "n_steps": 2, + } + param = { + "structures": ["conf"], + "interaction": {"type": "vasp", "incar": "INCAR"}, + "properties": [prop], + } + reports, errors, warnings = ( + self.validator.preflight_gamma_structures(param, root) + ) + self.assertFalse(errors) + self.assertFalse(warnings) + self.assertEqual(len(reports), 1) + self.assertEqual(reports[0]["parent_atom_count"], 4) + self.assertLessEqual(reports[0]["atom_count"], 100) + self.assertGreaterEqual(reports[0]["slab_height"], 1.0) + self.assertEqual(reports[0]["expected_task_count"], 3) + self.assertEqual( + reports[0]["kpoints"], + {"style": "Gamma", "grid": [1, 1, 1]}, + ) + + param["properties"][0]["max_atoms"] = 1 + reports, errors, _ = self.validator.preflight_gamma_structures( + param, root + ) + self.assertFalse(reports) + self.assertTrue(any("exceeding max_atoms=1" in item for item in errors)) + + def test_validate_vasp_parallel_and_vasp_gam_rules(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + incar = root / "INCAR" + command = ( + 'bash -c "source /opt/intel/oneapi/setvars.sh && ' + "ulimit -s unlimited && mpirun -n 8 " + '/opt/vasp.5.4.4/bin/vasp_gam"' + ) + global_config = { + "batch_type": "Bohrium", + "vasp_run_command": command, + "scass_type": "c8_m16_cpu", + } + param = { + "interaction": {"type": "vasp", "incar": "INCAR"}, + "properties": [{"type": "gamma"}], + } + gamma_report = { + "label": "synthetic Gamma slab", + "kpoints": {"style": "Gamma", "grid": [1, 1, 1]}, + } + + incar.write_text( + "NCORE = 2\nKPAR = 1\n", encoding="utf-8" + ) + errors, warnings = self.validator.validate_vasp_parallel_settings( + param, global_config, root, [gamma_report] + ) + self.assertEqual(errors, []) + self.assertEqual(warnings, []) + + report = dict(gamma_report) + report["kpoints"] = {"style": "Gamma", "grid": [2, 1, 1]} + errors, _ = self.validator.validate_vasp_parallel_settings( + param, global_config, root, [report] + ) + self.assertEqual(errors, []) + + incar.write_text( + "NCORE = 3\nKPAR = 2\n", encoding="utf-8" + ) + errors, _ = self.validator.validate_vasp_parallel_settings( + param, global_config, root, [gamma_report] + ) + self.assertTrue(any("require KPAR=1" in item for item in errors)) + self.assertTrue(any("NCORE=3" in item for item in errors)) + + global_config["scass_type"] = "c4_m8_cpu" + errors, _ = self.validator.validate_vasp_parallel_settings( + param, global_config, root, [gamma_report] + ) + self.assertTrue(any("must match Bohrium CPU" in item for item in errors)) + + global_config["scass_type"] = "c8_m16_cpu" + incar.write_text("KPAR = 1\n", encoding="utf-8") + errors, warnings = self.validator.validate_vasp_parallel_settings( + param, global_config, root, [gamma_report] + ) + self.assertFalse(errors) + self.assertTrue(any("NCORE is not set" in item for item in warnings)) + + incar.write_text( + "NCORE = 2\nNPAR = 4\nKPAR = 1\n", encoding="utf-8" + ) + errors, _ = self.validator.validate_vasp_parallel_settings( + param, global_config, root, [gamma_report] + ) + self.assertTrue(any("same time" in item for item in errors)) + + def test_non_bohrium_vasp_does_not_compare_ranks_with_scass_type(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "INCAR").write_text( + "NCORE = 2\nKPAR = 1\n", encoding="utf-8" + ) + param = { + "interaction": {"type": "vasp", "incar": "INCAR"}, + } + common = { + "vasp_run_command": "mpirun -n 8 /cluster/bin/vasp_std", + "scass_type": "c4_m8_cpu", + } + profiles = { + "local": { + "context_type": "Local", + "batch_type": "Shell", + }, + "dpdispatcher": { + "context_type": "Local", + "machine": { + "context_type": "Local", + "batch_type": "Slurm", + }, + }, + } + + for profile, config in profiles.items(): + with self.subTest(profile=profile): + errors, warnings = ( + self.validator.validate_vasp_parallel_settings( + param, {**common, **config}, root, [] + ) + ) + self.assertEqual(errors, []) + self.assertEqual(warnings, []) + + def test_validate_vasp_gam_selection_uses_sampling_not_property_name(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + conf = root / "confs" / "alloy" + conf.mkdir(parents=True) + (conf / "POSCAR").write_text( + "TiV\n" + "1.0\n" + "10 0 0\n" + "0 10 0\n" + "0 0 10\n" + "Ti V\n" + "1 1\n" + "Direct\n" + "0 0 0\n" + "0.5 0.5 0.5\n", + encoding="utf-8", + ) + incar = root / "INCAR" + incar.write_text( + "KSPACING = 1.0\nKGAMMA = True\nNCORE = 2\nKPAR = 2\n", + encoding="utf-8", + ) + param = { + "structures": ["confs/alloy"], + "interaction": {"type": "vasp", "incar": "INCAR"}, + "relaxation": {"req_calc": False}, + "properties": [ + { + "type": "finite_t_latt", + "supercell_size": [1, 1, 1], + } + ], + } + global_config = { + "vasp_run_command": ( + 'bash -c "source /opt/intel/oneapi/setvars.sh && ' + "ulimit -s unlimited && mpirun -n 8 " + '/opt/vasp.5.4.4/bin/vasp_std"' + ), + "scass_type": "c8_m16_cpu", + } + errors, _ = self.validator.validate_vasp_parallel_settings( + param, global_config, root, [] + ) + self.assertTrue(any("require KPAR=1" in item for item in errors)) + self.assertFalse( + any("non-Gamma properties" in item for item in errors) + ) + + incar.write_text( + "KSPACING = 1.0\nKGAMMA = True\nNCORE = 2\nKPAR = 1\n", + encoding="utf-8", + ) + errors, warnings = ( + self.validator.validate_vasp_parallel_settings( + param, global_config, root, [] + ) + ) + self.assertEqual(errors, []) + self.assertEqual(warnings, []) + def test_validate_interaction(self): cases = ( ({}, "Missing 'interaction.type'"), @@ -558,6 +1691,35 @@ def test_validate_property_required_fields(self): ) self.assertTrue(any("LAMMPS-only" in error for error in errors)) + valid_melting = { + "type": "melting_point", + "method": "two_phase", + "supercell_size": [1, 1, 2], + "cal_setting": { + "temperature": [1600, 1650, 1700], + "production_steps": 100000, + "interface_axis": "z", + }, + } + errors, _ = self.validator.validate_properties( + [valid_melting], "deepmd" + ) + self.assertEqual([], errors) + errors, _ = self.validator.validate_properties( + [valid_melting], "vasp" + ) + self.assertTrue(any("LAMMPS-only" in error for error in errors)) + invalid_melting = json.loads(json.dumps(valid_melting)) + invalid_melting["cal_setting"]["temperature"] = [] + invalid_melting["cal_setting"]["interface_axis"] = "bad" + invalid_melting["cal_setting"]["restart_interval"] = 0 + errors, _ = self.validator.validate_properties( + [invalid_melting], "deepmd" + ) + self.assertTrue(any("non-empty" in error for error in errors)) + self.assertTrue(any("interface_axis" in error for error in errors)) + self.assertTrue(any("restart_interval" in error for error in errors)) + def test_validate_gruneisen_and_gamma_geometry(self): errors, _ = self.validator.validate_properties( [{ @@ -614,6 +1776,57 @@ def test_validate_gamma_surface_steps(self): self.assertTrue(any("n_steps_x" in error for error in errors)) self.assertTrue(any("n_steps_y" in error for error in errors)) + def test_validate_parent_gamma_displacements_and_restart_boundary(self): + valid_gamma = { + "type": "gamma", + "parent_lattice": "bcc", + "plane_miller": [1, 1, 0], + "slip_direction": [-1, 1, 1], + "vacuum_size": 20, + "require_orthogonal_cell": True, + "displacement_points": [0.0, 0.5], + } + errors, _ = self.validator.validate_properties( + [valid_gamma], "lammps" + ) + self.assertEqual([], errors) + + for key, value, message in ( + ("parent_lattice", "b2", "parent_lattice"), + ("vacuum_size", -1, "vacuum_size"), + ("require_orthogonal_cell", "true", "must be a boolean"), + ("displacement_points", [0.5], "include 0"), + ): + prop = dict(valid_gamma, **{key: value}) + errors, _ = self.validator.validate_properties([prop], "lammps") + self.assertTrue(any(message in error for error in errors), errors) + + finite_latt = { + "type": "finite_t_latt", + "cal_setting": {"restart_files": ["restart.bad"]}, + } + errors, _ = self.validator.validate_properties( + [finite_latt], "lammps" + ) + self.assertTrue( + any("supported only by melting_point" in error for error in errors) + ) + + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "restart.1600").write_bytes(b"restart") + melting = { + "type": "melting_point", + "cal_setting": { + "temperature": [1600, 1700], + "restart_files": ["restart.1600", "restart.missing"], + }, + } + errors, _ = self.validator.validate_properties( + [melting], "deepmd", base_dir=root + ) + self.assertTrue(any("not found" in error for error in errors)) + def test_validate_structures(self): with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) @@ -684,6 +1897,62 @@ def test_main_success_and_strict_warning(self): self.assertEqual(code, 1) self.assertIn("strict mode", stderr) + def test_main_accepts_relaxation_only_with_missing_or_empty_properties(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "POSCAR").write_text("structure", encoding="utf-8") + param = root / "param.json" + payload = { + "structures": ["POSCAR"], + "interaction": { + "type": "deepmd", + "model": "model.pb", + "type_map": "auto", + }, + "relaxation": { + "cal_type": "relaxation", + "cal_setting": {}, + }, + } + + for properties in (None, []): + with self.subTest(properties=properties): + candidate = dict(payload) + if properties is not None: + candidate["properties"] = properties + param.write_text(json.dumps(candidate)) + + code, stdout, stderr = self._run_main([ + "validate_inputs.py", "--param", str(param), + ]) + + self.assertEqual(code, 0) + self.assertIn("Validation PASSED", stdout) + self.assertIn("Properties: []", stdout) + self.assertNotIn("No properties defined", stderr) + + def test_main_still_rejects_empty_property_only_input(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "POSCAR").write_text("structure", encoding="utf-8") + param = root / "param.json" + param.write_text(json.dumps({ + "structures": ["POSCAR"], + "interaction": { + "type": "deepmd", + "model": "model.pb", + "type_map": "auto", + }, + "properties": [], + })) + + code, _, stderr = self._run_main([ + "validate_inputs.py", "--param", str(param), + ]) + + self.assertEqual(code, 1) + self.assertIn("No properties defined", stderr) + def test_main_reports_missing_and_invalid_sections(self): with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) @@ -705,6 +1974,81 @@ def test_main_reports_missing_and_invalid_sections(self): self.assertIn("Missing 'interaction'", stderr) +class TestDpa4Profile(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.profile_mod = _load_script("dpa4_profile.py") + + def test_bundled_profile_is_pre_snapshot_and_fails_closed(self): + profile = self.profile_mod.load_dpa4_profile( + require_published=False + ) + self.assertFalse(profile["published"]) + self.assertFalse(profile["identity_finalized"]) + self.assertFalse(profile["qualified"]) + self.assertEqual( + profile["machine_compatibility"]["recommended"][0][ + "scass_type" + ], + "c4_m15_1 * NVIDIA T4", + ) + self.assertEqual( + profile["calculator"]["run_command"], + self.profile_mod.DPA4_LAMMPS_RUN_COMMAND, + ) + self.assertEqual( + profile["calculator"]["phonolammps_command"], + self.profile_mod.DPA4_PHONOLAMMPS_RUN_COMMAND, + ) + with self.assertRaisesRegex(RuntimeError, "not published"): + self.profile_mod.load_dpa4_profile(require_published=True) + + def test_published_profile_produces_exact_image_and_interaction(self): + source = json.loads( + self.profile_mod.DPA4_PROFILE_PATH.read_text(encoding="utf-8") + ) + source["qualification_status"] = "post_snapshot_passed" + source["image"] = { + "ref": "registry.example/dpa4:qualified", + "digest": "sha256:" + "e" * 64, + } + with tempfile.TemporaryDirectory() as tmpdir: + path = Path(tmpdir) / "profile.json" + path.write_text(json.dumps(source), encoding="utf-8") + profile = self.profile_mod.load_dpa4_profile(path=path) + self.assertEqual( + self.profile_mod.dpa4_image_name(profile), + "registry.example/dpa4:qualified@sha256:" + "e" * 64, + ) + interaction = self.profile_mod.dpa4_interaction(profile) + self.assertEqual( + interaction["model"], self.profile_mod.DPA4_RUNTIME_MODEL_PATH + ) + self.assertEqual( + interaction["source_checkpoint_sha256"], + self.profile_mod.DPA4_SOURCE_CHECKPOINT_SHA256, + ) + + def test_profile_rejects_tampered_runtime_or_wrapper(self): + source = json.loads( + self.profile_mod.DPA4_PROFILE_PATH.read_text(encoding="utf-8") + ) + source["runtime"]["model_sha256"] = "0" * 64 + with self.assertRaisesRegex(RuntimeError, "runtime.model_sha256"): + self.profile_mod.validate_dpa4_profile( + source, require_published=False + ) + + source = json.loads( + self.profile_mod.DPA4_PROFILE_PATH.read_text(encoding="utf-8") + ) + source["calculator"]["run_command"] = "dpa4-lmp -in in.lammps" + with self.assertRaisesRegex(RuntimeError, "audited dpa4-lmp"): + self.profile_mod.validate_dpa4_profile( + source, require_published=False + ) + + class TestValidateComboAdditionalPaths(unittest.TestCase): @classmethod def setUpClass(cls): @@ -761,6 +2105,79 @@ def test_list_and_recommend_cli_text_and_json(self): self.assertEqual(self.combo.main(argv), 0) self.assertIn(expected, stdout.getvalue()) + def test_dpa4_unpublished_profile_lists_candidate_but_never_recommends(self): + data = self.combo.list_combos( + runtime_profile=self.combo.DPA4_RUNTIME_PROFILE + ) + self.assertEqual(data["status"], "unpublished") + self.assertIsNone(data["recommended"]) + self.assertEqual( + data["candidate_after_publish"]["scass_type"], + "c4_m15_1 * NVIDIA T4", + ) + self.assertTrue(any( + "c8_m31_1" in item for item in data["unverified"] + )) + with self.assertRaisesRegex(RuntimeError, "not published"): + self.combo.recommend( + runtime_profile=self.combo.DPA4_RUNTIME_PROFILE + ) + + ok, errors = self.combo.check_combo( + "registry.example/dpa4:tag@sha256:" + "a" * 64, + "c4_m15_1 * NVIDIA T4", + runtime_profile=self.combo.DPA4_RUNTIME_PROFILE, + ) + self.assertFalse(ok) + self.assertTrue(any("not published" in error for error in errors)) + + stdout = io.StringIO() + with patch("sys.stdout", stdout): + self.assertEqual(self.combo.main([ + "list-combos", + "--runtime-profile", self.combo.DPA4_RUNTIME_PROFILE, + ]), 0) + text = stdout.getvalue() + self.assertIn("recommendation: LOCKED", text) + self.assertIn("Prohibited classes:", text) + self.assertIn("multi_gpu", text) + + def test_dpa4_exact_combo_rejects_unverified_gpu_and_multi_rank(self): + profile = { + "published": True, + "image": { + "ref": "registry.example/dpa4:qualified", + "digest": "sha256:" + "d" * 64, + }, + "machine_compatibility": { + "recommended": [{ + "scass_type": "c4_m15_1 * NVIDIA T4", + "mpi_ranks": 1, + "gpu_count": 1, + }], + "prohibited_exact": {}, + }, + } + exact = profile["image"]["ref"] + "@" + profile["image"]["digest"] + with patch.object(self.combo, "_dpa4_profile", return_value=profile): + ok, errors = self.combo.check_combo( + exact, + "c4_m15_1 * NVIDIA T4", + runtime_profile=self.combo.DPA4_RUNTIME_PROFILE, + ) + self.assertTrue(ok) + self.assertEqual(errors, []) + + ok, errors = self.combo.check_combo( + exact, + "c8_m31_1 * NVIDIA T4", + runtime_profile=self.combo.DPA4_RUNTIME_PROFILE, + mpi_ranks=2, + ) + self.assertFalse(ok) + self.assertTrue(any("unverified" in error for error in errors)) + self.assertTrue(any("one MPI rank" in error for error in errors)) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_structure.py b/tests/test_structure.py new file mode 100644 index 00000000..2c4a3163 --- /dev/null +++ b/tests/test_structure.py @@ -0,0 +1,29 @@ +import unittest + +from apex.core.structure import ( + normalize_parent_lattice_hint, + resolve_parent_lattice_hint, +) + + +class TestParentLatticeHint(unittest.TestCase): + def test_no_hint_preserves_detected_type(self): + self.assertEqual( + resolve_parent_lattice_hint("other"), + ("other", "auto_detected"), + ) + + def test_hint_overrides_detected_type(self): + self.assertEqual( + resolve_parent_lattice_hint("other", "BCC"), + ("bcc", "user_override"), + ) + + def test_normalization_and_validation(self): + self.assertEqual(normalize_parent_lattice_hint(" HCP "), "hcp") + with self.assertRaisesRegex(ValueError, "bcc, fcc, hcp"): + normalize_parent_lattice_hint("diamond") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_submit_path_validation.py b/tests/test_submit_path_validation.py index e880050c..4b0502ad 100644 --- a/tests/test_submit_path_validation.py +++ b/tests/test_submit_path_validation.py @@ -17,10 +17,48 @@ class TestSubmitPathValidation(unittest.TestCase): + DPA4_TEST_REF = "registry.example/apex/dpa4-runtime:tested" + DPA4_TEST_DIGEST = "sha256:" + "a" * 64 + + @classmethod + def _published_dpa4_profile(cls): + return { + "published": True, + "image": { + "ref": cls.DPA4_TEST_REF, + "digest": cls.DPA4_TEST_DIGEST, + }, + } + + @staticmethod + def _dpa4_interaction(): + return { + "type": "deepmd", + "deepmd_runtime": submit_module.DPA4_RUNTIME_KIND, + "model_in_image": True, + "model": submit_module.DPA4_RUNTIME_MODEL_PATH, + "runtime_model_sha256": submit_module.DPA4_RUNTIME_MODEL_SHA256, + "source_checkpoint": submit_module.DPA4_SOURCE_CHECKPOINT_PATH, + "source_checkpoint_sha256": ( + submit_module.DPA4_SOURCE_CHECKPOINT_SHA256 + ), + "type_map": "auto", + } + + def test_lammps_phonon_image_matches_validated_phonolammps_runtime(self): + self.assertEqual( + LAMMPS_PHONON_IMAGE, + "registry.dp.tech/dptech/dp/native/prod-16664/" + "dpa4-phonolammps:0.0.2", + ) + def test_lammps_phonon_forces_validated_image(self): selected = _select_run_image( "lammps", - {"properties": [{"type": "phonon"}]}, + { + "interaction": {"type": "deepmd"}, + "properties": [{"type": "phonon"}], + }, "registry.dp.tech/dptech/deepmd-kit:3.1.3", ) self.assertEqual(selected, LAMMPS_PHONON_IMAGE) @@ -28,11 +66,45 @@ def test_lammps_phonon_forces_validated_image(self): def test_lammps_gruneisen_forces_validated_image(self): selected = _select_run_image( "lammps", - {"properties": [{"type": "gruneisen"}]}, + { + "interaction": {"type": "deepmd"}, + "properties": [{"type": "gruneisen"}], + }, "registry.dp.tech/dptech/deepmd-kit:3.1.3", ) self.assertEqual(selected, LAMMPS_PHONON_IMAGE) + def test_cpu_lammps_phonon_keeps_configured_image(self): + configured = ( + "registry.dp.tech/dptech/dp/native/prod-397637/" + "apex-flow:1.3.0.post" + ) + selected = _select_run_image( + "lammps", + { + "interaction": {"type": "eam_alloy"}, + "properties": [{"type": "phonon"}], + }, + configured, + ) + self.assertEqual(selected, configured) + + def test_deepmd_cpu_phonon_does_not_force_gpu_image(self): + configured = ( + "registry.dp.tech/dptech/dp/native/prod-397637/" + "deepmd-kit-phonolammps:3.1.3" + ) + selected = _select_run_image( + "lammps", + { + "interaction": {"type": "deepmd"}, + "properties": [{"type": "phonon"}], + }, + configured, + "c8_m32_cpu", + ) + self.assertEqual(selected, configured) + def test_non_phonon_keeps_configured_lammps_image(self): configured = "registry.example/custom-lammps:latest" selected = _select_run_image( @@ -42,6 +114,328 @@ def test_non_phonon_keeps_configured_lammps_image(self): ) self.assertEqual(selected, configured) + def test_dpa4_image_placeholders_fail_closed(self): + with self.assertRaisesRegex(RuntimeError, "not published"): + submit_module._dpa4_image_name() + + partial = self._dpa4_interaction() + partial["deepmd_runtime"] = "legacy" + props = {"interaction": partial, "properties": [{"type": "eos"}]} + with tempfile.TemporaryDirectory() as work_dir, self.assertRaisesRegex( + RuntimeError, "deepmd_runtime" + ): + submit_module._validate_lammps_runtime_contract( + None, props, "props", [work_dir] + ) + + def test_dpa4_image_identity_comes_from_canonical_profile(self): + profile = self._published_dpa4_profile() + with patch.object( + submit_module, + "_load_dpa4_profile", + return_value=profile, + ) as loader: + self.assertEqual( + submit_module._dpa4_image_name(), + f"{self.DPA4_TEST_REF}@{self.DPA4_TEST_DIGEST}", + ) + loader.assert_called_once_with(require_published=True) + + def test_exact_dpa4_phonon_uses_candidate_not_legacy_override(self): + exact_image = f"{self.DPA4_TEST_REF}@{self.DPA4_TEST_DIGEST}" + with tempfile.TemporaryDirectory() as work_dir, patch.object( + submit_module, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile(), + ): + for prop_type in ("phonon", "gruneisen"): + with self.subTest(prop_type=prop_type): + props = { + "interaction": self._dpa4_interaction(), + "properties": [{"type": prop_type}], + } + contract = submit_module._validate_lammps_runtime_contract( + None, props, "props", [work_dir] + ) + selected = _select_run_image( + "lammps", + props, + LAMMPS_PHONON_IMAGE, + runtime_contract=contract, + ) + self.assertEqual( + contract, submit_module.DPA4_RUNTIME_KIND + ) + self.assertEqual(selected, exact_image) + self.assertNotEqual(selected, LAMMPS_PHONON_IMAGE) + + def test_dpa4_contract_rejects_wrong_hash_image_and_partial_intent(self): + exact_image = f"{self.DPA4_TEST_REF}@{self.DPA4_TEST_DIGEST}" + interaction = self._dpa4_interaction() + interaction["runtime_model_sha256"] = "0" * 64 + props = { + "interaction": interaction, + "properties": [{"type": "eos"}], + } + with tempfile.TemporaryDirectory() as work_dir, patch.object( + submit_module, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile(), + ): + with self.assertRaisesRegex(RuntimeError, "runtime_model_sha256"): + submit_module._validate_lammps_runtime_contract( + None, props, "props", [work_dir] + ) + + props["interaction"] = self._dpa4_interaction() + props["interaction"]["type_map"] = {"Ti": 0, "V": 1} + self.assertEqual( + submit_module._validate_lammps_runtime_contract( + None, props, "props", [work_dir] + ), + submit_module.DPA4_RUNTIME_KIND, + ) + + props["interaction"]["type_map"] = {"Ti": 1, "V": 3} + with self.assertRaisesRegex(RuntimeError, "contiguous"): + submit_module._validate_lammps_runtime_contract( + None, props, "props", [work_dir] + ) + + props["interaction"] = self._dpa4_interaction() + contract = submit_module._validate_lammps_runtime_contract( + None, props, "props", [work_dir] + ) + self.assertEqual(contract, submit_module.DPA4_RUNTIME_KIND) + self.assertEqual( + _select_run_image( + "lammps", + props, + "registry.example/wrong:tag", + runtime_contract=contract, + ), + exact_image, + ) + + def test_dpa4_contract_rejects_legacy_overwrite_and_joint_mix(self): + props = { + "interaction": self._dpa4_interaction(), + "properties": [ + {"type": "eos"}, + { + "type": "elastic", + "cal_setting": { + "overwrite_interaction": { + "type": "deepmd", + "model": "legacy.pb", + "type_map": {"Ti": 0, "V": 1}, + } + }, + }, + ], + } + with tempfile.TemporaryDirectory() as work_dir, patch.object( + submit_module, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile(), + ): + with self.assertRaisesRegex(RuntimeError, "cannot mix legacy"): + submit_module._validate_lammps_runtime_contract( + None, props, "props", [work_dir] + ) + + relax = { + "interaction": { + "type": "deepmd", + "model": "legacy.pb", + "type_map": {"Ti": 0, "V": 1}, + } + } + props["properties"] = [{"type": "eos"}] + with self.assertRaisesRegex(RuntimeError, "cannot mix legacy"): + submit_module._validate_lammps_runtime_contract( + relax, props, "joint", [work_dir] + ) + + def test_dpa4_overwrite_cannot_hide_under_vasp_base_calculator(self): + props = { + "interaction": { + "type": "vasp", + "potcars": {"Ti": "Ti"}, + "potcar_prefix": "/potcars", + }, + "properties": [ + { + "type": "eos", + "cal_setting": { + "overwrite_interaction": self._dpa4_interaction() + }, + } + ], + } + with tempfile.TemporaryDirectory() as work_dir, patch.object( + submit_module, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile(), + ), patch.object( + submit_module, + "judge_flow", + return_value=(object(), "vasp", "props", None, props), + ), self.assertRaisesRegex(RuntimeError, "base calculator is 'vasp'"): + submit_workflow([props], {}, [work_dir], "props") + + def test_dpa4_execution_profile_is_exact_t4_single_rank(self): + def make_config(**overrides): + values = { + "context_type": "Bohrium", + "batch_type": "Bohrium", + "scass_type": submit_module.DPA4_SCASS_TYPE, + "lammps_run_command": submit_module.DPA4_LAMMPS_RUN_COMMAND, + "phonolammps_run_command": ( + submit_module.DPA4_PHONOLAMMPS_RUN_COMMAND + ), + "group_size": submit_module.DPA4_GROUP_SIZE, + "pool_size": submit_module.DPA4_POOL_SIZE, + } + values.update(overrides) + return submit_module.Config(**values) + + phonon = {"properties": [{"type": "phonon"}]} + submit_module._validate_dpa4_execution_config(make_config(), phonon) + + invalid_cases = { + "wrong SKU": {"scass_type": "c8_m31_1 * NVIDIA T4"}, + "wrong job type": {"job_type": "not-container"}, + "wrong platform": {"platform": "not-ali"}, + "CPU command": {"lammps_run_command": "lmp -in in.lammps"}, + "missing phono wrapper": {"phonolammps_run_command": None}, + "grouped tasks": {"group_size": 2}, + "pooled tasks": {"pool_size": 2}, + "remote multi-rank wrapper": { + "dispatcher_remote_command": [ + "mpirun", + "-n", + "2", + "python3", + ] + }, + "dispatcher multi-rank wrapper": { + "dispatcher_config": { + "command": ["mpirun", "-n", "2", "python3"] + } + }, + "dispatcher remote multi-rank wrapper": { + "dispatcher_config": { + "remote_command": ["mpirun", "-n", "2", "python3"] + } + }, + "dispatcher JSON injection": { + "dispatcher_config": {"json_file": "attacker.json"} + }, + "resource override": { + "resources": {"number_node": 2, "gpu_per_node": 1} + }, + "task override": {"task": {"command": "mpirun -n 2 dpa4-lmp"}}, + "local context": { + "context_type": "LocalContext", + "batch_type": "Shell", + }, + } + for label, overrides in invalid_cases.items(): + with self.subTest(label=label), self.assertRaisesRegex( + RuntimeError, "Invalid DPA4 execution profile" + ): + submit_module._validate_dpa4_execution_config( + make_config(**overrides), phonon + ) + + nested_override = { + "machine_dict": { + "context_type": "Bohrium", + "batch_type": "Bohrium", + "remote_profile": { + "input_data": { + "scass_type": "c16_m62_1 * NVIDIA T4" + } + }, + } + } + with self.assertRaisesRegex(RuntimeError, "nested machine overrides"): + submit_module._validate_dpa4_execution_config( + make_config(dispatcher_config=nested_override), phonon + ) + + wrong_image_override = { + "machine_dict": { + "context_type": "Bohrium", + "batch_type": "Bohrium", + "remote_profile": { + "input_data": { + "scass_type": submit_module.DPA4_SCASS_TYPE, + "image_name": "registry.example/wrong:latest", + } + }, + } + } + with patch.object( + submit_module, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile(), + ), self.assertRaisesRegex(RuntimeError, "nested image overrides"): + submit_module._validate_dpa4_execution_config( + make_config(dispatcher_config=wrong_image_override), phonon + ) + + def test_dpa4_source_checkpoint_in_any_workdir_is_never_a_runtime(self): + props = { + "interaction": { + "type": "deepmd", + "model": "model.pt", + "type_map": {"Ti": 0, "V": 1}, + }, + "properties": [{"type": "eos"}], + } + with tempfile.TemporaryDirectory() as first, tempfile.TemporaryDirectory() as second: + with open(os.path.join(first, "model.pt"), "wb") as stream: + stream.write(b"exact-checkpoint-test") + digest = submit_module._sha256_file( + submit_module.Path(first) / "model.pt" + ) + with patch.object( + submit_module, "DPA4_SOURCE_CHECKPOINT_SHA256", digest + ): + with self.assertRaisesRegex(RuntimeError, "never model.pt"): + submit_module._validate_lammps_runtime_contract( + None, + props, + "props", + [first, second], + ) + + def test_staged_dpa4_pt2_in_any_workdir_is_rejected(self): + props = { + "interaction": { + "type": "deepmd", + "model": "runtime.pt2", + "type_map": {"Ti": 0, "V": 1}, + }, + "properties": [{"type": "eos"}], + } + with tempfile.TemporaryDirectory() as first, tempfile.TemporaryDirectory() as second: + runtime_path = submit_module.Path(second) / "runtime.pt2" + runtime_path.write_bytes(b"exact-runtime-test") + digest = submit_module._sha256_file(runtime_path) + with patch.object( + submit_module, "DPA4_RUNTIME_MODEL_SHA256", digest + ): + with self.assertRaisesRegex(RuntimeError, "image-resident path"): + submit_module._validate_lammps_runtime_contract( + None, + props, + "props", + [first, second], + ) + def test_with_lammps_retry_env_handles_empty_existing_and_new_commands(self): class DummyConfig: lammps_header_retry_attempts = 4 @@ -208,6 +602,91 @@ def test_auto_fill_type_map_from_poscar(self): {"Al": 0, "Co": 1, "Cr": 2, "Fe": 3, "Mn": 4, "Ni": 5}, ) + def test_dpa4_runtime_contract_accepts_cli_expanded_type_map(self): + with tempfile.TemporaryDirectory() as tmp, patch.object( + submit_module, + "_load_dpa4_profile", + return_value=self._published_dpa4_profile(), + ): + structure_dir = os.path.join(tmp, "TiV") + os.makedirs(structure_dir) + with open( + os.path.join(structure_dir, "POSCAR"), "w", encoding="utf-8" + ) as stream: + stream.write( + "TiV\n1.0\n1 0 0\n0 1 0\n0 0 1\nTi V\n1 1\n" + "Direct\n0 0 0\n0.5 0.5 0.5\n" + ) + payload = { + "structures": ["TiV"], + "interaction": self._dpa4_interaction(), + "properties": [{"type": "eos"}], + } + param_path = os.path.join(tmp, "param.json") + with open(param_path, "w", encoding="utf-8") as stream: + json.dump(payload, stream) + self.assertTrue(auto_fill_type_map_from_poscar(payload, param_path)) + self.assertEqual(payload["interaction"]["type_map"], {"Ti": 0, "V": 1}) + self.assertEqual( + submit_module._validate_lammps_runtime_contract( + None, payload, "props", [tmp] + ), + submit_module.DPA4_RUNTIME_KIND, + ) + + def test_auto_fill_type_map_uses_all_structures_and_overwrites(self): + with tempfile.TemporaryDirectory() as tmp: + for directory, symbol in (("first", "Ti"), ("second", "V")): + structure_dir = os.path.join(tmp, directory) + os.makedirs(structure_dir) + with open( + os.path.join(structure_dir, "POSCAR"), + "w", + encoding="utf-8", + ) as stream: + stream.write( + f"{symbol}\n1.0\n1 0 0\n0 1 0\n0 0 1\n" + f"{symbol}\n1\nDirect\n0 0 0\n" + ) + + overwrite = { + "type": "deepmd", + "model": "overwrite.pb", + "type_map": "auto", + } + payload = { + "structures": ["first", "second"], + "interaction": { + "type": "deepmd", + "model": "base.pb", + "type_map": "auto", + }, + "properties": [ + { + "type": "eos", + "cal_setting": { + "overwrite_interaction": overwrite, + }, + } + ], + } + param_path = os.path.join(tmp, "param.json") + with open(param_path, "w", encoding="utf-8") as stream: + json.dump(payload, stream) + + self.assertTrue(auto_fill_type_map_from_poscar(payload, param_path)) + expected = {"Ti": 0, "V": 1} + self.assertEqual(payload["interaction"]["type_map"], expected) + self.assertEqual(overwrite["type_map"], expected) + with open(param_path, "r", encoding="utf-8") as stream: + persisted = json.load(stream) + self.assertEqual( + persisted["properties"][0]["cal_setting"][ + "overwrite_interaction" + ]["type_map"], + expected, + ) + def test_auto_fill_type_map_from_rss_conf_subdir(self): with tempfile.TemporaryDirectory() as tmp: structure_dir = os.path.join(tmp, "B2_HEA", "conf_001") diff --git a/tests/test_vasp.py b/tests/test_vasp.py index 365555fa..a7dc20fa 100644 --- a/tests/test_vasp.py +++ b/tests/test_vasp.py @@ -3,6 +3,7 @@ import os import shutil import sys +import tempfile import unittest import numpy as np @@ -13,7 +14,11 @@ __package__ = "tests" from apex.core.calculator.VASP import VASP -from apex.core.calculator.lib.vasp_utils import incar_upper +from apex.core.calculator.lib.vasp_utils import ( + incar_upper, + regulate_poscar, + sort_poscar, +) class TestVASP(unittest.TestCase): @@ -194,3 +199,38 @@ def test_backward_files(self): self.VASP.backward_files("gruneisen"), ["OUTCAR", "outlog", "CONTCAR", "OSZICAR", "XDATCAR", "vasprun.xml"], ) + + +class TestVASPPoscarUtilities(unittest.TestCase): + def test_regulate_and_sort_preserve_selective_dynamics(self): + contents = """TiV selective +1.0 +2 0 0 +0 2 0 +0 0 2 +V Ti V +1 1 1 +Selective dynamics +Direct +0.0 0.0 0.0 F F T V +0.5 0.5 0.5 F F T Ti +0.25 0.25 0.25 F F T V +""" + with tempfile.TemporaryDirectory() as tmp: + source = os.path.join(tmp, "POSCAR.in") + regulated = os.path.join(tmp, "POSCAR.regulated") + sorted_path = os.path.join(tmp, "POSCAR.sorted") + with open(source, "w") as fp: + fp.write(contents) + + regulate_poscar(source, regulated) + sort_poscar(regulated, sorted_path, ["Ti", "V"]) + + with open(sorted_path) as fp: + lines = fp.read().splitlines() + self.assertEqual(lines[5], "Ti V") + self.assertEqual(lines[6], "1 2") + self.assertEqual(lines[7], "Selective dynamics") + self.assertEqual(lines[8], "Direct") + self.assertEqual(lines[9].split()[-4:], ["F", "F", "T", "Ti"]) + self.assertTrue(all(line.split()[-4:-1] == ["F", "F", "T"] for line in lines[9:12]))