From f1c0978db9de581993ecf102ffdd6457a8f71479 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 18 Sep 2026 20:15:45 -0700 Subject: [PATCH 1/9] SOF-8063: formation energy beside the band structure for graphene C28N3 Structure NB: save and name the pristine 4x4 supercell as "graphene 4x4"; rename the defective material to "graphene 4x4 N3V pyridinic (C28N3)". Simulation NB, rewritten in place on the merged SOF-8044 skeleton: drop the RUN_PROFILE debug/production toggle, the Standata fallback and workflow.add_relaxation(); add the RELAX chain, three Total Energy jobs (C28N3 nspin 2, pristine graphene, solid N2 mp-154), the formation-energy arithmetic and one comparison block against Fujimoto & Saito (2011) Table I's 2.51 eV; keep the K-Gamma-M-K band structure off the same cell. LDA (pz) with GBRV ultrasoft, the only LDA family the platform publishes for C and N. Co-Authored-By: Claude Opus 5 (1M context) --- .../defect_point_substitution_graphene.ipynb | 12 +- ...int_substitution_graphene_SIMULATION.ipynb | 640 +++++++++++------- 2 files changed, 386 insertions(+), 266 deletions(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb index 4b0b8ecd5..f78e5f2b6 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb @@ -222,7 +222,10 @@ "id": "16", "metadata": {}, "source": [ - "## 4. Write resulting material to the folder" + "## 4. Write resulting materials to the folder\n", + "\n", + "Both materials are saved to `uploads` under the names the simulation notebook loads by exact name:\n", + "`graphene 4x4` is the reference cell, `graphene 4x4 N3V pyridinic (C28N3)` is the defective one." ] }, { @@ -235,9 +238,10 @@ "from mat3ra.notebooks_utils.io import download_content_to_file\n", "from mat3ra.notebooks_utils.material import set_materials\n", "\n", - "material_with_defect.name = \"N-doped Graphene\"\n", - "set_materials(material_with_defect)\n", - "download_content_to_file(material_with_defect.to_json(), \"N-doped_Graphene.json\")" + "supercell.name = \"graphene 4x4\"\n", + "material_with_defect.name = \"graphene 4x4 N3V pyridinic (C28N3)\"\n", + "set_materials([supercell, material_with_defect])\n", + "download_content_to_file(material_with_defect.to_json(), \"graphene_4x4_N3V_pyridinic.json\")" ] } ], diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index 4902ba679..53e00cb7c 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -5,31 +5,35 @@ "id": "0", "metadata": {}, "source": [ - "# Calculating Band Structure of N-doped Graphene.\n", + "# Formation Energy and Band Structure of the Pyridinic N₃-Vacancy Defect in Graphene\n", "\n", - "> **Yoshitaka Fujimoto and Susumu Saito**, \"Formation, stabilities, and electronic properties of nitrogen defects in graphene\", Physical Review B, 2011. [DOI: 10.1103/PhysRevB.84.245446](https://journals.aps.org/prb/abstract/10.1103/PhysRevB.84.245446).\n", + "> **Yoshitaka Fujimoto and Susumu Saito**, \"Formation, stabilities, and electronic properties of\n", + "> nitrogen defects in graphene\", Physical Review B 84, 245446 (2011).\n", + "> [DOI:10.1103/PhysRevB.84.245446](https://doi.org/10.1103/PhysRevB.84.245446)\n", "\n", - "\n", - "Calculate the band structure of material created in the specialized notebook using Quantum ESPRESSO application and the band structure workflow available in Standata.\n", + "Computes the formation energy and the band structure of the trimerized pyridine-type C₂₈N₃ defect\n", + "created in the [structure notebook](defect_point_substitution_graphene.ipynb), both off the same\n", + "cell, and compares the formation energy with the manuscript's Table I entry, 2.51 eV.\n", "\n", "

Usage

\n", "\n", - "1. Create the material in the [defect creation notebook](defect_point_substitution_graphene.ipynb) or place the material file in `../uploads` folder.\n", - "2. Set material and calculation parameters in cell 1.2. below (or use the default values). Choose between \"debug\" and \"production\" modes using `RUN_PROFILE` variable.\n", + "1. Create the materials in the [structure notebook](defect_point_substitution_graphene.ipynb), which saves `graphene 4x4` and `graphene 4x4 N3V pyridinic (C28N3)` to the `uploads` folder.\n", + "1. Set the material names and parameters in cells 1.2-1.4 (or use the defaults); `RELAX = True` uses the relaxed structure, relaxing once and saving it as ` relaxed` for reuse by this and other notebooks.\n", "1. Click \"Run\" > \"Run All\" to run all cells.\n", - "1. Wait for the job to complete.\n", - "1. Scroll down to view the result.\n", + "1. Wait for the jobs to complete.\n", + "1. Scroll down to view the results.\n", "\n", "## Summary\n", "\n", - "1. Set up the environment and parameters: install packages (JupyterLite only) and configure parameters for material, workflow, compute resources, and job.\n", + "1. Set up the environment and parameters: install packages (JupyterLite only) and configure the material names, model and compute parameters.\n", "1. Authenticate and initialize API client: authenticate via browser, initialize the client, then select account and project.\n", - "1. Create material: materials are read from the `../uploads` folder — place files there manually or run a [material creation notebook](defect_point_substitution_graphene.ipynb) first. If the material is not found by name, Standata is used as a fallback. The material is then saved to the platform.\n", - "1. Create workflow and set its parameters: select application, load total energy workflow from Standata, optionally add relaxation or adjust model/method parameters, and save the workflow to the platform.\n", - "1. Configure compute: get list of clusters and create compute configuration with selected cluster, queue, and number of processors.\n", - "1. Create the job with material and workflow configuration: assemble the job from material, workflow, project, and compute configuration.\n", - "1. Submit the job and monitor the status: submit the job and wait for completion.\n", - "1. Retrieve results: get and display the properties." + "1. Load the materials by name from the `uploads` folder, resolve the nitrogen reference, and save them to the platform.\n", + "1. Configure the shared DFT model and k-grid: one model and a per-material k-grid for every job below.\n", + "1. Configure compute: get the list of clusters and create a compute configuration.\n", + "1. Relax the defective cell if `RELAX`, reusing a saved relaxed structure when one already exists.\n", + "1. Create the missing Total Energy jobs for the defective, pristine and nitrogen cells and wait for them.\n", + "1. Configure the Band Structure workflow and run it on the defective cell.\n", + "1. Retrieve the band structure, assemble the formation energy and compare it with the manuscript." ] }, { @@ -50,7 +54,7 @@ "source": [ "from mat3ra.notebooks_utils.packages import install_packages\n", "\n", - "await install_packages(\"api_examples\")" + "await install_packages(\"made|specific_examples|api_examples\")" ] }, { @@ -58,14 +62,7 @@ "id": "3", "metadata": {}, "source": [ - "### 1.2. Set parameters and configurations for the workflow and job\n", - "\n", - "Use `RUN_PROFILE` in the next cell to switch between two predefined modes:\n", - "\n", - "- `\"debug\"`: quick validation run (small grids/path, no relaxation) to finish in a few minutes.\n", - "- `\"production\"`: paper-like settings (relaxation + denser sampling) for final-quality results; this can take much longer.\n", - "\n", - "For first-time use, start with `\"debug\"`. Once everything works, switch to `\"production\"` and rerun the notebook." + "### 1.2. Material names" ] }, { @@ -74,100 +71,114 @@ "id": "4", "metadata": {}, "outputs": [], + "source": [ + "# Names saved by defect_point_substitution_graphene.ipynb.\n", + "PRISTINE_NAME = \"graphene 4x4\"\n", + "DEFECTIVE_NAME = \"graphene 4x4 N3V pyridinic (C28N3)\"\n", + "# Standata's solid nitrogen -- the platform carries no free N2 molecule.\n", + "NITROGEN_NAME = \"N2, Nitrogen, FCC (P2_13) 3D (Bulk), mp-154\"" + ] + }, + { + "cell_type": "markdown", + "id": "5", + "metadata": {}, + "source": [ + "### 1.3. Parameters" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "6", + "metadata": {}, + "outputs": [], "source": [ "from datetime import datetime\n", "from mat3ra.ide.compute import QueueName\n", "\n", - "# 2. Auth and organization parameters\n", "ORGANIZATION_NAME = None # set to your organization name (full or partial); otherwise, your default one is used\n", - "\n", - "# 3. Material parameters\n", "FOLDER = \"./uploads\"\n", - "MATERIAL_NAME = \"N-doped Graphene\"\n", "\n", - "# 4. Workflow parameters\n", - "WORKFLOW_SEARCH_TERM = \"band_structure.json\"\n", - "RUN_PROFILE = \"debug\" # Change to \"production\" for paper-quality results\n", + "TOTAL_ENERGY_SEARCH_TERM = \"total_energy.json\"\n", + "RELAX_WORKFLOW_SEARCH_TERM = \"fixed_cell_relaxation.json\"\n", + "BAND_STRUCTURE_WORKFLOW_SEARCH_TERM = \"band_structure.json\"\n", "MY_WORKFLOW_NAME = \"Band Structure\"\n", "APPLICATION_NAME = \"espresso\"\n", + "TOTAL_ENERGY_SOURCE = \"my_account\"\n", "\n", - "# DFT model (same for both profiles)\n", - "MODEL_SUBTYPE = \"lda\"\n", + "# False: use the structures as given, fast. True: use the relaxed defective structure, running the\n", + "# relaxation once if it does not exist yet.\n", + "RELAX = False\n", "\n", - "# 5. Compute parameters\n", - "CLUSTER_NAME = \"101\" # specify full or partial name i.e. \"cluster-101\" to select\n", - "QUEUE_NAME = QueueName.D\n", - "PPN = 1\n", + "CLUSTER_NAME = \"001\" # specify full or partial name i.e. \"cluster-001\" to select\n", + "QUEUE_NAME = QueueName.OR\n", + "PPN = 40\n", + "TIME_LIMIT = \"12:00:00\" # covers the optional relaxation (~1 h)\n", "\n", - "# 6. Job parameters\n", "timestamp = datetime.now().strftime(\"%Y-%m-%d %H:%M\")\n", - "JOB_NAME = MY_WORKFLOW_NAME + \" \" + MATERIAL_NAME + f\" ({RUN_PROFILE}) \" + timestamp\n", - "POLL_INTERVAL = 60 # seconds\n" + "POLL_INTERVAL = 60 # seconds" ] }, { "cell_type": "markdown", - "id": "5", + "id": "7", "metadata": {}, "source": [ - "### 1.3. Set specific parameters" + "### 1.4. DFT model parameters" ] }, { "cell_type": "code", "execution_count": null, - "id": "6", + "id": "8", "metadata": {}, "outputs": [], "source": [ - "# Parameters for the Method Selected\n", - "PSEUDOPOTENTIAL_TYPE = \"nc\"\n", + "MODEL_SUBTYPE = \"lda\"\n", "FUNCTIONAL = \"pz\"\n", - "\n", - "# Energy cutoffs (check per the ONCV guidelines - https://www.pseudo-dojo.org/)\n", - "ECUTWFC = 50\n", - "# Density cutoff in relation to wavefunction cutoff\n", - "ECUTRHO = 4 * ECUTWFC\n", - "\n", - "# Parameters for the specific property\n", - "if RUN_PROFILE == \"debug\":\n", - " # Quick validation run (finishes in a few minutes)\n", - " ADD_RELAXATION = False\n", - " RELAXATION_KGRID = [1, 1, 1]\n", - " SCF_KGRID = [1, 1, 1]\n", - " KPATH = [\n", - " {\"point\": \"K\", \"steps\": 6},\n", - " {\"point\": \"Γ\", \"steps\": 6},\n", - " {\"point\": \"M\", \"steps\": 6},\n", - " {\"point\": \"K\", \"steps\": 1},\n", - " ]\n", - "elif RUN_PROFILE == \"production\":\n", - " # Paper settings (Fujimoto & Saito 2011): relaxation + dense sampling\n", - " ADD_RELAXATION = True\n", - " RELAXATION_KGRID = [6, 6, 1]\n", - " SCF_KGRID = [6, 6, 1]\n", - " KPATH = [\n", - " {\"point\": \"K\", \"steps\": 20},\n", - " {\"point\": \"Γ\", \"steps\": 20},\n", - " {\"point\": \"M\", \"steps\": 20},\n", - " {\"point\": \"K\", \"steps\": 1},\n", - " ]" + "PSEUDOPOTENTIAL_TYPE = \"us\" # GBRV ultrasoft, the only LDA family the platform publishes for C and N\n", + "ECUTWFC = 50 # Ry, Fujimoto & Saito Sec. II\n", + "ECUTRHO = 200 # Ry, GBRV's recommended charge-density cutoff\n", + "\n", + "KPOINT_DENSITY = 7 # points per Å⁻¹; gives 6 x 6 x 1 on the 4x4 cell, Fujimoto & Saito Sec. II\n", + "MODEL_TAG = f\"{FUNCTIONAL}-{PSEUDOPOTENTIAL_TYPE} {ECUTWFC}-{ECUTRHO}Ry k{KPOINT_DENSITY}\"\n", + "\n", + "SCF_UNIT = \"pw_scf\"\n", + "BANDS_UNIT = \"pw_bands\"\n", + "RELAX_UNIT = \"pw_relax\"\n", + "# C28N3 carries 127 valence electrons, so its ground state is a doublet.\n", + "SPIN_SETTINGS = {\n", + " DEFECTIVE_NAME: {\"nspin\": 2, \"tot_magnetization\": 1},\n", + " PRISTINE_NAME: {\"nspin\": 1},\n", + " NITROGEN_NAME: {\"nspin\": 1},\n", + "}\n", + "RELAXATION_SETTINGS = {\"forc_conv_thr\": 1.9e-3, \"nstep\": 100} # 0.05 eV/Å, Fujimoto & Saito Sec. II\n", + "# Names the relaxation job; the relaxed structure itself is found by content hash.\n", + "RELAX_TAG = f\"{MODEL_TAG} nspin2 f{RELAXATION_SETTINGS['forc_conv_thr']}\"\n", + "\n", + "KPATH_STEPS = 20\n", + "KPATH = [\n", + " {\"point\": \"K\", \"steps\": KPATH_STEPS},\n", + " {\"point\": \"Γ\", \"steps\": KPATH_STEPS},\n", + " {\"point\": \"M\", \"steps\": KPATH_STEPS},\n", + " {\"point\": \"K\", \"steps\": 1},\n", + "]" ] }, { "cell_type": "markdown", - "id": "7", + "id": "9", "metadata": {}, "source": [ "## 2. Authenticate and initialize API client\n", - "### 2.1. Authenticate\n", - "Authenticate in the browser and have credentials stored in environment variable \"OIDC_ACCESS_TOKEN\".\n" + "### 2.1. Authenticate" ] }, { "cell_type": "code", "execution_count": null, - "id": "8", + "id": "10", "metadata": {}, "outputs": [], "source": [ @@ -178,16 +189,16 @@ }, { "cell_type": "markdown", - "id": "9", + "id": "11", "metadata": {}, "source": [ - "### 2.2. Initialize API Client\n" + "### 2.2. Initialize API client" ] }, { "cell_type": "code", "execution_count": null, - "id": "10", + "id": "12", "metadata": {}, "outputs": [], "source": [ @@ -199,16 +210,16 @@ }, { "cell_type": "markdown", - "id": "11", + "id": "13", "metadata": {}, "source": [ - "### 2.3. Select account to work under" + "### 2.3. Select account" ] }, { "cell_type": "code", "execution_count": null, - "id": "12", + "id": "14", "metadata": {}, "outputs": [], "source": [ @@ -218,7 +229,7 @@ { "cell_type": "code", "execution_count": null, - "id": "13", + "id": "15", "metadata": {}, "outputs": [], "source": [ @@ -233,7 +244,7 @@ }, { "cell_type": "markdown", - "id": "14", + "id": "16", "metadata": {}, "source": [ "### 2.4. Select project" @@ -242,7 +253,7 @@ { "cell_type": "code", "execution_count": null, - "id": "15", + "id": "17", "metadata": {}, "outputs": [], "source": [ @@ -251,41 +262,13 @@ "print(f\"✅ Using project: {projects[0]['name']} ({project_id})\")" ] }, - { - "cell_type": "markdown", - "id": "16", - "metadata": {}, - "source": [ - "## 3. Create material\n", - "### 3.1. Load material from local file (or Standata)" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "id": "17", - "metadata": {}, - "outputs": [], - "source": [ - "from mat3ra.made.material import Material\n", - "from mat3ra.standata.materials import Materials\n", - "from mat3ra.notebooks_utils.ipython.entity.material.visualize import visualize_materials as visualize\n", - "from mat3ra.notebooks_utils.material import load_material_from_folder\n", - "\n", - "material = load_material_from_folder(FOLDER, MATERIAL_NAME) or Material.create(\n", - " Materials.get_by_name_first_match(MATERIAL_NAME))\n", - "\n", - "# set lattice for the correct display of Point Path for this material\n", - "material.lattice.type = \"HEX\"\n", - "visualize(material)" - ] - }, { "cell_type": "markdown", "id": "18", "metadata": {}, "source": [ - "### 3.2. Save material to the platform" + "## 3. Load the materials\n", + "### 3.1. Load from the uploads folder or the platform, and print provenance" ] }, { @@ -295,10 +278,15 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.core.entity.material.api import get_or_create_material\n", + "from collections import Counter\n", + "from mat3ra.notebooks_utils.core.entity.material.api import load_material\n", "\n", - "saved_material_response = get_or_create_material(client, material, ACCOUNT_ID)\n", - "saved_material = Material.create(saved_material_response)" + "materials_by_name = {name: load_material(client, FOLDER, name, ACCOUNT_ID) for name in (PRISTINE_NAME, DEFECTIVE_NAME)}\n", + "pristine, defective = materials_by_name[PRISTINE_NAME], materials_by_name[DEFECTIVE_NAME]\n", + "for name, material in materials_by_name.items():\n", + " composition = \"\".join(f\"{e}{n}\" for e, n in sorted(Counter(material.basis.elements.values).items()))\n", + " print(f\"{name}: {composition}, {material.basis.number_of_atoms} atoms, \"\n", + " f\"cell {material.lattice.a:.2f} x {material.lattice.b:.2f} x {material.lattice.c:.2f} Å\")" ] }, { @@ -306,8 +294,7 @@ "id": "20", "metadata": {}, "source": [ - "## 4. Create workflow and set its parameters\n", - "### 4.1. Get list of applications and select one" + "### 3.2. Resolve the nitrogen reference material" ] }, { @@ -317,12 +304,11 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.standata.applications import ApplicationStandata\n", - "from mat3ra.ade.application import Application\n", + "from mat3ra.made.material import Material\n", + "from mat3ra.standata.materials import Materials\n", "\n", - "app_config = ApplicationStandata.get_by_name_first_match(APPLICATION_NAME)\n", - "app = Application(**app_config)\n", - "print(f\"Using application: {app.name}\")" + "nitrogen = Material.create(Materials.get_by_name_first_match(NITROGEN_NAME))\n", + "print(f\"{nitrogen.name}: {nitrogen.basis.number_of_atoms} atoms, cell {nitrogen.lattice.a:.2f} Å\")" ] }, { @@ -330,7 +316,7 @@ "id": "22", "metadata": {}, "source": [ - "### 4.2. Create workflow from standard workflows and preview it" + "### 3.3. Save the materials to the platform" ] }, { @@ -340,15 +326,11 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.standata.workflows import WorkflowStandata\n", - "from mat3ra.wode.workflows import Workflow\n", - "from mat3ra.notebooks_utils.ipython.entity.workflow.visualize import visualize_workflow\n", - "\n", - "workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(WORKFLOW_SEARCH_TERM)\n", - "workflow = Workflow.create(workflow_config)\n", - "workflow.name = MY_WORKFLOW_NAME\n", + "from mat3ra.notebooks_utils.core.entity.material.api import get_or_create_material\n", "\n", - "visualize_workflow(workflow)" + "saved_pristine = get_or_create_material(client, pristine, ACCOUNT_ID)\n", + "saved_defective = get_or_create_material(client, defective, ACCOUNT_ID)\n", + "saved_nitrogen = get_or_create_material(client, nitrogen, ACCOUNT_ID)" ] }, { @@ -356,8 +338,8 @@ "id": "24", "metadata": {}, "source": [ - "### 4.3. Modify workflow (Optional)\n", - "#### 4.3.1. Add relaxation" + "## 4. Configure the shared DFT model and k-grid\n", + "### 4.1. DFT model" ] }, { @@ -367,8 +349,18 @@ "metadata": {}, "outputs": [], "source": [ - "if ADD_RELAXATION:\n", - " workflow.add_relaxation()\n" + "from mat3ra.standata.applications import ApplicationStandata\n", + "from mat3ra.standata.model_tree import ModelTreeStandata\n", + "from mat3ra.ade.application import Application\n", + "from mat3ra.mode import ModelFactory\n", + "\n", + "app_config = ApplicationStandata.get_by_name_first_match(APPLICATION_NAME)\n", + "app = Application(**app_config)\n", + "\n", + "model_config = ModelTreeStandata.get_model_by_parameters(type=\"dft\", subtype=MODEL_SUBTYPE, functional=FUNCTIONAL)\n", + "model_config[\"method\"] = {\"type\": \"pseudopotential\", \"subtype\": PSEUDOPOTENTIAL_TYPE}\n", + "model = ModelFactory.create(model_config)\n", + "print(f\"Using application: {app.name}, model: {MODEL_TAG}\")" ] }, { @@ -376,8 +368,7 @@ "id": "26", "metadata": {}, "source": [ - "#### 4.3.2. Modify model and method parameters (Optional)\n", - "Uncomment the code below and adjust selection of model parameters as needed." + "### 4.2. k-grid per material" ] }, { @@ -387,19 +378,16 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.standata.model_tree import ModelTreeStandata\n", - "from mat3ra.mode.model import Model\n", - "\n", - "model_config = ModelTreeStandata.get_model_by_parameters(\n", - " type=\"dft\", subtype=MODEL_SUBTYPE, functional=FUNCTIONAL\n", - ")\n", - "model_config[\"method\"] = {\"type\": \"pseudopotential\", \"subtype\": PSEUDOPOTENTIAL_TYPE}\n", - "model = Model.create(model_config)\n", + "from mat3ra.notebooks_utils.workflow import kgrid_from_density\n", "\n", - "for subworkflow in workflow.subworkflows:\n", - " subworkflow.model = model\n", - "\n", - "visualize_workflow(workflow)" + "material_objects = {PRISTINE_NAME: pristine, DEFECTIVE_NAME: defective, NITROGEN_NAME: nitrogen}\n", + "kgrid = {\n", + " PRISTINE_NAME: kgrid_from_density(pristine, KPOINT_DENSITY, periodic_dims=(0, 1)),\n", + " DEFECTIVE_NAME: kgrid_from_density(defective, KPOINT_DENSITY, periodic_dims=(0, 1)),\n", + " NITROGEN_NAME: kgrid_from_density(nitrogen, KPOINT_DENSITY),\n", + "}\n", + "for name, grid in kgrid.items():\n", + " print(f\"{name}: k-grid {grid}\")" ] }, { @@ -407,8 +395,8 @@ "id": "28", "metadata": {}, "source": [ - "#### 4.3.3. Modify important settings\n", - "Set k-grid, k-path and energy cutoffs, as well as other settings." + "## 5. Create the compute configuration\n", + "### 5.1. Select cluster" ] }, { @@ -418,48 +406,8 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.wode.context.providers import PlanewaveCutoffsContextProvider, PointsGridDataProvider, \\\n", - " PointsPathDataProvider\n", - "\n", - "bs_subworkflow = workflow.subworkflows[1 if ADD_RELAXATION else 0]\n", - "cutoffs_context = PlanewaveCutoffsContextProvider(wavefunction=ECUTWFC, density=ECUTRHO, isEdited=True).get_context_item_data()\n", - "\n", - "if RELAXATION_KGRID is not None and ADD_RELAXATION:\n", - " unit = workflow.subworkflows[0].get_unit_by_name(name_regex=\"relax\")\n", - " unit.add_context(PointsGridDataProvider(material=material, dimensions=RELAXATION_KGRID, isEdited=True).get_context_item_data())\n", - " workflow.subworkflows[0].set_unit(unit)\n", - "\n", - "if SCF_KGRID is not None:\n", - " unit = bs_subworkflow.get_unit_by_name(name=\"pw_scf\")\n", - " unit.add_context(PointsGridDataProvider(material=material, dimensions=SCF_KGRID, isEdited=True).get_context_item_data())\n", - " bs_subworkflow.set_unit(unit)\n", - "\n", - "if KPATH is not None:\n", - " unit = bs_subworkflow.get_unit_by_name(name=\"pw_bands\")\n", - " unit.add_context(PointsPathDataProvider(path=KPATH, isEdited=True).get_context_item_data())\n", - " bs_subworkflow.set_unit(unit)\n", - "\n", - "for unit_name in [\"pw_relax\", \"pw_vc-relax\", \"pw_scf\", \"pw_bands\"]:\n", - " for swf in workflow.subworkflows:\n", - " unit = swf.get_unit_by_name(name=unit_name)\n", - " if unit:\n", - " unit.add_context(cutoffs_context)\n", - " swf.set_unit(unit)\n", - "\n", - "# Below is the example of how to modify the input file content directly.\n", - "# bands_unit = bs_subworkflow.get_unit_by_name(name=\"bands\")\n", - "# bands_unit.input = [{\"name\": \"bands.in\", \"content\": (\n", - "# \"&BANDS\\n\"\n", - "# \" prefix = '__prefix__'\\n\"\n", - "# \" lsym = .false.\\n\"\n", - "# \" outdir = {% raw %}'{{ JOB_WORK_DIR }}/outdir'{% endraw %}\\n\"\n", - "# \" filband = {% raw %}'{{ JOB_WORK_DIR }}/bands.dat'{% endraw %}\\n\"\n", - "# \" no_overlap = .true.\\n\"\n", - "# \"/\"\n", - "# )}]\n", - "# bs_subworkflow.set_unit(bands_unit)\n", - "\n", - "visualize_workflow(workflow)" + "clusters = client.clusters.list()\n", + "print(f\"Available clusters: {[c['hostname'] for c in clusters]}\")" ] }, { @@ -467,7 +415,7 @@ "id": "30", "metadata": {}, "source": [ - "### 4.4. Save workflow to collection" + "### 5.2. Create the compute configuration for the jobs" ] }, { @@ -477,12 +425,17 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.utils.namespace import dict_to_namespace_recursive\n", - "from mat3ra.notebooks_utils.core.entity.workflow.api import get_or_create_workflow\n", + "from mat3ra.ide.compute import Compute\n", "\n", - "saved_workflow_response = get_or_create_workflow(client, workflow, ACCOUNT_ID)\n", - "saved_workflow = Workflow.create(saved_workflow_response)\n", - "print(f\"Workflow ID: {saved_workflow.id}\")" + "if CLUSTER_NAME:\n", + " cluster = next((c for c in clusters if CLUSTER_NAME in c[\"hostname\"]), None)\n", + " if cluster is None:\n", + " raise ValueError(f\"Cluster '{CLUSTER_NAME}' not found. Available: {[c['hostname'] for c in clusters]}\")\n", + "else:\n", + " cluster = clusters[0]\n", + "compute = Compute(cluster=cluster, queue=QUEUE_NAME, ppn=PPN, timeLimit=TIME_LIMIT)\n", + "print(f\"Using cluster: {compute.cluster.hostname}, queue: {QUEUE_NAME}, ppn: {PPN}, \"\n", + " f\"time limit: {TIME_LIMIT}\")" ] }, { @@ -490,8 +443,11 @@ "id": "32", "metadata": {}, "source": [ - "## 5. Create the compute configuration\n", - "### 5.1. Get list of clusters" + "## 6. Relax the defective cell (optional)\n", + "\n", + "Runs only if `RELAX`: finds any relaxed version of this structure already on the account,\n", + "regardless of who relaxed it or with what -- before spending any compute on a new one. The band\n", + "structure and the total energy are then both taken off that one geometry." ] }, { @@ -501,8 +457,58 @@ "metadata": {}, "outputs": [], "source": [ - "clusters = client.clusters.list()\n", - "print(f\"Available clusters: {[c['hostname'] for c in clusters]}\")" + "from mat3ra.standata.workflows import WorkflowStandata\n", + "from mat3ra.wode.workflows import Workflow\n", + "from mat3ra.notebooks_utils.workflow import apply_planewave_cutoffs, apply_scf_kgrid, patch_workflow_qe_input\n", + "from mat3ra.notebooks_utils.core.entity.job.api import find_job_for_material\n", + "from mat3ra.notebooks_utils.core.entity.material.api import find_relaxed_material, get_final_structure_for_job\n", + "from mat3ra.notebooks_utils.core.entity.property.api import get_properties_for_job\n", + "from mat3ra.notebooks_utils.job import create_job\n", + "from mat3ra.notebooks_utils.api.job import submit_jobs, wait_for_jobs_to_finish_async\n", + "\n", + "relaxed_defective = None\n", + "if RELAX:\n", + " relax_workflow_name = f\"Fixed-cell Relaxation {DEFECTIVE_NAME} {RELAX_TAG}\"\n", + " relaxed_defective = find_relaxed_material(client, defective, ACCOUNT_ID)\n", + " if relaxed_defective is not None:\n", + " print(f\"♻️ Relaxed defective material: {relaxed_defective.name} ({relaxed_defective.id})\")\n", + " else:\n", + " relax_workflow = Workflow.create(\n", + " WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(RELAX_WORKFLOW_SEARCH_TERM)\n", + " )\n", + " relax_workflow.name = relax_workflow_name\n", + " relax_workflow.subworkflows[0].model = model\n", + " apply_planewave_cutoffs(relax_workflow, ECUTWFC, ECUTRHO, unit_name=RELAX_UNIT)\n", + " apply_scf_kgrid(relax_workflow, kgrid[DEFECTIVE_NAME], material=defective, unit_name=RELAX_UNIT)\n", + " patch_workflow_qe_input(relax_workflow, {\"system\": SPIN_SETTINGS[DEFECTIVE_NAME]}, [RELAX_UNIT])\n", + " patch_workflow_qe_input(relax_workflow, {\"control\": RELAXATION_SETTINGS}, [RELAX_UNIT])\n", + "\n", + " relax_job = find_job_for_material(\n", + " client, saved_defective[\"_id\"], relax_workflow_name, ACCOUNT_ID,\n", + " statuses=(\"submitted\", \"queued\", \"active\", \"finished\"),\n", + " )\n", + " if relax_job is None:\n", + " relax_job = create_job(\n", + " api_client=client, materials=[saved_defective], workflow=relax_workflow,\n", + " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n", + " prefix=f\"{relax_workflow_name} {timestamp}\",\n", + " )\n", + " submit_jobs(client.jobs, [relax_job[\"_id\"]])\n", + " await wait_for_jobs_to_finish_async(client.jobs, [relax_job[\"_id\"]], poll_interval=POLL_INTERVAL)\n", + "\n", + " relaxed_material = get_final_structure_for_job(client, relax_job[\"_id\"])\n", + " client.materials.update(relaxed_material.id, {\"name\": f\"{DEFECTIVE_NAME} relaxed\"})\n", + " relaxed_defective = Material.create(client.materials.get(relaxed_material.id))\n", + " print(f\"✅ Relaxed defective material: {relaxed_defective.name} ({relaxed_defective.id})\")\n", + "\n", + " total_force = get_properties_for_job(client, relax_job[\"_id\"], \"total_force\")[0]\n", + " print(f\"Residual force after relaxation (norm over all atoms): \"\n", + " f\"{total_force['value']:.4f} {total_force['units']}\")\n", + "\n", + "defective_for_jobs = relaxed_defective if RELAX else defective\n", + "defective_for_jobs.lattice.type = \"HEX\" # the symbolic K-Γ-M-K path needs a named lattice type\n", + "material_objects[DEFECTIVE_NAME] = defective_for_jobs\n", + "saved_defective_for_jobs = get_or_create_material(client, defective_for_jobs, ACCOUNT_ID)" ] }, { @@ -510,7 +516,7 @@ "id": "34", "metadata": {}, "source": [ - "### 5.2. Create compute configuration for the job\n" + "## 7. Total Energy jobs for the defective, pristine and nitrogen cells" ] }, { @@ -520,122 +526,232 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.ide.compute import Compute\n", - "\n", - "# Select cluster: use specified name if provided, otherwise use first available\n", - "if CLUSTER_NAME:\n", - " cluster = next((c for c in clusters if CLUSTER_NAME in c[\"hostname\"]), None)\n", - "else:\n", - " cluster = clusters[0]\n", + "from mat3ra.notebooks_utils.core.entity.property.api import find_total_energy_for_material\n", "\n", - "compute = Compute(\n", - " cluster=cluster,\n", - " queue=QUEUE_NAME,\n", - " ppn=PPN\n", + "total_energy_workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(\n", + " TOTAL_ENERGY_SEARCH_TERM\n", ")\n", - "print(f\"Using cluster: {compute.cluster.hostname}, queue: {QUEUE_NAME}, ppn: {PPN}\")" + "saved_materials = {\n", + " DEFECTIVE_NAME: saved_defective_for_jobs,\n", + " PRISTINE_NAME: saved_pristine,\n", + " NITROGEN_NAME: saved_nitrogen,\n", + "}\n", + "total_energies = {}\n", + "total_energy_job_ids = {}\n", + "new_job_ids = []\n", + "for name, saved_material in saved_materials.items():\n", + " existing = find_total_energy_for_material(client, saved_material[\"_id\"], source=TOTAL_ENERGY_SOURCE)\n", + " if existing is not None:\n", + " total_energies[name] = existing[\"data\"][\"value\"]\n", + " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV\")\n", + " continue\n", + " workflow = Workflow.create(total_energy_workflow_config)\n", + " workflow.name = f\"Total Energy {name} {MODEL_TAG}\"\n", + " workflow.subworkflows[0].model = model\n", + " apply_planewave_cutoffs(workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n", + " apply_scf_kgrid(workflow, kgrid[name], material=material_objects[name])\n", + " patch_workflow_qe_input(workflow, {\"system\": SPIN_SETTINGS[name]}, [SCF_UNIT])\n", + " job = create_job(\n", + " api_client=client, materials=[saved_material], workflow=workflow, project_id=project_id,\n", + " owner_id=ACCOUNT_ID, compute=compute.to_dict(), prefix=f\"{workflow.name} {timestamp}\",\n", + " )\n", + " total_energy_job_ids[name] = job[\"_id\"]\n", + " new_job_ids.append(job[\"_id\"])\n", + " print(f\"✅ {name}: created Total Energy job {job['_id']}\")" ] }, { - "cell_type": "markdown", + "cell_type": "code", + "execution_count": null, "id": "36", "metadata": {}, + "outputs": [], + "source": [ + "if new_job_ids:\n", + " submit_jobs(client.jobs, new_job_ids)\n", + " print(f\"✅ Submitted {len(new_job_ids)} Total Energy job(s).\")\n", + " await wait_for_jobs_to_finish_async(client.jobs, new_job_ids, poll_interval=POLL_INTERVAL)\n", + "\n", + "for name, job_id in total_energy_job_ids.items():\n", + " total_energies[name] = get_properties_for_job(client, job_id, \"total_energy\")[0][\"value\"]\n", + " print(f\"{name}: {total_energies[name]:.4f} eV\")" + ] + }, + { + "cell_type": "markdown", + "id": "37", + "metadata": {}, "source": [ - "## 6. Create the job with material and workflow configuration\n", - "### 6.1. Create job" + "## 8. Band structure of the defective cell\n", + "### 8.1. Configure the workflow" ] }, { "cell_type": "code", "execution_count": null, - "id": "37", + "id": "38", "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.api.job import create_job\n", - "from mat3ra.notebooks_utils.ui import display_JSON\n", - "\n", - "print(f\"Material: {saved_material.id}\")\n", - "print(f\"Workflow: {saved_workflow.id}\")\n", - "print(f\"Project: {project_id}\")\n", - "\n", - "job_response = create_job(\n", - " api_client=client,\n", - " materials=[saved_material],\n", - " workflow=workflow,\n", - " project_id=project_id,\n", - " owner_id=ACCOUNT_ID,\n", - " prefix=JOB_NAME,\n", - " compute=compute.to_dict(),\n", + "from mat3ra.wode.context.providers import PointsPathDataProvider\n", + "from mat3ra.notebooks_utils.ipython.entity.workflow.visualize import visualize_workflow\n", + "\n", + "band_structure_workflow = Workflow.create(\n", + " WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(BAND_STRUCTURE_WORKFLOW_SEARCH_TERM)\n", ")\n", + "band_structure_workflow.name = f\"{MY_WORKFLOW_NAME} {DEFECTIVE_NAME} {MODEL_TAG}\"\n", + "band_structure_workflow.subworkflows[0].model = model\n", + "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n", + "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=BANDS_UNIT)\n", + "apply_scf_kgrid(band_structure_workflow, kgrid[DEFECTIVE_NAME], material=defective_for_jobs)\n", + "patch_workflow_qe_input(band_structure_workflow, {\"system\": SPIN_SETTINGS[DEFECTIVE_NAME]}, [SCF_UNIT, BANDS_UNIT])\n", + "\n", + "bands_subworkflow = band_structure_workflow.subworkflows[0]\n", + "bands_unit = bands_subworkflow.get_unit_by_name(name=BANDS_UNIT)\n", + "bands_unit.add_context(PointsPathDataProvider(path=KPATH, isEdited=True).get_context_item_data())\n", + "bands_subworkflow.set_unit(bands_unit)\n", "\n", - "job = dict_to_namespace_recursive(job_response)\n", - "job_id = job._id\n", - "print(\"✅ Job created successfully!\")\n", - "print(f\"Job ID: {job_id}\")\n", - "display_JSON(job_response)" + "visualize_workflow(band_structure_workflow)" ] }, { "cell_type": "markdown", - "id": "38", + "id": "39", "metadata": {}, "source": [ - "## 7. Submit the job and monitor the status" + "### 8.2. Create the job" ] }, { "cell_type": "code", "execution_count": null, - "id": "39", + "id": "40", + "metadata": {}, + "outputs": [], + "source": [ + "band_structure_job = create_job(\n", + " api_client=client, materials=[saved_defective_for_jobs], workflow=band_structure_workflow,\n", + " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n", + " prefix=f\"{band_structure_workflow.name} {timestamp}\",\n", + ")\n", + "band_structure_job_id = band_structure_job[\"_id\"]\n", + "print(f\"✅ Band Structure job created: {band_structure_job_id}\")" + ] + }, + { + "cell_type": "markdown", + "id": "41", + "metadata": {}, + "source": [ + "### 8.3. Submit and monitor the job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "42", "metadata": {}, "outputs": [], "source": [ - "client.jobs.submit(job_id)\n", - "print(f\"✅ Job {job_id} submitted successfully!\")" + "client.jobs.submit(band_structure_job_id)\n", + "print(f\"✅ Job {band_structure_job_id} submitted successfully!\")\n", + "await wait_for_jobs_to_finish_async(client.jobs, [band_structure_job_id], poll_interval=POLL_INTERVAL)" + ] + }, + { + "cell_type": "markdown", + "id": "43", + "metadata": {}, + "source": [ + "## 9. Retrieve the results\n", + "### 9.1. Band structure" ] }, { "cell_type": "code", "execution_count": null, - "id": "40", + "id": "44", "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.api.job import wait_for_jobs_to_finish_async\n", + "from mat3ra.notebooks_utils.ipython.entity.property.visualize import visualize_properties\n", "\n", - "await wait_for_jobs_to_finish_async(client.jobs, [job_id], poll_interval=POLL_INTERVAL)" + "band_structure_data = get_properties_for_job(client, band_structure_job_id, property_name=\"band_structure\")\n", + "visualize_properties(band_structure_data, title=\"Band Structure\",\n", + " extra_config={\"material\": defective_for_jobs.to_dict()})" ] }, { "cell_type": "markdown", - "id": "41", + "id": "45", "metadata": {}, "source": [ - "## 8. Retrieve results" + "### 9.2. Formation energy" ] }, { "cell_type": "code", "execution_count": null, - "id": "42", + "id": "46", "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.core.entity.property.api import get_properties_for_job\n", - "from mat3ra.notebooks_utils.ipython.entity.property.visualize import visualize_properties\n", + "defective_composition = Counter(defective_for_jobs.basis.elements.values)\n", + "mu_carbon = total_energies[PRISTINE_NAME] / pristine.basis.number_of_atoms\n", + "mu_nitrogen = total_energies[NITROGEN_NAME] / nitrogen.basis.number_of_atoms\n", + "e_formation = (\n", + " total_energies[DEFECTIVE_NAME]\n", + " - defective_composition[\"C\"] * mu_carbon\n", + " - defective_composition[\"N\"] * mu_nitrogen\n", + ")\n", "\n", - "property_data = get_properties_for_job(client, job_id, property_name=\"band_structure\")\n", - "visualize_properties(property_data, title=\"Band Structure\", extra_config={\"material\": material.to_dict()})" + "print(f\"μ_C = {mu_carbon:.4f} eV/atom ({pristine.basis.number_of_atoms} atoms)\")\n", + "print(f\"μ_N = {mu_nitrogen:.4f} eV/atom ({nitrogen.basis.number_of_atoms} atoms)\")\n", + "print(f\"E_f = {total_energies[DEFECTIVE_NAME]:.4f} eV - {defective_composition['C']} μ_C \"\n", + " f\"- {defective_composition['N']} μ_N = {e_formation:.3f} eV\")" + ] + }, + { + "cell_type": "markdown", + "id": "47", + "metadata": {}, + "source": [ + "### 9.3. Compare with Fujimoto & Saito (2011)" ] }, { "cell_type": "code", "execution_count": null, - "id": "43", + "id": "48", "metadata": {}, "outputs": [], - "source": [] + "source": [ + "FUJIMOTO_FORMATION_ENERGY = 2.51 # eV, Table I, trimerized pyridine-type C28N3\n", + "TOLERANCE_FRACTION = 0.15\n", + "\n", + "deviation = e_formation - FUJIMOTO_FORMATION_ENERGY\n", + "verdict = \"yes\" if abs(deviation) <= TOLERANCE_FRACTION * FUJIMOTO_FORMATION_ENERGY else \"no\"\n", + "regime = \"relaxed\" if RELAX else \"unrelaxed\"\n", + "print(f\"E_f (this notebook): {e_formation:.3f} eV\")\n", + "print(f\"E_f (Fujimoto & Saito 2011, Table I): {FUJIMOTO_FORMATION_ENERGY:.3f} eV\")\n", + "print(f\"Deviation: {deviation:+.3f} eV ({100 * deviation / FUJIMOTO_FORMATION_ENERGY:+.1f} %, \"\n", + " f\"tolerance {100 * TOLERANCE_FRACTION:.0f} %)\")\n", + "print(\"Offsets: μ_N comes from solid N2 (mp-154) rather than the free molecule the manuscript \"\n", + " \"uses, and GBRV ultrasoft LDA stands in for its Troullier-Martins norm-conserving set.\")\n", + "print(f\"Reproduces Fujimoto & Saito (2011): {verdict} ({regime})\")" + ] + }, + { + "cell_type": "markdown", + "id": "49", + "metadata": {}, + "source": [ + "## References\n", + "\n", + "[1] Y. Fujimoto and S. Saito, \"Formation, stabilities, and electronic properties of nitrogen defects in graphene\", Phys. Rev. B 84, 245446 (2011). https://doi.org/10.1103/PhysRevB.84.245446\n", + "\n", + "[2] K. F. Garrity, J. W. Bennett, K. M. Rabe and D. Vanderbilt, \"Pseudopotentials for high-throughput DFT calculations\", Comput. Mater. Sci. 81, 446 (2014). https://doi.org/10.1016/j.commatsci.2013.08.053" + ] } ], "metadata": { From 0c416d672677dd9e4239aec18bc8c1caafbff01b Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 18 Sep 2026 20:19:06 -0700 Subject: [PATCH 2/9] SOF-8063: NOTE lines saying what a result close to the manuscript needs Follows api-examples #370 (SOF-8045): a NOTE above RELAX = False, a NOTE above the DFT parameters naming what the manuscript's own value needed, and the regime printed before the numbers in the comparison cell. Here relaxation also changes the band structure -- the pyridinic C-N bond contracts 1.41 A -> 1.33 A and the acceptor-like states near E_F belong to the relaxed geometry -- so the RELAX section markdown says so and the NOTE names both properties. RELAX_TAG drops its nspin marker to match #370. Co-Authored-By: Claude Opus 5 (1M context) --- ...t_point_substitution_graphene_SIMULATION.ipynb | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index 53e00cb7c..3594d4ee3 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -107,8 +107,8 @@ "APPLICATION_NAME = \"espresso\"\n", "TOTAL_ENERGY_SOURCE = \"my_account\"\n", "\n", - "# False: use the structures as given, fast. True: use the relaxed defective structure, running the\n", - "# relaxation once if it does not exist yet.\n", + "# NOTE: set to True for results close to the manuscript (relaxes C28N3 once, ~1 h); the formation\n", + "# energy and the band structure both depend on it.\n", "RELAX = False\n", "\n", "CLUSTER_NAME = \"001\" # specify full or partial name i.e. \"cluster-001\" to select\n", @@ -135,6 +135,8 @@ "metadata": {}, "outputs": [], "source": [ + "# NOTE: the manuscript's value needs its own settings — Troullier-Martins norm-conserving LDA;\n", + "# its 4x4 cell, 50 Ry cutoff and 6x6x1 k-grid are matched below, its pseudopotentials are not.\n", "MODEL_SUBTYPE = \"lda\"\n", "FUNCTIONAL = \"pz\"\n", "PSEUDOPOTENTIAL_TYPE = \"us\" # GBRV ultrasoft, the only LDA family the platform publishes for C and N\n", @@ -155,7 +157,7 @@ "}\n", "RELAXATION_SETTINGS = {\"forc_conv_thr\": 1.9e-3, \"nstep\": 100} # 0.05 eV/Å, Fujimoto & Saito Sec. II\n", "# Names the relaxation job; the relaxed structure itself is found by content hash.\n", - "RELAX_TAG = f\"{MODEL_TAG} nspin2 f{RELAXATION_SETTINGS['forc_conv_thr']}\"\n", + "RELAX_TAG = f\"{MODEL_TAG} f{RELAXATION_SETTINGS['forc_conv_thr']}\"\n", "\n", "KPATH_STEPS = 20\n", "KPATH = [\n", @@ -447,7 +449,9 @@ "\n", "Runs only if `RELAX`: finds any relaxed version of this structure already on the account,\n", "regardless of who relaxed it or with what -- before spending any compute on a new one. The band\n", - "structure and the total energy are then both taken off that one geometry." + "structure and the total energy are then both taken off that one geometry. In Fujimoto & Saito the\n", + "pyridinic C-N bond contracts from 1.41 Å to 1.33 Å and the acceptor-like states near E_F belong to\n", + "the relaxed geometry, so an unrelaxed run is a different system rather than a rougher answer." ] }, { @@ -729,9 +733,10 @@ "FUJIMOTO_FORMATION_ENERGY = 2.51 # eV, Table I, trimerized pyridine-type C28N3\n", "TOLERANCE_FRACTION = 0.15\n", "\n", + "regime = \"relaxed defect\" if RELAX else \"unrelaxed SCF\"\n", "deviation = e_formation - FUJIMOTO_FORMATION_ENERGY\n", "verdict = \"yes\" if abs(deviation) <= TOLERANCE_FRACTION * FUJIMOTO_FORMATION_ENERGY else \"no\"\n", - "regime = \"relaxed\" if RELAX else \"unrelaxed\"\n", + "print(f\"Regime: {regime}\")\n", "print(f\"E_f (this notebook): {e_formation:.3f} eV\")\n", "print(f\"E_f (Fujimoto & Saito 2011, Table I): {FUJIMOTO_FORMATION_ENERGY:.3f} eV\")\n", "print(f\"Deviation: {deviation:+.3f} eV ({100 * deviation / FUJIMOTO_FORMATION_ENERGY:+.1f} %, \"\n", From c90cf3c3ba1d1526f8d41601c36b1a4e51f8ef94 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 18 Sep 2026 21:36:08 -0700 Subject: [PATCH 3/9] SOF-8063: prefer an exact name in load_material_from_folder The lookup took the first filename containing the search string in sorted order, so "graphene 4x4" resolved to "graphene 4x4 N3V pyridinic (C28N3)" -- ' ' sorts before '.' -- and load_material rejected it and raised. Rank every material in the folder instead: exact material name, exact filename, the same two ignoring case, then the previous substring behaviour, filename before material name. Co-Authored-By: Claude Opus 5 (1M context) --- .../core/entity/material/io.py | 51 ++++++++++++------ .../py/unit/core/entity/test_material_api.py | 1 - tests/py/unit/core/entity/test_material_io.py | 52 +++++++++++++++++++ 3 files changed, 86 insertions(+), 18 deletions(-) diff --git a/src/py/mat3ra/notebooks_utils/core/entity/material/io.py b/src/py/mat3ra/notebooks_utils/core/entity/material/io.py index 0a4cbe3f1..ef7136aa2 100644 --- a/src/py/mat3ra/notebooks_utils/core/entity/material/io.py +++ b/src/py/mat3ra/notebooks_utils/core/entity/material/io.py @@ -122,35 +122,52 @@ def load_materials_from_folder(folder_path: Optional[str] = None, verbose: bool return materials +def _get_name_match_rank(requested_name: str, material_name: str, filename_without_extension: str) -> Optional[int]: + """ + Rank how well one folder entry matches a requested name: lower is better, None is no match. + + An exact match wins over a substring match, and a case-sensitive match over a case-insensitive + one. Among exact matches the material name wins; among substring matches the filename wins. + """ + requested_name_lower = requested_name.lower() + ranked_candidates = [ + material_name == requested_name, + filename_without_extension == requested_name, + material_name.lower() == requested_name_lower, + filename_without_extension.lower() == requested_name_lower, + requested_name_lower in filename_without_extension.lower(), + requested_name_lower in material_name.lower(), + ] + return next((rank for rank, matches in enumerate(ranked_candidates) if matches), None) + + def load_material_from_folder(folder_path: str, name: str, verbose: bool = True) -> Optional[Any]: """ - Load a single material from the specified folder by matching a substring of the name or filename. + Load a single material from the specified folder by matching the name or filename. Args: folder_path (str): The path to the folder containing material files. - name (str): The substring to match against material names or filenames (case-insensitive). + name (str): The name to match against material names or filenames. An exact match is + preferred; otherwise a case-insensitive substring match is accepted. verbose (bool): Whether to log verbose messages. Returns: - Optional[Material]: The first Material object that matches, or None if not found. + Optional[Material]: The best matching Material object, or None if not found. """ - name_lower = name.lower() + best_rank = None resulting_material = None for filename in sorted(os.listdir(folder_path)): - if filename.endswith(".json") and name_lower in os.path.splitext(filename)[0].lower(): - with open(os.path.join(folder_path, filename), "r") as file: - data = json.load(file) - if MaterialWithBuildMetadata.is_valid(data): - resulting_material = MaterialWithBuildMetadata.create(data) - break - - if not resulting_material: - materials = load_materials_from_folder(folder_path, verbose=verbose) - for material in materials: - if name_lower in material.name.lower(): - resulting_material = material - break + if not filename.endswith(".json"): + continue + with open(os.path.join(folder_path, filename), "r") as file: + data = json.load(file) + if not MaterialWithBuildMetadata.is_valid(data): + continue + material = MaterialWithBuildMetadata.create(data) + rank = _get_name_match_rank(name, material.name, os.path.splitext(filename)[0]) + if rank is not None and (best_rank is None or rank < best_rank): + best_rank, resulting_material = rank, material if resulting_material: log(f"Found: '{resulting_material.name}'", SeverityLevelEnum.INFO, force_verbose=verbose) diff --git a/tests/py/unit/core/entity/test_material_api.py b/tests/py/unit/core/entity/test_material_api.py index 44e7df45d..1d81528ca 100644 --- a/tests/py/unit/core/entity/test_material_api.py +++ b/tests/py/unit/core/entity/test_material_api.py @@ -382,7 +382,6 @@ def test_load_material_raises_when_neither_has_it(tmp_path): def test_load_material_falls_through_a_folder_near_miss(tmp_path): - (tmp_path / "silicon.json").write_text(json.dumps(SILICON_NAMED)) (tmp_path / "silicon relaxed.json").write_text(json.dumps({**SILICON_NAMED, "name": "Silicon relaxed"})) client = MagicMock() client.materials.list.return_value = [SILICON_NAMED] diff --git a/tests/py/unit/core/entity/test_material_io.py b/tests/py/unit/core/entity/test_material_io.py index 5771ca68d..51ebe827a 100644 --- a/tests/py/unit/core/entity/test_material_io.py +++ b/tests/py/unit/core/entity/test_material_io.py @@ -1,10 +1,33 @@ import json +import pytest from mat3ra.notebooks_utils.core.entity.material.io import load_material_from_folder, load_materials_from_folder from mat3ra.standata.materials import Materials RADII = {"Si": 1.11, "C": 0.76} +GRAPHENE_STANDATA_NAME = "C, Graphene, HEX (P6/mmm) 2D (Monolayer), 2dm-3993" +PRISTINE_NAME = "graphene 4x4" +DEFECTIVE_NAME = "graphene 4x4 N3V pyridinic (C28N3)" +RELAXED_NAME = "graphene 4x4 relaxed" + +# (filename without extension, material name) pairs; the relaxed material is stored under a +# filename that is not its name. +GRAPHENE_FOLDER = [ + (GRAPHENE_STANDATA_NAME.replace("/", "-"), GRAPHENE_STANDATA_NAME), + (PRISTINE_NAME, PRISTINE_NAME), + (DEFECTIVE_NAME, DEFECTIVE_NAME), + ("pristine graphene", RELAXED_NAME), +] +RENAMED_FILE_FOLDER = [ + (PRISTINE_NAME, RELAXED_NAME), + (DEFECTIVE_NAME, DEFECTIVE_NAME), +] +RECASED_NAME_FOLDER = [ + (PRISTINE_NAME, PRISTINE_NAME.upper()), + ("n3v defect", PRISTINE_NAME), +] + def _uploads_folder(tmp_path): (tmp_path / "silicon.json").write_text(json.dumps(Materials.get_by_name_first_match("Silicon"))) @@ -18,3 +41,32 @@ def test_data_file_beside_a_material_is_skipped(tmp_path): def test_lookup_by_name_works_alongside_a_data_file(tmp_path): assert load_material_from_folder(_uploads_folder(tmp_path), "Silicon", verbose=False) is not None + + +def _material_folder(tmp_path, files): + for filename, name in files: + graphene = {**Materials.get_by_name_first_match("Graphene"), "name": name} + (tmp_path / f"{filename}.json").write_text(json.dumps(graphene)) + return str(tmp_path) + + +@pytest.mark.parametrize( + ("files", "requested_name", "expected_name"), + [ + (GRAPHENE_FOLDER, PRISTINE_NAME, PRISTINE_NAME), + (GRAPHENE_FOLDER, DEFECTIVE_NAME, DEFECTIVE_NAME), + (GRAPHENE_FOLDER, "GRAPHENE 4X4", PRISTINE_NAME), + (GRAPHENE_FOLDER, "N3V", DEFECTIVE_NAME), + (GRAPHENE_FOLDER, "pristine", RELAXED_NAME), + (GRAPHENE_FOLDER, "relaxed", RELAXED_NAME), + (GRAPHENE_FOLDER, "Graphene", GRAPHENE_STANDATA_NAME), + (GRAPHENE_FOLDER, "Germanium", None), + (RENAMED_FILE_FOLDER, PRISTINE_NAME, RELAXED_NAME), + (RENAMED_FILE_FOLDER, "GRAPHENE 4X4", RELAXED_NAME), + (RECASED_NAME_FOLDER, PRISTINE_NAME, PRISTINE_NAME), + ], +) +def test_load_material_from_folder(tmp_path, files, requested_name, expected_name): + material = load_material_from_folder(_material_folder(tmp_path, files), requested_name, verbose=False) + + assert (material.name if material else None) == expected_name From a704c5587d967a236b255ab0789b0b2ef8a42f23 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 18 Sep 2026 21:45:49 -0700 Subject: [PATCH 4/9] SOF-8063: scope the Total Energy reuse to this notebook's own workflow The reuse check took any total_energy property carried by the material, so the live run mixed energies from unrelated models (-5195.0 eV for C28N3 beside -313.6 eV for C32) and printed E_f = -4098.5 eV. It now looks for a finished job of that material whose workflow name is the one this notebook creates, and reads total_energy off that job; a new job is created when there is none. PPN 40 -> 16: queue OR on cluster-001 allows at most 16 cores per node, and the band structure job errored 1.5 s after submission with compute.errors domain "celim", every unit still idle. Co-Authored-By: Claude Opus 5 (1M context) --- ...ct_point_substitution_graphene_SIMULATION.ipynb | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index 3594d4ee3..f6109b3d6 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -105,7 +105,6 @@ "BAND_STRUCTURE_WORKFLOW_SEARCH_TERM = \"band_structure.json\"\n", "MY_WORKFLOW_NAME = \"Band Structure\"\n", "APPLICATION_NAME = \"espresso\"\n", - "TOTAL_ENERGY_SOURCE = \"my_account\"\n", "\n", "# NOTE: set to True for results close to the manuscript (relaxes C28N3 once, ~1 h); the formation\n", "# energy and the band structure both depend on it.\n", @@ -113,7 +112,7 @@ "\n", "CLUSTER_NAME = \"001\" # specify full or partial name i.e. \"cluster-001\" to select\n", "QUEUE_NAME = QueueName.OR\n", - "PPN = 40\n", + "PPN = 16 # queue OR on cluster-001 allows at most 16 cores per node\n", "TIME_LIMIT = \"12:00:00\" # covers the optional relaxation (~1 h)\n", "\n", "timestamp = datetime.now().strftime(\"%Y-%m-%d %H:%M\")\n", @@ -530,8 +529,6 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.core.entity.property.api import find_total_energy_for_material\n", - "\n", "total_energy_workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(\n", " TOTAL_ENERGY_SEARCH_TERM\n", ")\n", @@ -544,13 +541,14 @@ "total_energy_job_ids = {}\n", "new_job_ids = []\n", "for name, saved_material in saved_materials.items():\n", - " existing = find_total_energy_for_material(client, saved_material[\"_id\"], source=TOTAL_ENERGY_SOURCE)\n", - " if existing is not None:\n", - " total_energies[name] = existing[\"data\"][\"value\"]\n", + " workflow_name = f\"Total Energy {name} {MODEL_TAG}\"\n", + " existing_job = find_job_for_material(client, saved_material[\"_id\"], workflow_name, ACCOUNT_ID)\n", + " if existing_job is not None:\n", + " total_energies[name] = get_properties_for_job(client, existing_job[\"_id\"], \"total_energy\")[0][\"value\"]\n", " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV\")\n", " continue\n", " workflow = Workflow.create(total_energy_workflow_config)\n", - " workflow.name = f\"Total Energy {name} {MODEL_TAG}\"\n", + " workflow.name = workflow_name\n", " workflow.subworkflows[0].model = model\n", " apply_planewave_cutoffs(workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n", " apply_scf_kgrid(workflow, kgrid[name], material=material_objects[name])\n", From 34ef46a4dc485d6c5e3d2e09f116d259b079839e Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 18 Sep 2026 22:32:14 -0700 Subject: [PATCH 5/9] SOF-8063: reuse an existing Band Structure job instead of submitting a second The Band Structure job was created and submitted on every run, so a notebook that died after its Total Energy jobs finished would put a second job on the cluster beside the one still running. It now looks for a job of this material and this workflow name in any live or finished state, and creates one only when there is none; the wait runs either way, so a job found while still active is waited on before its band structure is fetched. Co-Authored-By: Claude Opus 5 (1M context) --- ...int_substitution_graphene_SIMULATION.ipynb | 29 +++++++++++++------ 1 file changed, 20 insertions(+), 9 deletions(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index f6109b3d6..1873a981b 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -598,10 +598,11 @@ "from mat3ra.wode.context.providers import PointsPathDataProvider\n", "from mat3ra.notebooks_utils.ipython.entity.workflow.visualize import visualize_workflow\n", "\n", + "band_structure_workflow_name = f\"{MY_WORKFLOW_NAME} {DEFECTIVE_NAME} {MODEL_TAG}\"\n", "band_structure_workflow = Workflow.create(\n", " WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(BAND_STRUCTURE_WORKFLOW_SEARCH_TERM)\n", ")\n", - "band_structure_workflow.name = f\"{MY_WORKFLOW_NAME} {DEFECTIVE_NAME} {MODEL_TAG}\"\n", + "band_structure_workflow.name = band_structure_workflow_name\n", "band_structure_workflow.subworkflows[0].model = model\n", "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n", "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=BANDS_UNIT)\n", @@ -631,13 +632,22 @@ "metadata": {}, "outputs": [], "source": [ - "band_structure_job = create_job(\n", - " api_client=client, materials=[saved_defective_for_jobs], workflow=band_structure_workflow,\n", - " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n", - " prefix=f\"{band_structure_workflow.name} {timestamp}\",\n", + "band_structure_job = find_job_for_material(\n", + " client, saved_defective_for_jobs[\"_id\"], band_structure_workflow_name, ACCOUNT_ID,\n", + " statuses=(\"submitted\", \"queued\", \"active\", \"finished\"),\n", ")\n", - "band_structure_job_id = band_structure_job[\"_id\"]\n", - "print(f\"✅ Band Structure job created: {band_structure_job_id}\")" + "if band_structure_job is None:\n", + " band_structure_job = create_job(\n", + " api_client=client, materials=[saved_defective_for_jobs], workflow=band_structure_workflow,\n", + " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n", + " prefix=f\"{band_structure_workflow.name} {timestamp}\",\n", + " )\n", + " new_band_structure_job_id = band_structure_job[\"_id\"]\n", + " print(f\"✅ Band Structure job created: {new_band_structure_job_id}\")\n", + "else:\n", + " new_band_structure_job_id = None\n", + " print(f\"♻️ Reusing Band Structure job {band_structure_job['_id']}\")\n", + "band_structure_job_id = band_structure_job[\"_id\"]" ] }, { @@ -655,8 +665,9 @@ "metadata": {}, "outputs": [], "source": [ - "client.jobs.submit(band_structure_job_id)\n", - "print(f\"✅ Job {band_structure_job_id} submitted successfully!\")\n", + "if new_band_structure_job_id:\n", + " client.jobs.submit(new_band_structure_job_id)\n", + " print(f\"✅ Job {new_band_structure_job_id} submitted successfully!\")\n", "await wait_for_jobs_to_finish_async(client.jobs, [band_structure_job_id], poll_interval=POLL_INTERVAL)" ] }, From 374f7502f02de4f882626f75f96980ae4707cbc9 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 2 Oct 2026 13:05:37 -0700 Subject: [PATCH 6/9] =?UTF-8?q?SOF-8063:=20TIME=5FLIMIT=204=20h=20?= =?UTF-8?q?=E2=80=94=2012=20h=20x=2016=20ppn=20is=20refused=20at=20submiss?= =?UTF-8?q?ion?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5 (1M context) --- .../defect_point_substitution_graphene_SIMULATION.ipynb | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index 1873a981b..9ebaa4cff 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -113,7 +113,7 @@ "CLUSTER_NAME = \"001\" # specify full or partial name i.e. \"cluster-001\" to select\n", "QUEUE_NAME = QueueName.OR\n", "PPN = 16 # queue OR on cluster-001 allows at most 16 cores per node\n", - "TIME_LIMIT = \"12:00:00\" # covers the optional relaxation (~1 h)\n", + "TIME_LIMIT = \"04:00:00\" # covers the optional relaxation (~1 h); walltime x ppn is reserved up front\n", "\n", "timestamp = datetime.now().strftime(\"%Y-%m-%d %H:%M\")\n", "POLL_INTERVAL = 60 # seconds" From bf9c3a71b918931964982530a2cbf5aff83ea20d Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 2 Oct 2026 14:10:05 -0700 Subject: [PATCH 7/9] =?UTF-8?q?SOF-8063:=20RUN=5FTIER=20fast|production=20?= =?UTF-8?q?replaces=20RELAX=20=E2=80=94=20one=20tier=20sets=20k-grid=20and?= =?UTF-8?q?=20relaxation?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fast (default) runs 3x3x1 with no relaxation, in minutes, and its verdict line reads "no (fast tier)" whatever the number; production is the manuscript's 6x6x1 and the relaxation to 0.05 eV/A, i.e. the old RELAX = True. Both come off one TIER_SETTINGS entry in the parameter cell, so the k-density no longer has a second home in 1.4. Co-Authored-By: Claude Opus 5 (1M context) --- ...int_substitution_graphene_SIMULATION.ipynb | 33 +++++++++++-------- 1 file changed, 20 insertions(+), 13 deletions(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index 9ebaa4cff..b4e9f5810 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -18,7 +18,7 @@ "

Usage

\n", "\n", "1. Create the materials in the [structure notebook](defect_point_substitution_graphene.ipynb), which saves `graphene 4x4` and `graphene 4x4 N3V pyridinic (C28N3)` to the `uploads` folder.\n", - "1. Set the material names and parameters in cells 1.2-1.4 (or use the defaults); `RELAX = True` uses the relaxed structure, relaxing once and saving it as ` relaxed` for reuse by this and other notebooks.\n", + "1. Set the material names and parameters in cells 1.2-1.4 (or use the defaults); `RUN_TIER = \"production\"` runs the manuscript's k-grid and relaxes C₂₈N₃ once, saving it as ` relaxed` for reuse by this and other notebooks.\n", "1. Click \"Run\" > \"Run All\" to run all cells.\n", "1. Wait for the jobs to complete.\n", "1. Scroll down to view the results.\n", @@ -30,7 +30,7 @@ "1. Load the materials by name from the `uploads` folder, resolve the nitrogen reference, and save them to the platform.\n", "1. Configure the shared DFT model and k-grid: one model and a per-material k-grid for every job below.\n", "1. Configure compute: get the list of clusters and create a compute configuration.\n", - "1. Relax the defective cell if `RELAX`, reusing a saved relaxed structure when one already exists.\n", + "1. Relax the defective cell on the `production` tier, reusing a saved relaxed structure when one already exists.\n", "1. Create the missing Total Energy jobs for the defective, pristine and nitrogen cells and wait for them.\n", "1. Configure the Band Structure workflow and run it on the defective cell.\n", "1. Retrieve the band structure, assemble the formation energy and compare it with the manuscript." @@ -84,7 +84,9 @@ "id": "5", "metadata": {}, "source": [ - "### 1.3. Parameters" + "### 1.3. Parameters\n", + "\n", + "`RUN_TIER = \"fast\"` runs end to end in minutes; `\"production\"` takes about 1.5 hours from cold, which outlives the browser session — every job here is looked up by name before it is created, so signing in again and re-running attaches to the jobs already on the cluster instead of resubmitting them." ] }, { @@ -106,9 +108,15 @@ "MY_WORKFLOW_NAME = \"Band Structure\"\n", "APPLICATION_NAME = \"espresso\"\n", "\n", - "# NOTE: set to True for results close to the manuscript (relaxes C28N3 once, ~1 h); the formation\n", - "# energy and the band structure both depend on it.\n", - "RELAX = False\n", + "RUN_TIER = \"fast\" # minutes; coarse k-grid, no relaxation — will NOT match the manuscript\n", + "# RUN_TIER = \"production\" # ~1.5 h; 6x6x1 and relaxation to 0.05 eV/Å — reproduces Table I\n", + "TIER_SETTINGS = {\n", + " \"fast\": {\"kpoint_density\": 3.5, \"relax\": False},\n", + " \"production\": {\"kpoint_density\": 7, \"relax\": True},\n", + "}\n", + "# Points per Å⁻¹: 3.5 gives 3 x 3 x 1 on the 4x4 cell, 7 gives the manuscript's 6 x 6 x 1.\n", + "KPOINT_DENSITY = TIER_SETTINGS[RUN_TIER][\"kpoint_density\"]\n", + "RELAX = TIER_SETTINGS[RUN_TIER][\"relax\"]\n", "\n", "CLUSTER_NAME = \"001\" # specify full or partial name i.e. \"cluster-001\" to select\n", "QUEUE_NAME = QueueName.OR\n", @@ -135,14 +143,14 @@ "outputs": [], "source": [ "# NOTE: the manuscript's value needs its own settings — Troullier-Martins norm-conserving LDA;\n", - "# its 4x4 cell, 50 Ry cutoff and 6x6x1 k-grid are matched below, its pseudopotentials are not.\n", + "# its 4x4 cell and 50 Ry cutoff are matched below, its 6x6x1 k-grid on the production tier,\n", + "# its pseudopotentials not at all.\n", "MODEL_SUBTYPE = \"lda\"\n", "FUNCTIONAL = \"pz\"\n", "PSEUDOPOTENTIAL_TYPE = \"us\" # GBRV ultrasoft, the only LDA family the platform publishes for C and N\n", "ECUTWFC = 50 # Ry, Fujimoto & Saito Sec. II\n", "ECUTRHO = 200 # Ry, GBRV's recommended charge-density cutoff\n", "\n", - "KPOINT_DENSITY = 7 # points per Å⁻¹; gives 6 x 6 x 1 on the 4x4 cell, Fujimoto & Saito Sec. II\n", "MODEL_TAG = f\"{FUNCTIONAL}-{PSEUDOPOTENTIAL_TYPE} {ECUTWFC}-{ECUTRHO}Ry k{KPOINT_DENSITY}\"\n", "\n", "SCF_UNIT = \"pw_scf\"\n", @@ -446,7 +454,7 @@ "source": [ "## 6. Relax the defective cell (optional)\n", "\n", - "Runs only if `RELAX`: finds any relaxed version of this structure already on the account,\n", + "Runs only on the `production` tier: finds any relaxed version of this structure already on the account,\n", "regardless of who relaxed it or with what -- before spending any compute on a new one. The band\n", "structure and the total energy are then both taken off that one geometry. In Fujimoto & Saito the\n", "pyridinic C-N bond contracts from 1.41 Å to 1.33 Å and the acceptor-like states near E_F belong to\n", @@ -742,17 +750,16 @@ "FUJIMOTO_FORMATION_ENERGY = 2.51 # eV, Table I, trimerized pyridine-type C28N3\n", "TOLERANCE_FRACTION = 0.15\n", "\n", - "regime = \"relaxed defect\" if RELAX else \"unrelaxed SCF\"\n", "deviation = e_formation - FUJIMOTO_FORMATION_ENERGY\n", - "verdict = \"yes\" if abs(deviation) <= TOLERANCE_FRACTION * FUJIMOTO_FORMATION_ENERGY else \"no\"\n", - "print(f\"Regime: {regime}\")\n", + "verdict = \"yes\" if RELAX and abs(deviation) <= TOLERANCE_FRACTION * FUJIMOTO_FORMATION_ENERGY else \"no\"\n", + "print(f\"Regime: {RUN_TIER} tier\")\n", "print(f\"E_f (this notebook): {e_formation:.3f} eV\")\n", "print(f\"E_f (Fujimoto & Saito 2011, Table I): {FUJIMOTO_FORMATION_ENERGY:.3f} eV\")\n", "print(f\"Deviation: {deviation:+.3f} eV ({100 * deviation / FUJIMOTO_FORMATION_ENERGY:+.1f} %, \"\n", " f\"tolerance {100 * TOLERANCE_FRACTION:.0f} %)\")\n", "print(\"Offsets: μ_N comes from solid N2 (mp-154) rather than the free molecule the manuscript \"\n", " \"uses, and GBRV ultrasoft LDA stands in for its Troullier-Martins norm-conserving set.\")\n", - "print(f\"Reproduces Fujimoto & Saito (2011): {verdict} ({regime})\")" + "print(f\"Reproduces Fujimoto & Saito (2011): {verdict} ({RUN_TIER} tier)\")" ] }, { From 44a85c036215552e2ba443ad806c663b3e665f92 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 2 Oct 2026 15:26:11 -0700 Subject: [PATCH 8/9] SOF-8063: model-scoped total energy reuse, a source switch, and one prerequisite job at a time find_total_energy_for_material sorted by precision and took the first, so an LDA notebook could pick up a stranger's PBE energy for the same material (measured: -5195.0138 eV at KPPRA 2000, group qe:dft:gga:pbe, beside our own -5208.9120 eV at KPPRA 1116, group qe:dft:lda:pz) and print E_f = -4098 eV. The platform's own "Resolve Total Energies for Elemental Materials" subworkflow constrains 'group' in its query; the helper, which documents itself as mirroring it, did not. Optional group and precision_value arguments restore that, leaving the merged callers unchanged. The notebook keeps its own-account job-name lookup as the first choice and falls back to a TOTAL_ENERGY_SOURCE property scoped by MODEL_GROUP and the cell's KPPRA. Prerequisite jobs are submitted one at a time: celim is a concurrency limit, so a fan-out of three errors whenever anything else is active on the account. Co-Authored-By: Claude Opus 5 (1M context) --- ...int_substitution_graphene_SIMULATION.ipynb | 40 ++++++++++++++----- .../core/entity/property/api.py | 15 ++++++- .../py/unit/core/entity/test_property_api.py | 19 +++++++++ 3 files changed, 64 insertions(+), 10 deletions(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index b4e9f5810..8dd67afbb 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -166,6 +166,13 @@ "# Names the relaxation job; the relaxed structure itself is found by content hash.\n", "RELAX_TAG = f\"{MODEL_TAG} f{RELAXATION_SETTINGS['forc_conv_thr']}\"\n", "\n", + "# Whose total_energy properties to reuse when this notebook has no job of its own for a material:\n", + "# \"public\" (any owner), \"curators\" (only curators'), or \"my_account\" (only your own).\n", + "TOTAL_ENERGY_SOURCE = \"my_account\"\n", + "# The model identifier a reusable property must carry. It stops at the functional, so it pins\n", + "# neither the pseudopotential family nor the cutoffs; the k-point density is matched separately.\n", + "MODEL_GROUP = f\"qe:dft:{MODEL_SUBTYPE}:{FUNCTIONAL}\"\n", + "\n", "KPATH_STEPS = 20\n", "KPATH = [\n", " {\"point\": \"K\", \"steps\": KPATH_STEPS},\n", @@ -387,6 +394,8 @@ "metadata": {}, "outputs": [], "source": [ + "import math\n", + "\n", "from mat3ra.notebooks_utils.workflow import kgrid_from_density\n", "\n", "material_objects = {PRISTINE_NAME: pristine, DEFECTIVE_NAME: defective, NITROGEN_NAME: nitrogen}\n", @@ -395,8 +404,9 @@ " DEFECTIVE_NAME: kgrid_from_density(defective, KPOINT_DENSITY, periodic_dims=(0, 1)),\n", " NITROGEN_NAME: kgrid_from_density(nitrogen, KPOINT_DENSITY),\n", "}\n", + "kppra = {name: math.prod(grid) * material_objects[name].basis.number_of_atoms for name, grid in kgrid.items()}\n", "for name, grid in kgrid.items():\n", - " print(f\"{name}: k-grid {grid}\")" + " print(f\"{name}: k-grid {grid}, KPPRA {kppra[name]}\")" ] }, { @@ -527,7 +537,12 @@ "id": "34", "metadata": {}, "source": [ - "## 7. Total Energy jobs for the defective, pristine and nitrogen cells" + "## 7. Total Energy jobs for the defective, pristine and nitrogen cells\n", + "\n", + "A total energy already on the platform is reused when it comes from this notebook's own job for the\n", + "material, or, failing that, from a `TOTAL_ENERGY_SOURCE` property computed with the same model and\n", + "the same k-point density. The jobs that remain are submitted one at a time: the account's\n", + "concurrency limit rejects a batch whenever another job of yours is already running." ] }, { @@ -537,6 +552,8 @@ "metadata": {}, "outputs": [], "source": [ + "from mat3ra.notebooks_utils.core.entity.property.api import find_total_energy_for_material\n", + "\n", "total_energy_workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(\n", " TOTAL_ENERGY_SEARCH_TERM\n", ")\n", @@ -547,7 +564,6 @@ "}\n", "total_energies = {}\n", "total_energy_job_ids = {}\n", - "new_job_ids = []\n", "for name, saved_material in saved_materials.items():\n", " workflow_name = f\"Total Energy {name} {MODEL_TAG}\"\n", " existing_job = find_job_for_material(client, saved_material[\"_id\"], workflow_name, ACCOUNT_ID)\n", @@ -555,6 +571,15 @@ " total_energies[name] = get_properties_for_job(client, existing_job[\"_id\"], \"total_energy\")[0][\"value\"]\n", " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV\")\n", " continue\n", + " existing_property = find_total_energy_for_material(\n", + " client, saved_material[\"_id\"], source=TOTAL_ENERGY_SOURCE,\n", + " group=MODEL_GROUP, precision_value=kppra[name],\n", + " )\n", + " if existing_property is not None:\n", + " total_energies[name] = existing_property[\"data\"][\"value\"]\n", + " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV \"\n", + " f\"from {existing_property['owner']['slug']}\")\n", + " continue\n", " workflow = Workflow.create(total_energy_workflow_config)\n", " workflow.name = workflow_name\n", " workflow.subworkflows[0].model = model\n", @@ -566,7 +591,6 @@ " owner_id=ACCOUNT_ID, compute=compute.to_dict(), prefix=f\"{workflow.name} {timestamp}\",\n", " )\n", " total_energy_job_ids[name] = job[\"_id\"]\n", - " new_job_ids.append(job[\"_id\"])\n", " print(f\"✅ {name}: created Total Energy job {job['_id']}\")" ] }, @@ -577,12 +601,10 @@ "metadata": {}, "outputs": [], "source": [ - "if new_job_ids:\n", - " submit_jobs(client.jobs, new_job_ids)\n", - " print(f\"✅ Submitted {len(new_job_ids)} Total Energy job(s).\")\n", - " await wait_for_jobs_to_finish_async(client.jobs, new_job_ids, poll_interval=POLL_INTERVAL)\n", - "\n", "for name, job_id in total_energy_job_ids.items():\n", + " submit_jobs(client.jobs, [job_id])\n", + " print(f\"✅ {name}: submitted Total Energy job {job_id}\")\n", + " await wait_for_jobs_to_finish_async(client.jobs, [job_id], poll_interval=POLL_INTERVAL)\n", " total_energies[name] = get_properties_for_job(client, job_id, \"total_energy\")[0][\"value\"]\n", " print(f\"{name}: {total_energies[name]:.4f} eV\")" ] diff --git a/src/py/mat3ra/notebooks_utils/core/entity/property/api.py b/src/py/mat3ra/notebooks_utils/core/entity/property/api.py index 050c16f4f..65e59feb7 100644 --- a/src/py/mat3ra/notebooks_utils/core/entity/property/api.py +++ b/src/py/mat3ra/notebooks_utils/core/entity/property/api.py @@ -71,7 +71,13 @@ def update_property_holder_value(client: APIClient, property_holder_id: str, val return client.properties.update(property_holder_id, {"$set": {"data.value": value}}) -def find_total_energy_for_material(client: APIClient, material_id: str, source: str = "my_account") -> Optional[dict]: +def find_total_energy_for_material( + client: APIClient, + material_id: str, + source: str = "my_account", + group: Optional[str] = None, + precision_value: Optional[float] = None, +) -> Optional[dict]: """ Find the best-precision total_energy property for a material. Mirrors the platform's "Resolve Total Energies for Elemental Materials" subworkflow, @@ -86,6 +92,9 @@ def find_total_energy_for_material(client: APIClient, material_id: str, source: material_id (str): Material _id to look up the total_energy property for. source (str): Source of the total energy property: `my_account` (default), `curators` or `public`. + group (str, optional): Model identifier the property must carry, e.g. `qe:dft:lda:pz`. + Without it a property computed with a different model is returned. + precision_value (float, optional): `precision.value` (KPPRA) the property must carry. Returns: The best-precision total_energy property, or None if none exists. @@ -95,6 +104,10 @@ def find_total_energy_for_material(client: APIClient, material_id: str, source: if not exabyte_id: return None query = {"exabyteId": exabyte_id, "slug": "total_energy"} + if group: + query["group"] = group + if precision_value is not None: + query["precision.value"] = precision_value if source == "curators": query["owner.slug"] = "curators" elif source == "my_account": diff --git a/tests/py/unit/core/entity/test_property_api.py b/tests/py/unit/core/entity/test_property_api.py index 8dc5585c8..f1693bc32 100644 --- a/tests/py/unit/core/entity/test_property_api.py +++ b/tests/py/unit/core/entity/test_property_api.py @@ -9,6 +9,8 @@ EXABYTE_ID = "exabyte-a" MATERIAL = {"_id": MATERIAL_ID, "exabyteId": EXABYTE_ID} OWNER_ACCOUNT_ID = "account-a" +GROUP = "qe:dft:lda:pz" +PRECISION_VALUE = 1116 TOTAL_ENERGY_PROPERTY = {"data": {"value": -12.34}, "precision": {"value": 0.001}} @@ -84,6 +86,23 @@ def test_find_total_energy_for_material_public_scope_has_no_owner_filter(): ) +def test_find_total_energy_for_material_constrains_group_and_precision(): + client = _client() + + find_total_energy_for_material(client, MATERIAL_ID, group=GROUP, precision_value=PRECISION_VALUE) + + client.properties.list.assert_called_once_with( + query={ + "exabyteId": EXABYTE_ID, + "slug": "total_energy", + "group": GROUP, + "precision.value": PRECISION_VALUE, + "owner._id": OWNER_ACCOUNT_ID, + }, + projection={"sort": {"precision.value": -1}, "limit": 1}, + ) + + def test_find_total_energy_for_material_rejects_invalid_source(): client = _client() From 46d836743e33c17981ab6abf472a044253381ad5 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Fri, 2 Oct 2026 15:34:05 -0700 Subject: [PATCH 9/9] SOF-8063: default the total-energy source to public, safe now that group is matched Co-Authored-By: Claude Opus 5 (1M context) --- .../defect_point_substitution_graphene_SIMULATION.ipynb | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index 8dd67afbb..f662193de 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -168,7 +168,7 @@ "\n", "# Whose total_energy properties to reuse when this notebook has no job of its own for a material:\n", "# \"public\" (any owner), \"curators\" (only curators'), or \"my_account\" (only your own).\n", - "TOTAL_ENERGY_SOURCE = \"my_account\"\n", + "TOTAL_ENERGY_SOURCE = \"public\" # so a first run reuses references someone has already computed\n", "# The model identifier a reusable property must carry. It stops at the functional, so it pins\n", "# neither the pseudopotential family nor the cutoffs; the k-point density is matched separately.\n", "MODEL_GROUP = f\"qe:dft:{MODEL_SUBTYPE}:{FUNCTIONAL}\"\n",