diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb
index 4b0b8ecd5..f78e5f2b6 100644
--- a/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb
+++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb
@@ -222,7 +222,10 @@
"id": "16",
"metadata": {},
"source": [
- "## 4. Write resulting material to the folder"
+ "## 4. Write resulting materials to the folder\n",
+ "\n",
+ "Both materials are saved to `uploads` under the names the simulation notebook loads by exact name:\n",
+ "`graphene 4x4` is the reference cell, `graphene 4x4 N3V pyridinic (C28N3)` is the defective one."
]
},
{
@@ -235,9 +238,10 @@
"from mat3ra.notebooks_utils.io import download_content_to_file\n",
"from mat3ra.notebooks_utils.material import set_materials\n",
"\n",
- "material_with_defect.name = \"N-doped Graphene\"\n",
- "set_materials(material_with_defect)\n",
- "download_content_to_file(material_with_defect.to_json(), \"N-doped_Graphene.json\")"
+ "supercell.name = \"graphene 4x4\"\n",
+ "material_with_defect.name = \"graphene 4x4 N3V pyridinic (C28N3)\"\n",
+ "set_materials([supercell, material_with_defect])\n",
+ "download_content_to_file(material_with_defect.to_json(), \"graphene_4x4_N3V_pyridinic.json\")"
]
}
],
diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb
index 4902ba679..f662193de 100644
--- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb
+++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb
@@ -5,31 +5,35 @@
"id": "0",
"metadata": {},
"source": [
- "# Calculating Band Structure of N-doped Graphene.\n",
+ "# Formation Energy and Band Structure of the Pyridinic N₃-Vacancy Defect in Graphene\n",
"\n",
- "> **Yoshitaka Fujimoto and Susumu Saito**, \"Formation, stabilities, and electronic properties of nitrogen defects in graphene\", Physical Review B, 2011. [DOI: 10.1103/PhysRevB.84.245446](https://journals.aps.org/prb/abstract/10.1103/PhysRevB.84.245446).\n",
+ "> **Yoshitaka Fujimoto and Susumu Saito**, \"Formation, stabilities, and electronic properties of\n",
+ "> nitrogen defects in graphene\", Physical Review B 84, 245446 (2011).\n",
+ "> [DOI:10.1103/PhysRevB.84.245446](https://doi.org/10.1103/PhysRevB.84.245446)\n",
"\n",
- "\n",
- "Calculate the band structure of material created in the specialized notebook using Quantum ESPRESSO application and the band structure workflow available in Standata.\n",
+ "Computes the formation energy and the band structure of the trimerized pyridine-type C₂₈N₃ defect\n",
+ "created in the [structure notebook](defect_point_substitution_graphene.ipynb), both off the same\n",
+ "cell, and compares the formation energy with the manuscript's Table I entry, 2.51 eV.\n",
"\n",
"
Usage
\n",
"\n",
- "1. Create the material in the [defect creation notebook](defect_point_substitution_graphene.ipynb) or place the material file in `../uploads` folder.\n",
- "2. Set material and calculation parameters in cell 1.2. below (or use the default values). Choose between \"debug\" and \"production\" modes using `RUN_PROFILE` variable.\n",
+ "1. Create the materials in the [structure notebook](defect_point_substitution_graphene.ipynb), which saves `graphene 4x4` and `graphene 4x4 N3V pyridinic (C28N3)` to the `uploads` folder.\n",
+ "1. Set the material names and parameters in cells 1.2-1.4 (or use the defaults); `RUN_TIER = \"production\"` runs the manuscript's k-grid and relaxes C₂₈N₃ once, saving it as ` relaxed` for reuse by this and other notebooks.\n",
"1. Click \"Run\" > \"Run All\" to run all cells.\n",
- "1. Wait for the job to complete.\n",
- "1. Scroll down to view the result.\n",
+ "1. Wait for the jobs to complete.\n",
+ "1. Scroll down to view the results.\n",
"\n",
"## Summary\n",
"\n",
- "1. Set up the environment and parameters: install packages (JupyterLite only) and configure parameters for material, workflow, compute resources, and job.\n",
+ "1. Set up the environment and parameters: install packages (JupyterLite only) and configure the material names, model and compute parameters.\n",
"1. Authenticate and initialize API client: authenticate via browser, initialize the client, then select account and project.\n",
- "1. Create material: materials are read from the `../uploads` folder — place files there manually or run a [material creation notebook](defect_point_substitution_graphene.ipynb) first. If the material is not found by name, Standata is used as a fallback. The material is then saved to the platform.\n",
- "1. Create workflow and set its parameters: select application, load total energy workflow from Standata, optionally add relaxation or adjust model/method parameters, and save the workflow to the platform.\n",
- "1. Configure compute: get list of clusters and create compute configuration with selected cluster, queue, and number of processors.\n",
- "1. Create the job with material and workflow configuration: assemble the job from material, workflow, project, and compute configuration.\n",
- "1. Submit the job and monitor the status: submit the job and wait for completion.\n",
- "1. Retrieve results: get and display the properties."
+ "1. Load the materials by name from the `uploads` folder, resolve the nitrogen reference, and save them to the platform.\n",
+ "1. Configure the shared DFT model and k-grid: one model and a per-material k-grid for every job below.\n",
+ "1. Configure compute: get the list of clusters and create a compute configuration.\n",
+ "1. Relax the defective cell on the `production` tier, reusing a saved relaxed structure when one already exists.\n",
+ "1. Create the missing Total Energy jobs for the defective, pristine and nitrogen cells and wait for them.\n",
+ "1. Configure the Band Structure workflow and run it on the defective cell.\n",
+ "1. Retrieve the band structure, assemble the formation energy and compare it with the manuscript."
]
},
{
@@ -50,7 +54,7 @@
"source": [
"from mat3ra.notebooks_utils.packages import install_packages\n",
"\n",
- "await install_packages(\"api_examples\")"
+ "await install_packages(\"made|specific_examples|api_examples\")"
]
},
{
@@ -58,14 +62,7 @@
"id": "3",
"metadata": {},
"source": [
- "### 1.2. Set parameters and configurations for the workflow and job\n",
- "\n",
- "Use `RUN_PROFILE` in the next cell to switch between two predefined modes:\n",
- "\n",
- "- `\"debug\"`: quick validation run (small grids/path, no relaxation) to finish in a few minutes.\n",
- "- `\"production\"`: paper-like settings (relaxation + denser sampling) for final-quality results; this can take much longer.\n",
- "\n",
- "For first-time use, start with `\"debug\"`. Once everything works, switch to `\"production\"` and rerun the notebook."
+ "### 1.2. Material names"
]
},
{
@@ -74,100 +71,130 @@
"id": "4",
"metadata": {},
"outputs": [],
+ "source": [
+ "# Names saved by defect_point_substitution_graphene.ipynb.\n",
+ "PRISTINE_NAME = \"graphene 4x4\"\n",
+ "DEFECTIVE_NAME = \"graphene 4x4 N3V pyridinic (C28N3)\"\n",
+ "# Standata's solid nitrogen -- the platform carries no free N2 molecule.\n",
+ "NITROGEN_NAME = \"N2, Nitrogen, FCC (P2_13) 3D (Bulk), mp-154\""
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "5",
+ "metadata": {},
+ "source": [
+ "### 1.3. Parameters\n",
+ "\n",
+ "`RUN_TIER = \"fast\"` runs end to end in minutes; `\"production\"` takes about 1.5 hours from cold, which outlives the browser session — every job here is looked up by name before it is created, so signing in again and re-running attaches to the jobs already on the cluster instead of resubmitting them."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "6",
+ "metadata": {},
+ "outputs": [],
"source": [
"from datetime import datetime\n",
"from mat3ra.ide.compute import QueueName\n",
"\n",
- "# 2. Auth and organization parameters\n",
"ORGANIZATION_NAME = None # set to your organization name (full or partial); otherwise, your default one is used\n",
- "\n",
- "# 3. Material parameters\n",
"FOLDER = \"./uploads\"\n",
- "MATERIAL_NAME = \"N-doped Graphene\"\n",
"\n",
- "# 4. Workflow parameters\n",
- "WORKFLOW_SEARCH_TERM = \"band_structure.json\"\n",
- "RUN_PROFILE = \"debug\" # Change to \"production\" for paper-quality results\n",
+ "TOTAL_ENERGY_SEARCH_TERM = \"total_energy.json\"\n",
+ "RELAX_WORKFLOW_SEARCH_TERM = \"fixed_cell_relaxation.json\"\n",
+ "BAND_STRUCTURE_WORKFLOW_SEARCH_TERM = \"band_structure.json\"\n",
"MY_WORKFLOW_NAME = \"Band Structure\"\n",
"APPLICATION_NAME = \"espresso\"\n",
"\n",
- "# DFT model (same for both profiles)\n",
- "MODEL_SUBTYPE = \"lda\"\n",
- "\n",
- "# 5. Compute parameters\n",
- "CLUSTER_NAME = \"101\" # specify full or partial name i.e. \"cluster-101\" to select\n",
- "QUEUE_NAME = QueueName.D\n",
- "PPN = 1\n",
+ "RUN_TIER = \"fast\" # minutes; coarse k-grid, no relaxation — will NOT match the manuscript\n",
+ "# RUN_TIER = \"production\" # ~1.5 h; 6x6x1 and relaxation to 0.05 eV/Å — reproduces Table I\n",
+ "TIER_SETTINGS = {\n",
+ " \"fast\": {\"kpoint_density\": 3.5, \"relax\": False},\n",
+ " \"production\": {\"kpoint_density\": 7, \"relax\": True},\n",
+ "}\n",
+ "# Points per Å⁻¹: 3.5 gives 3 x 3 x 1 on the 4x4 cell, 7 gives the manuscript's 6 x 6 x 1.\n",
+ "KPOINT_DENSITY = TIER_SETTINGS[RUN_TIER][\"kpoint_density\"]\n",
+ "RELAX = TIER_SETTINGS[RUN_TIER][\"relax\"]\n",
+ "\n",
+ "CLUSTER_NAME = \"001\" # specify full or partial name i.e. \"cluster-001\" to select\n",
+ "QUEUE_NAME = QueueName.OR\n",
+ "PPN = 16 # queue OR on cluster-001 allows at most 16 cores per node\n",
+ "TIME_LIMIT = \"04:00:00\" # covers the optional relaxation (~1 h); walltime x ppn is reserved up front\n",
"\n",
- "# 6. Job parameters\n",
"timestamp = datetime.now().strftime(\"%Y-%m-%d %H:%M\")\n",
- "JOB_NAME = MY_WORKFLOW_NAME + \" \" + MATERIAL_NAME + f\" ({RUN_PROFILE}) \" + timestamp\n",
- "POLL_INTERVAL = 60 # seconds\n"
+ "POLL_INTERVAL = 60 # seconds"
]
},
{
"cell_type": "markdown",
- "id": "5",
+ "id": "7",
"metadata": {},
"source": [
- "### 1.3. Set specific parameters"
+ "### 1.4. DFT model parameters"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "6",
+ "id": "8",
"metadata": {},
"outputs": [],
"source": [
- "# Parameters for the Method Selected\n",
- "PSEUDOPOTENTIAL_TYPE = \"nc\"\n",
+ "# NOTE: the manuscript's value needs its own settings — Troullier-Martins norm-conserving LDA;\n",
+ "# its 4x4 cell and 50 Ry cutoff are matched below, its 6x6x1 k-grid on the production tier,\n",
+ "# its pseudopotentials not at all.\n",
+ "MODEL_SUBTYPE = \"lda\"\n",
"FUNCTIONAL = \"pz\"\n",
- "\n",
- "# Energy cutoffs (check per the ONCV guidelines - https://www.pseudo-dojo.org/)\n",
- "ECUTWFC = 50\n",
- "# Density cutoff in relation to wavefunction cutoff\n",
- "ECUTRHO = 4 * ECUTWFC\n",
- "\n",
- "# Parameters for the specific property\n",
- "if RUN_PROFILE == \"debug\":\n",
- " # Quick validation run (finishes in a few minutes)\n",
- " ADD_RELAXATION = False\n",
- " RELAXATION_KGRID = [1, 1, 1]\n",
- " SCF_KGRID = [1, 1, 1]\n",
- " KPATH = [\n",
- " {\"point\": \"K\", \"steps\": 6},\n",
- " {\"point\": \"Γ\", \"steps\": 6},\n",
- " {\"point\": \"M\", \"steps\": 6},\n",
- " {\"point\": \"K\", \"steps\": 1},\n",
- " ]\n",
- "elif RUN_PROFILE == \"production\":\n",
- " # Paper settings (Fujimoto & Saito 2011): relaxation + dense sampling\n",
- " ADD_RELAXATION = True\n",
- " RELAXATION_KGRID = [6, 6, 1]\n",
- " SCF_KGRID = [6, 6, 1]\n",
- " KPATH = [\n",
- " {\"point\": \"K\", \"steps\": 20},\n",
- " {\"point\": \"Γ\", \"steps\": 20},\n",
- " {\"point\": \"M\", \"steps\": 20},\n",
- " {\"point\": \"K\", \"steps\": 1},\n",
- " ]"
+ "PSEUDOPOTENTIAL_TYPE = \"us\" # GBRV ultrasoft, the only LDA family the platform publishes for C and N\n",
+ "ECUTWFC = 50 # Ry, Fujimoto & Saito Sec. II\n",
+ "ECUTRHO = 200 # Ry, GBRV's recommended charge-density cutoff\n",
+ "\n",
+ "MODEL_TAG = f\"{FUNCTIONAL}-{PSEUDOPOTENTIAL_TYPE} {ECUTWFC}-{ECUTRHO}Ry k{KPOINT_DENSITY}\"\n",
+ "\n",
+ "SCF_UNIT = \"pw_scf\"\n",
+ "BANDS_UNIT = \"pw_bands\"\n",
+ "RELAX_UNIT = \"pw_relax\"\n",
+ "# C28N3 carries 127 valence electrons, so its ground state is a doublet.\n",
+ "SPIN_SETTINGS = {\n",
+ " DEFECTIVE_NAME: {\"nspin\": 2, \"tot_magnetization\": 1},\n",
+ " PRISTINE_NAME: {\"nspin\": 1},\n",
+ " NITROGEN_NAME: {\"nspin\": 1},\n",
+ "}\n",
+ "RELAXATION_SETTINGS = {\"forc_conv_thr\": 1.9e-3, \"nstep\": 100} # 0.05 eV/Å, Fujimoto & Saito Sec. II\n",
+ "# Names the relaxation job; the relaxed structure itself is found by content hash.\n",
+ "RELAX_TAG = f\"{MODEL_TAG} f{RELAXATION_SETTINGS['forc_conv_thr']}\"\n",
+ "\n",
+ "# Whose total_energy properties to reuse when this notebook has no job of its own for a material:\n",
+ "# \"public\" (any owner), \"curators\" (only curators'), or \"my_account\" (only your own).\n",
+ "TOTAL_ENERGY_SOURCE = \"public\" # so a first run reuses references someone has already computed\n",
+ "# The model identifier a reusable property must carry. It stops at the functional, so it pins\n",
+ "# neither the pseudopotential family nor the cutoffs; the k-point density is matched separately.\n",
+ "MODEL_GROUP = f\"qe:dft:{MODEL_SUBTYPE}:{FUNCTIONAL}\"\n",
+ "\n",
+ "KPATH_STEPS = 20\n",
+ "KPATH = [\n",
+ " {\"point\": \"K\", \"steps\": KPATH_STEPS},\n",
+ " {\"point\": \"Γ\", \"steps\": KPATH_STEPS},\n",
+ " {\"point\": \"M\", \"steps\": KPATH_STEPS},\n",
+ " {\"point\": \"K\", \"steps\": 1},\n",
+ "]"
]
},
{
"cell_type": "markdown",
- "id": "7",
+ "id": "9",
"metadata": {},
"source": [
"## 2. Authenticate and initialize API client\n",
- "### 2.1. Authenticate\n",
- "Authenticate in the browser and have credentials stored in environment variable \"OIDC_ACCESS_TOKEN\".\n"
+ "### 2.1. Authenticate"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "8",
+ "id": "10",
"metadata": {},
"outputs": [],
"source": [
@@ -178,16 +205,16 @@
},
{
"cell_type": "markdown",
- "id": "9",
+ "id": "11",
"metadata": {},
"source": [
- "### 2.2. Initialize API Client\n"
+ "### 2.2. Initialize API client"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "10",
+ "id": "12",
"metadata": {},
"outputs": [],
"source": [
@@ -199,16 +226,16 @@
},
{
"cell_type": "markdown",
- "id": "11",
+ "id": "13",
"metadata": {},
"source": [
- "### 2.3. Select account to work under"
+ "### 2.3. Select account"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "12",
+ "id": "14",
"metadata": {},
"outputs": [],
"source": [
@@ -218,7 +245,7 @@
{
"cell_type": "code",
"execution_count": null,
- "id": "13",
+ "id": "15",
"metadata": {},
"outputs": [],
"source": [
@@ -233,7 +260,7 @@
},
{
"cell_type": "markdown",
- "id": "14",
+ "id": "16",
"metadata": {},
"source": [
"### 2.4. Select project"
@@ -242,7 +269,7 @@
{
"cell_type": "code",
"execution_count": null,
- "id": "15",
+ "id": "17",
"metadata": {},
"outputs": [],
"source": [
@@ -251,41 +278,13 @@
"print(f\"✅ Using project: {projects[0]['name']} ({project_id})\")"
]
},
- {
- "cell_type": "markdown",
- "id": "16",
- "metadata": {},
- "source": [
- "## 3. Create material\n",
- "### 3.1. Load material from local file (or Standata)"
- ]
- },
- {
- "cell_type": "code",
- "execution_count": null,
- "id": "17",
- "metadata": {},
- "outputs": [],
- "source": [
- "from mat3ra.made.material import Material\n",
- "from mat3ra.standata.materials import Materials\n",
- "from mat3ra.notebooks_utils.ipython.entity.material.visualize import visualize_materials as visualize\n",
- "from mat3ra.notebooks_utils.material import load_material_from_folder\n",
- "\n",
- "material = load_material_from_folder(FOLDER, MATERIAL_NAME) or Material.create(\n",
- " Materials.get_by_name_first_match(MATERIAL_NAME))\n",
- "\n",
- "# set lattice for the correct display of Point Path for this material\n",
- "material.lattice.type = \"HEX\"\n",
- "visualize(material)"
- ]
- },
{
"cell_type": "markdown",
"id": "18",
"metadata": {},
"source": [
- "### 3.2. Save material to the platform"
+ "## 3. Load the materials\n",
+ "### 3.1. Load from the uploads folder or the platform, and print provenance"
]
},
{
@@ -295,10 +294,15 @@
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.notebooks_utils.core.entity.material.api import get_or_create_material\n",
+ "from collections import Counter\n",
+ "from mat3ra.notebooks_utils.core.entity.material.api import load_material\n",
"\n",
- "saved_material_response = get_or_create_material(client, material, ACCOUNT_ID)\n",
- "saved_material = Material.create(saved_material_response)"
+ "materials_by_name = {name: load_material(client, FOLDER, name, ACCOUNT_ID) for name in (PRISTINE_NAME, DEFECTIVE_NAME)}\n",
+ "pristine, defective = materials_by_name[PRISTINE_NAME], materials_by_name[DEFECTIVE_NAME]\n",
+ "for name, material in materials_by_name.items():\n",
+ " composition = \"\".join(f\"{e}{n}\" for e, n in sorted(Counter(material.basis.elements.values).items()))\n",
+ " print(f\"{name}: {composition}, {material.basis.number_of_atoms} atoms, \"\n",
+ " f\"cell {material.lattice.a:.2f} x {material.lattice.b:.2f} x {material.lattice.c:.2f} Å\")"
]
},
{
@@ -306,8 +310,7 @@
"id": "20",
"metadata": {},
"source": [
- "## 4. Create workflow and set its parameters\n",
- "### 4.1. Get list of applications and select one"
+ "### 3.2. Resolve the nitrogen reference material"
]
},
{
@@ -317,12 +320,11 @@
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.standata.applications import ApplicationStandata\n",
- "from mat3ra.ade.application import Application\n",
+ "from mat3ra.made.material import Material\n",
+ "from mat3ra.standata.materials import Materials\n",
"\n",
- "app_config = ApplicationStandata.get_by_name_first_match(APPLICATION_NAME)\n",
- "app = Application(**app_config)\n",
- "print(f\"Using application: {app.name}\")"
+ "nitrogen = Material.create(Materials.get_by_name_first_match(NITROGEN_NAME))\n",
+ "print(f\"{nitrogen.name}: {nitrogen.basis.number_of_atoms} atoms, cell {nitrogen.lattice.a:.2f} Å\")"
]
},
{
@@ -330,7 +332,7 @@
"id": "22",
"metadata": {},
"source": [
- "### 4.2. Create workflow from standard workflows and preview it"
+ "### 3.3. Save the materials to the platform"
]
},
{
@@ -340,15 +342,11 @@
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.standata.workflows import WorkflowStandata\n",
- "from mat3ra.wode.workflows import Workflow\n",
- "from mat3ra.notebooks_utils.ipython.entity.workflow.visualize import visualize_workflow\n",
- "\n",
- "workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(WORKFLOW_SEARCH_TERM)\n",
- "workflow = Workflow.create(workflow_config)\n",
- "workflow.name = MY_WORKFLOW_NAME\n",
+ "from mat3ra.notebooks_utils.core.entity.material.api import get_or_create_material\n",
"\n",
- "visualize_workflow(workflow)"
+ "saved_pristine = get_or_create_material(client, pristine, ACCOUNT_ID)\n",
+ "saved_defective = get_or_create_material(client, defective, ACCOUNT_ID)\n",
+ "saved_nitrogen = get_or_create_material(client, nitrogen, ACCOUNT_ID)"
]
},
{
@@ -356,8 +354,8 @@
"id": "24",
"metadata": {},
"source": [
- "### 4.3. Modify workflow (Optional)\n",
- "#### 4.3.1. Add relaxation"
+ "## 4. Configure the shared DFT model and k-grid\n",
+ "### 4.1. DFT model"
]
},
{
@@ -367,8 +365,18 @@
"metadata": {},
"outputs": [],
"source": [
- "if ADD_RELAXATION:\n",
- " workflow.add_relaxation()\n"
+ "from mat3ra.standata.applications import ApplicationStandata\n",
+ "from mat3ra.standata.model_tree import ModelTreeStandata\n",
+ "from mat3ra.ade.application import Application\n",
+ "from mat3ra.mode import ModelFactory\n",
+ "\n",
+ "app_config = ApplicationStandata.get_by_name_first_match(APPLICATION_NAME)\n",
+ "app = Application(**app_config)\n",
+ "\n",
+ "model_config = ModelTreeStandata.get_model_by_parameters(type=\"dft\", subtype=MODEL_SUBTYPE, functional=FUNCTIONAL)\n",
+ "model_config[\"method\"] = {\"type\": \"pseudopotential\", \"subtype\": PSEUDOPOTENTIAL_TYPE}\n",
+ "model = ModelFactory.create(model_config)\n",
+ "print(f\"Using application: {app.name}, model: {MODEL_TAG}\")"
]
},
{
@@ -376,8 +384,7 @@
"id": "26",
"metadata": {},
"source": [
- "#### 4.3.2. Modify model and method parameters (Optional)\n",
- "Uncomment the code below and adjust selection of model parameters as needed."
+ "### 4.2. k-grid per material"
]
},
{
@@ -387,19 +394,19 @@
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.standata.model_tree import ModelTreeStandata\n",
- "from mat3ra.mode.model import Model\n",
- "\n",
- "model_config = ModelTreeStandata.get_model_by_parameters(\n",
- " type=\"dft\", subtype=MODEL_SUBTYPE, functional=FUNCTIONAL\n",
- ")\n",
- "model_config[\"method\"] = {\"type\": \"pseudopotential\", \"subtype\": PSEUDOPOTENTIAL_TYPE}\n",
- "model = Model.create(model_config)\n",
+ "import math\n",
"\n",
- "for subworkflow in workflow.subworkflows:\n",
- " subworkflow.model = model\n",
+ "from mat3ra.notebooks_utils.workflow import kgrid_from_density\n",
"\n",
- "visualize_workflow(workflow)"
+ "material_objects = {PRISTINE_NAME: pristine, DEFECTIVE_NAME: defective, NITROGEN_NAME: nitrogen}\n",
+ "kgrid = {\n",
+ " PRISTINE_NAME: kgrid_from_density(pristine, KPOINT_DENSITY, periodic_dims=(0, 1)),\n",
+ " DEFECTIVE_NAME: kgrid_from_density(defective, KPOINT_DENSITY, periodic_dims=(0, 1)),\n",
+ " NITROGEN_NAME: kgrid_from_density(nitrogen, KPOINT_DENSITY),\n",
+ "}\n",
+ "kppra = {name: math.prod(grid) * material_objects[name].basis.number_of_atoms for name, grid in kgrid.items()}\n",
+ "for name, grid in kgrid.items():\n",
+ " print(f\"{name}: k-grid {grid}, KPPRA {kppra[name]}\")"
]
},
{
@@ -407,8 +414,8 @@
"id": "28",
"metadata": {},
"source": [
- "#### 4.3.3. Modify important settings\n",
- "Set k-grid, k-path and energy cutoffs, as well as other settings."
+ "## 5. Create the compute configuration\n",
+ "### 5.1. Select cluster"
]
},
{
@@ -418,48 +425,8 @@
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.wode.context.providers import PlanewaveCutoffsContextProvider, PointsGridDataProvider, \\\n",
- " PointsPathDataProvider\n",
- "\n",
- "bs_subworkflow = workflow.subworkflows[1 if ADD_RELAXATION else 0]\n",
- "cutoffs_context = PlanewaveCutoffsContextProvider(wavefunction=ECUTWFC, density=ECUTRHO, isEdited=True).get_context_item_data()\n",
- "\n",
- "if RELAXATION_KGRID is not None and ADD_RELAXATION:\n",
- " unit = workflow.subworkflows[0].get_unit_by_name(name_regex=\"relax\")\n",
- " unit.add_context(PointsGridDataProvider(material=material, dimensions=RELAXATION_KGRID, isEdited=True).get_context_item_data())\n",
- " workflow.subworkflows[0].set_unit(unit)\n",
- "\n",
- "if SCF_KGRID is not None:\n",
- " unit = bs_subworkflow.get_unit_by_name(name=\"pw_scf\")\n",
- " unit.add_context(PointsGridDataProvider(material=material, dimensions=SCF_KGRID, isEdited=True).get_context_item_data())\n",
- " bs_subworkflow.set_unit(unit)\n",
- "\n",
- "if KPATH is not None:\n",
- " unit = bs_subworkflow.get_unit_by_name(name=\"pw_bands\")\n",
- " unit.add_context(PointsPathDataProvider(path=KPATH, isEdited=True).get_context_item_data())\n",
- " bs_subworkflow.set_unit(unit)\n",
- "\n",
- "for unit_name in [\"pw_relax\", \"pw_vc-relax\", \"pw_scf\", \"pw_bands\"]:\n",
- " for swf in workflow.subworkflows:\n",
- " unit = swf.get_unit_by_name(name=unit_name)\n",
- " if unit:\n",
- " unit.add_context(cutoffs_context)\n",
- " swf.set_unit(unit)\n",
- "\n",
- "# Below is the example of how to modify the input file content directly.\n",
- "# bands_unit = bs_subworkflow.get_unit_by_name(name=\"bands\")\n",
- "# bands_unit.input = [{\"name\": \"bands.in\", \"content\": (\n",
- "# \"&BANDS\\n\"\n",
- "# \" prefix = '__prefix__'\\n\"\n",
- "# \" lsym = .false.\\n\"\n",
- "# \" outdir = {% raw %}'{{ JOB_WORK_DIR }}/outdir'{% endraw %}\\n\"\n",
- "# \" filband = {% raw %}'{{ JOB_WORK_DIR }}/bands.dat'{% endraw %}\\n\"\n",
- "# \" no_overlap = .true.\\n\"\n",
- "# \"/\"\n",
- "# )}]\n",
- "# bs_subworkflow.set_unit(bands_unit)\n",
- "\n",
- "visualize_workflow(workflow)"
+ "clusters = client.clusters.list()\n",
+ "print(f\"Available clusters: {[c['hostname'] for c in clusters]}\")"
]
},
{
@@ -467,7 +434,7 @@
"id": "30",
"metadata": {},
"source": [
- "### 4.4. Save workflow to collection"
+ "### 5.2. Create the compute configuration for the jobs"
]
},
{
@@ -477,12 +444,17 @@
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.utils.namespace import dict_to_namespace_recursive\n",
- "from mat3ra.notebooks_utils.core.entity.workflow.api import get_or_create_workflow\n",
+ "from mat3ra.ide.compute import Compute\n",
"\n",
- "saved_workflow_response = get_or_create_workflow(client, workflow, ACCOUNT_ID)\n",
- "saved_workflow = Workflow.create(saved_workflow_response)\n",
- "print(f\"Workflow ID: {saved_workflow.id}\")"
+ "if CLUSTER_NAME:\n",
+ " cluster = next((c for c in clusters if CLUSTER_NAME in c[\"hostname\"]), None)\n",
+ " if cluster is None:\n",
+ " raise ValueError(f\"Cluster '{CLUSTER_NAME}' not found. Available: {[c['hostname'] for c in clusters]}\")\n",
+ "else:\n",
+ " cluster = clusters[0]\n",
+ "compute = Compute(cluster=cluster, queue=QUEUE_NAME, ppn=PPN, timeLimit=TIME_LIMIT)\n",
+ "print(f\"Using cluster: {compute.cluster.hostname}, queue: {QUEUE_NAME}, ppn: {PPN}, \"\n",
+ " f\"time limit: {TIME_LIMIT}\")"
]
},
{
@@ -490,8 +462,13 @@
"id": "32",
"metadata": {},
"source": [
- "## 5. Create the compute configuration\n",
- "### 5.1. Get list of clusters"
+ "## 6. Relax the defective cell (optional)\n",
+ "\n",
+ "Runs only on the `production` tier: finds any relaxed version of this structure already on the account,\n",
+ "regardless of who relaxed it or with what -- before spending any compute on a new one. The band\n",
+ "structure and the total energy are then both taken off that one geometry. In Fujimoto & Saito the\n",
+ "pyridinic C-N bond contracts from 1.41 Å to 1.33 Å and the acceptor-like states near E_F belong to\n",
+ "the relaxed geometry, so an unrelaxed run is a different system rather than a rougher answer."
]
},
{
@@ -501,8 +478,58 @@
"metadata": {},
"outputs": [],
"source": [
- "clusters = client.clusters.list()\n",
- "print(f\"Available clusters: {[c['hostname'] for c in clusters]}\")"
+ "from mat3ra.standata.workflows import WorkflowStandata\n",
+ "from mat3ra.wode.workflows import Workflow\n",
+ "from mat3ra.notebooks_utils.workflow import apply_planewave_cutoffs, apply_scf_kgrid, patch_workflow_qe_input\n",
+ "from mat3ra.notebooks_utils.core.entity.job.api import find_job_for_material\n",
+ "from mat3ra.notebooks_utils.core.entity.material.api import find_relaxed_material, get_final_structure_for_job\n",
+ "from mat3ra.notebooks_utils.core.entity.property.api import get_properties_for_job\n",
+ "from mat3ra.notebooks_utils.job import create_job\n",
+ "from mat3ra.notebooks_utils.api.job import submit_jobs, wait_for_jobs_to_finish_async\n",
+ "\n",
+ "relaxed_defective = None\n",
+ "if RELAX:\n",
+ " relax_workflow_name = f\"Fixed-cell Relaxation {DEFECTIVE_NAME} {RELAX_TAG}\"\n",
+ " relaxed_defective = find_relaxed_material(client, defective, ACCOUNT_ID)\n",
+ " if relaxed_defective is not None:\n",
+ " print(f\"♻️ Relaxed defective material: {relaxed_defective.name} ({relaxed_defective.id})\")\n",
+ " else:\n",
+ " relax_workflow = Workflow.create(\n",
+ " WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(RELAX_WORKFLOW_SEARCH_TERM)\n",
+ " )\n",
+ " relax_workflow.name = relax_workflow_name\n",
+ " relax_workflow.subworkflows[0].model = model\n",
+ " apply_planewave_cutoffs(relax_workflow, ECUTWFC, ECUTRHO, unit_name=RELAX_UNIT)\n",
+ " apply_scf_kgrid(relax_workflow, kgrid[DEFECTIVE_NAME], material=defective, unit_name=RELAX_UNIT)\n",
+ " patch_workflow_qe_input(relax_workflow, {\"system\": SPIN_SETTINGS[DEFECTIVE_NAME]}, [RELAX_UNIT])\n",
+ " patch_workflow_qe_input(relax_workflow, {\"control\": RELAXATION_SETTINGS}, [RELAX_UNIT])\n",
+ "\n",
+ " relax_job = find_job_for_material(\n",
+ " client, saved_defective[\"_id\"], relax_workflow_name, ACCOUNT_ID,\n",
+ " statuses=(\"submitted\", \"queued\", \"active\", \"finished\"),\n",
+ " )\n",
+ " if relax_job is None:\n",
+ " relax_job = create_job(\n",
+ " api_client=client, materials=[saved_defective], workflow=relax_workflow,\n",
+ " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n",
+ " prefix=f\"{relax_workflow_name} {timestamp}\",\n",
+ " )\n",
+ " submit_jobs(client.jobs, [relax_job[\"_id\"]])\n",
+ " await wait_for_jobs_to_finish_async(client.jobs, [relax_job[\"_id\"]], poll_interval=POLL_INTERVAL)\n",
+ "\n",
+ " relaxed_material = get_final_structure_for_job(client, relax_job[\"_id\"])\n",
+ " client.materials.update(relaxed_material.id, {\"name\": f\"{DEFECTIVE_NAME} relaxed\"})\n",
+ " relaxed_defective = Material.create(client.materials.get(relaxed_material.id))\n",
+ " print(f\"✅ Relaxed defective material: {relaxed_defective.name} ({relaxed_defective.id})\")\n",
+ "\n",
+ " total_force = get_properties_for_job(client, relax_job[\"_id\"], \"total_force\")[0]\n",
+ " print(f\"Residual force after relaxation (norm over all atoms): \"\n",
+ " f\"{total_force['value']:.4f} {total_force['units']}\")\n",
+ "\n",
+ "defective_for_jobs = relaxed_defective if RELAX else defective\n",
+ "defective_for_jobs.lattice.type = \"HEX\" # the symbolic K-Γ-M-K path needs a named lattice type\n",
+ "material_objects[DEFECTIVE_NAME] = defective_for_jobs\n",
+ "saved_defective_for_jobs = get_or_create_material(client, defective_for_jobs, ACCOUNT_ID)"
]
},
{
@@ -510,7 +537,12 @@
"id": "34",
"metadata": {},
"source": [
- "### 5.2. Create compute configuration for the job\n"
+ "## 7. Total Energy jobs for the defective, pristine and nitrogen cells\n",
+ "\n",
+ "A total energy already on the platform is reused when it comes from this notebook's own job for the\n",
+ "material, or, failing that, from a `TOTAL_ENERGY_SOURCE` property computed with the same model and\n",
+ "the same k-point density. The jobs that remain are submitted one at a time: the account's\n",
+ "concurrency limit rejects a batch whenever another job of yours is already running."
]
},
{
@@ -520,122 +552,249 @@
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.ide.compute import Compute\n",
- "\n",
- "# Select cluster: use specified name if provided, otherwise use first available\n",
- "if CLUSTER_NAME:\n",
- " cluster = next((c for c in clusters if CLUSTER_NAME in c[\"hostname\"]), None)\n",
- "else:\n",
- " cluster = clusters[0]\n",
+ "from mat3ra.notebooks_utils.core.entity.property.api import find_total_energy_for_material\n",
"\n",
- "compute = Compute(\n",
- " cluster=cluster,\n",
- " queue=QUEUE_NAME,\n",
- " ppn=PPN\n",
+ "total_energy_workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(\n",
+ " TOTAL_ENERGY_SEARCH_TERM\n",
")\n",
- "print(f\"Using cluster: {compute.cluster.hostname}, queue: {QUEUE_NAME}, ppn: {PPN}\")"
+ "saved_materials = {\n",
+ " DEFECTIVE_NAME: saved_defective_for_jobs,\n",
+ " PRISTINE_NAME: saved_pristine,\n",
+ " NITROGEN_NAME: saved_nitrogen,\n",
+ "}\n",
+ "total_energies = {}\n",
+ "total_energy_job_ids = {}\n",
+ "for name, saved_material in saved_materials.items():\n",
+ " workflow_name = f\"Total Energy {name} {MODEL_TAG}\"\n",
+ " existing_job = find_job_for_material(client, saved_material[\"_id\"], workflow_name, ACCOUNT_ID)\n",
+ " if existing_job is not None:\n",
+ " total_energies[name] = get_properties_for_job(client, existing_job[\"_id\"], \"total_energy\")[0][\"value\"]\n",
+ " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV\")\n",
+ " continue\n",
+ " existing_property = find_total_energy_for_material(\n",
+ " client, saved_material[\"_id\"], source=TOTAL_ENERGY_SOURCE,\n",
+ " group=MODEL_GROUP, precision_value=kppra[name],\n",
+ " )\n",
+ " if existing_property is not None:\n",
+ " total_energies[name] = existing_property[\"data\"][\"value\"]\n",
+ " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV \"\n",
+ " f\"from {existing_property['owner']['slug']}\")\n",
+ " continue\n",
+ " workflow = Workflow.create(total_energy_workflow_config)\n",
+ " workflow.name = workflow_name\n",
+ " workflow.subworkflows[0].model = model\n",
+ " apply_planewave_cutoffs(workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n",
+ " apply_scf_kgrid(workflow, kgrid[name], material=material_objects[name])\n",
+ " patch_workflow_qe_input(workflow, {\"system\": SPIN_SETTINGS[name]}, [SCF_UNIT])\n",
+ " job = create_job(\n",
+ " api_client=client, materials=[saved_material], workflow=workflow, project_id=project_id,\n",
+ " owner_id=ACCOUNT_ID, compute=compute.to_dict(), prefix=f\"{workflow.name} {timestamp}\",\n",
+ " )\n",
+ " total_energy_job_ids[name] = job[\"_id\"]\n",
+ " print(f\"✅ {name}: created Total Energy job {job['_id']}\")"
]
},
{
- "cell_type": "markdown",
+ "cell_type": "code",
+ "execution_count": null,
"id": "36",
"metadata": {},
+ "outputs": [],
+ "source": [
+ "for name, job_id in total_energy_job_ids.items():\n",
+ " submit_jobs(client.jobs, [job_id])\n",
+ " print(f\"✅ {name}: submitted Total Energy job {job_id}\")\n",
+ " await wait_for_jobs_to_finish_async(client.jobs, [job_id], poll_interval=POLL_INTERVAL)\n",
+ " total_energies[name] = get_properties_for_job(client, job_id, \"total_energy\")[0][\"value\"]\n",
+ " print(f\"{name}: {total_energies[name]:.4f} eV\")"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "37",
+ "metadata": {},
"source": [
- "## 6. Create the job with material and workflow configuration\n",
- "### 6.1. Create job"
+ "## 8. Band structure of the defective cell\n",
+ "### 8.1. Configure the workflow"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "37",
+ "id": "38",
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.notebooks_utils.api.job import create_job\n",
- "from mat3ra.notebooks_utils.ui import display_JSON\n",
- "\n",
- "print(f\"Material: {saved_material.id}\")\n",
- "print(f\"Workflow: {saved_workflow.id}\")\n",
- "print(f\"Project: {project_id}\")\n",
- "\n",
- "job_response = create_job(\n",
- " api_client=client,\n",
- " materials=[saved_material],\n",
- " workflow=workflow,\n",
- " project_id=project_id,\n",
- " owner_id=ACCOUNT_ID,\n",
- " prefix=JOB_NAME,\n",
- " compute=compute.to_dict(),\n",
+ "from mat3ra.wode.context.providers import PointsPathDataProvider\n",
+ "from mat3ra.notebooks_utils.ipython.entity.workflow.visualize import visualize_workflow\n",
+ "\n",
+ "band_structure_workflow_name = f\"{MY_WORKFLOW_NAME} {DEFECTIVE_NAME} {MODEL_TAG}\"\n",
+ "band_structure_workflow = Workflow.create(\n",
+ " WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(BAND_STRUCTURE_WORKFLOW_SEARCH_TERM)\n",
")\n",
+ "band_structure_workflow.name = band_structure_workflow_name\n",
+ "band_structure_workflow.subworkflows[0].model = model\n",
+ "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n",
+ "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=BANDS_UNIT)\n",
+ "apply_scf_kgrid(band_structure_workflow, kgrid[DEFECTIVE_NAME], material=defective_for_jobs)\n",
+ "patch_workflow_qe_input(band_structure_workflow, {\"system\": SPIN_SETTINGS[DEFECTIVE_NAME]}, [SCF_UNIT, BANDS_UNIT])\n",
+ "\n",
+ "bands_subworkflow = band_structure_workflow.subworkflows[0]\n",
+ "bands_unit = bands_subworkflow.get_unit_by_name(name=BANDS_UNIT)\n",
+ "bands_unit.add_context(PointsPathDataProvider(path=KPATH, isEdited=True).get_context_item_data())\n",
+ "bands_subworkflow.set_unit(bands_unit)\n",
"\n",
- "job = dict_to_namespace_recursive(job_response)\n",
- "job_id = job._id\n",
- "print(\"✅ Job created successfully!\")\n",
- "print(f\"Job ID: {job_id}\")\n",
- "display_JSON(job_response)"
+ "visualize_workflow(band_structure_workflow)"
]
},
{
"cell_type": "markdown",
- "id": "38",
+ "id": "39",
"metadata": {},
"source": [
- "## 7. Submit the job and monitor the status"
+ "### 8.2. Create the job"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "39",
+ "id": "40",
"metadata": {},
"outputs": [],
"source": [
- "client.jobs.submit(job_id)\n",
- "print(f\"✅ Job {job_id} submitted successfully!\")"
+ "band_structure_job = find_job_for_material(\n",
+ " client, saved_defective_for_jobs[\"_id\"], band_structure_workflow_name, ACCOUNT_ID,\n",
+ " statuses=(\"submitted\", \"queued\", \"active\", \"finished\"),\n",
+ ")\n",
+ "if band_structure_job is None:\n",
+ " band_structure_job = create_job(\n",
+ " api_client=client, materials=[saved_defective_for_jobs], workflow=band_structure_workflow,\n",
+ " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n",
+ " prefix=f\"{band_structure_workflow.name} {timestamp}\",\n",
+ " )\n",
+ " new_band_structure_job_id = band_structure_job[\"_id\"]\n",
+ " print(f\"✅ Band Structure job created: {new_band_structure_job_id}\")\n",
+ "else:\n",
+ " new_band_structure_job_id = None\n",
+ " print(f\"♻️ Reusing Band Structure job {band_structure_job['_id']}\")\n",
+ "band_structure_job_id = band_structure_job[\"_id\"]"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "41",
+ "metadata": {},
+ "source": [
+ "### 8.3. Submit and monitor the job"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "40",
+ "id": "42",
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.notebooks_utils.api.job import wait_for_jobs_to_finish_async\n",
- "\n",
- "await wait_for_jobs_to_finish_async(client.jobs, [job_id], poll_interval=POLL_INTERVAL)"
+ "if new_band_structure_job_id:\n",
+ " client.jobs.submit(new_band_structure_job_id)\n",
+ " print(f\"✅ Job {new_band_structure_job_id} submitted successfully!\")\n",
+ "await wait_for_jobs_to_finish_async(client.jobs, [band_structure_job_id], poll_interval=POLL_INTERVAL)"
]
},
{
"cell_type": "markdown",
- "id": "41",
+ "id": "43",
"metadata": {},
"source": [
- "## 8. Retrieve results"
+ "## 9. Retrieve the results\n",
+ "### 9.1. Band structure"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "42",
+ "id": "44",
"metadata": {},
"outputs": [],
"source": [
- "from mat3ra.notebooks_utils.core.entity.property.api import get_properties_for_job\n",
"from mat3ra.notebooks_utils.ipython.entity.property.visualize import visualize_properties\n",
"\n",
- "property_data = get_properties_for_job(client, job_id, property_name=\"band_structure\")\n",
- "visualize_properties(property_data, title=\"Band Structure\", extra_config={\"material\": material.to_dict()})"
+ "band_structure_data = get_properties_for_job(client, band_structure_job_id, property_name=\"band_structure\")\n",
+ "visualize_properties(band_structure_data, title=\"Band Structure\",\n",
+ " extra_config={\"material\": defective_for_jobs.to_dict()})"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "45",
+ "metadata": {},
+ "source": [
+ "### 9.2. Formation energy"
]
},
{
"cell_type": "code",
"execution_count": null,
- "id": "43",
+ "id": "46",
"metadata": {},
"outputs": [],
- "source": []
+ "source": [
+ "defective_composition = Counter(defective_for_jobs.basis.elements.values)\n",
+ "mu_carbon = total_energies[PRISTINE_NAME] / pristine.basis.number_of_atoms\n",
+ "mu_nitrogen = total_energies[NITROGEN_NAME] / nitrogen.basis.number_of_atoms\n",
+ "e_formation = (\n",
+ " total_energies[DEFECTIVE_NAME]\n",
+ " - defective_composition[\"C\"] * mu_carbon\n",
+ " - defective_composition[\"N\"] * mu_nitrogen\n",
+ ")\n",
+ "\n",
+ "print(f\"μ_C = {mu_carbon:.4f} eV/atom ({pristine.basis.number_of_atoms} atoms)\")\n",
+ "print(f\"μ_N = {mu_nitrogen:.4f} eV/atom ({nitrogen.basis.number_of_atoms} atoms)\")\n",
+ "print(f\"E_f = {total_energies[DEFECTIVE_NAME]:.4f} eV - {defective_composition['C']} μ_C \"\n",
+ " f\"- {defective_composition['N']} μ_N = {e_formation:.3f} eV\")"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "47",
+ "metadata": {},
+ "source": [
+ "### 9.3. Compare with Fujimoto & Saito (2011)"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "48",
+ "metadata": {},
+ "outputs": [],
+ "source": [
+ "FUJIMOTO_FORMATION_ENERGY = 2.51 # eV, Table I, trimerized pyridine-type C28N3\n",
+ "TOLERANCE_FRACTION = 0.15\n",
+ "\n",
+ "deviation = e_formation - FUJIMOTO_FORMATION_ENERGY\n",
+ "verdict = \"yes\" if RELAX and abs(deviation) <= TOLERANCE_FRACTION * FUJIMOTO_FORMATION_ENERGY else \"no\"\n",
+ "print(f\"Regime: {RUN_TIER} tier\")\n",
+ "print(f\"E_f (this notebook): {e_formation:.3f} eV\")\n",
+ "print(f\"E_f (Fujimoto & Saito 2011, Table I): {FUJIMOTO_FORMATION_ENERGY:.3f} eV\")\n",
+ "print(f\"Deviation: {deviation:+.3f} eV ({100 * deviation / FUJIMOTO_FORMATION_ENERGY:+.1f} %, \"\n",
+ " f\"tolerance {100 * TOLERANCE_FRACTION:.0f} %)\")\n",
+ "print(\"Offsets: μ_N comes from solid N2 (mp-154) rather than the free molecule the manuscript \"\n",
+ " \"uses, and GBRV ultrasoft LDA stands in for its Troullier-Martins norm-conserving set.\")\n",
+ "print(f\"Reproduces Fujimoto & Saito (2011): {verdict} ({RUN_TIER} tier)\")"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "49",
+ "metadata": {},
+ "source": [
+ "## References\n",
+ "\n",
+ "[1] Y. Fujimoto and S. Saito, \"Formation, stabilities, and electronic properties of nitrogen defects in graphene\", Phys. Rev. B 84, 245446 (2011). https://doi.org/10.1103/PhysRevB.84.245446\n",
+ "\n",
+ "[2] K. F. Garrity, J. W. Bennett, K. M. Rabe and D. Vanderbilt, \"Pseudopotentials for high-throughput DFT calculations\", Comput. Mater. Sci. 81, 446 (2014). https://doi.org/10.1016/j.commatsci.2013.08.053"
+ ]
}
],
"metadata": {
diff --git a/src/py/mat3ra/notebooks_utils/core/entity/material/io.py b/src/py/mat3ra/notebooks_utils/core/entity/material/io.py
index 0a4cbe3f1..ef7136aa2 100644
--- a/src/py/mat3ra/notebooks_utils/core/entity/material/io.py
+++ b/src/py/mat3ra/notebooks_utils/core/entity/material/io.py
@@ -122,35 +122,52 @@ def load_materials_from_folder(folder_path: Optional[str] = None, verbose: bool
return materials
+def _get_name_match_rank(requested_name: str, material_name: str, filename_without_extension: str) -> Optional[int]:
+ """
+ Rank how well one folder entry matches a requested name: lower is better, None is no match.
+
+ An exact match wins over a substring match, and a case-sensitive match over a case-insensitive
+ one. Among exact matches the material name wins; among substring matches the filename wins.
+ """
+ requested_name_lower = requested_name.lower()
+ ranked_candidates = [
+ material_name == requested_name,
+ filename_without_extension == requested_name,
+ material_name.lower() == requested_name_lower,
+ filename_without_extension.lower() == requested_name_lower,
+ requested_name_lower in filename_without_extension.lower(),
+ requested_name_lower in material_name.lower(),
+ ]
+ return next((rank for rank, matches in enumerate(ranked_candidates) if matches), None)
+
+
def load_material_from_folder(folder_path: str, name: str, verbose: bool = True) -> Optional[Any]:
"""
- Load a single material from the specified folder by matching a substring of the name or filename.
+ Load a single material from the specified folder by matching the name or filename.
Args:
folder_path (str): The path to the folder containing material files.
- name (str): The substring to match against material names or filenames (case-insensitive).
+ name (str): The name to match against material names or filenames. An exact match is
+ preferred; otherwise a case-insensitive substring match is accepted.
verbose (bool): Whether to log verbose messages.
Returns:
- Optional[Material]: The first Material object that matches, or None if not found.
+ Optional[Material]: The best matching Material object, or None if not found.
"""
- name_lower = name.lower()
+ best_rank = None
resulting_material = None
for filename in sorted(os.listdir(folder_path)):
- if filename.endswith(".json") and name_lower in os.path.splitext(filename)[0].lower():
- with open(os.path.join(folder_path, filename), "r") as file:
- data = json.load(file)
- if MaterialWithBuildMetadata.is_valid(data):
- resulting_material = MaterialWithBuildMetadata.create(data)
- break
-
- if not resulting_material:
- materials = load_materials_from_folder(folder_path, verbose=verbose)
- for material in materials:
- if name_lower in material.name.lower():
- resulting_material = material
- break
+ if not filename.endswith(".json"):
+ continue
+ with open(os.path.join(folder_path, filename), "r") as file:
+ data = json.load(file)
+ if not MaterialWithBuildMetadata.is_valid(data):
+ continue
+ material = MaterialWithBuildMetadata.create(data)
+ rank = _get_name_match_rank(name, material.name, os.path.splitext(filename)[0])
+ if rank is not None and (best_rank is None or rank < best_rank):
+ best_rank, resulting_material = rank, material
if resulting_material:
log(f"Found: '{resulting_material.name}'", SeverityLevelEnum.INFO, force_verbose=verbose)
diff --git a/src/py/mat3ra/notebooks_utils/core/entity/property/api.py b/src/py/mat3ra/notebooks_utils/core/entity/property/api.py
index 050c16f4f..65e59feb7 100644
--- a/src/py/mat3ra/notebooks_utils/core/entity/property/api.py
+++ b/src/py/mat3ra/notebooks_utils/core/entity/property/api.py
@@ -71,7 +71,13 @@ def update_property_holder_value(client: APIClient, property_holder_id: str, val
return client.properties.update(property_holder_id, {"$set": {"data.value": value}})
-def find_total_energy_for_material(client: APIClient, material_id: str, source: str = "my_account") -> Optional[dict]:
+def find_total_energy_for_material(
+ client: APIClient,
+ material_id: str,
+ source: str = "my_account",
+ group: Optional[str] = None,
+ precision_value: Optional[float] = None,
+) -> Optional[dict]:
"""
Find the best-precision total_energy property for a material. Mirrors the
platform's "Resolve Total Energies for Elemental Materials" subworkflow,
@@ -86,6 +92,9 @@ def find_total_energy_for_material(client: APIClient, material_id: str, source:
material_id (str): Material _id to look up the total_energy property for.
source (str): Source of the total energy property: `my_account` (default), `curators` or
`public`.
+ group (str, optional): Model identifier the property must carry, e.g. `qe:dft:lda:pz`.
+ Without it a property computed with a different model is returned.
+ precision_value (float, optional): `precision.value` (KPPRA) the property must carry.
Returns:
The best-precision total_energy property, or None if none exists.
@@ -95,6 +104,10 @@ def find_total_energy_for_material(client: APIClient, material_id: str, source:
if not exabyte_id:
return None
query = {"exabyteId": exabyte_id, "slug": "total_energy"}
+ if group:
+ query["group"] = group
+ if precision_value is not None:
+ query["precision.value"] = precision_value
if source == "curators":
query["owner.slug"] = "curators"
elif source == "my_account":
diff --git a/tests/py/unit/core/entity/test_material_io.py b/tests/py/unit/core/entity/test_material_io.py
index 5771ca68d..51ebe827a 100644
--- a/tests/py/unit/core/entity/test_material_io.py
+++ b/tests/py/unit/core/entity/test_material_io.py
@@ -1,10 +1,33 @@
import json
+import pytest
from mat3ra.notebooks_utils.core.entity.material.io import load_material_from_folder, load_materials_from_folder
from mat3ra.standata.materials import Materials
RADII = {"Si": 1.11, "C": 0.76}
+GRAPHENE_STANDATA_NAME = "C, Graphene, HEX (P6/mmm) 2D (Monolayer), 2dm-3993"
+PRISTINE_NAME = "graphene 4x4"
+DEFECTIVE_NAME = "graphene 4x4 N3V pyridinic (C28N3)"
+RELAXED_NAME = "graphene 4x4 relaxed"
+
+# (filename without extension, material name) pairs; the relaxed material is stored under a
+# filename that is not its name.
+GRAPHENE_FOLDER = [
+ (GRAPHENE_STANDATA_NAME.replace("/", "-"), GRAPHENE_STANDATA_NAME),
+ (PRISTINE_NAME, PRISTINE_NAME),
+ (DEFECTIVE_NAME, DEFECTIVE_NAME),
+ ("pristine graphene", RELAXED_NAME),
+]
+RENAMED_FILE_FOLDER = [
+ (PRISTINE_NAME, RELAXED_NAME),
+ (DEFECTIVE_NAME, DEFECTIVE_NAME),
+]
+RECASED_NAME_FOLDER = [
+ (PRISTINE_NAME, PRISTINE_NAME.upper()),
+ ("n3v defect", PRISTINE_NAME),
+]
+
def _uploads_folder(tmp_path):
(tmp_path / "silicon.json").write_text(json.dumps(Materials.get_by_name_first_match("Silicon")))
@@ -18,3 +41,32 @@ def test_data_file_beside_a_material_is_skipped(tmp_path):
def test_lookup_by_name_works_alongside_a_data_file(tmp_path):
assert load_material_from_folder(_uploads_folder(tmp_path), "Silicon", verbose=False) is not None
+
+
+def _material_folder(tmp_path, files):
+ for filename, name in files:
+ graphene = {**Materials.get_by_name_first_match("Graphene"), "name": name}
+ (tmp_path / f"{filename}.json").write_text(json.dumps(graphene))
+ return str(tmp_path)
+
+
+@pytest.mark.parametrize(
+ ("files", "requested_name", "expected_name"),
+ [
+ (GRAPHENE_FOLDER, PRISTINE_NAME, PRISTINE_NAME),
+ (GRAPHENE_FOLDER, DEFECTIVE_NAME, DEFECTIVE_NAME),
+ (GRAPHENE_FOLDER, "GRAPHENE 4X4", PRISTINE_NAME),
+ (GRAPHENE_FOLDER, "N3V", DEFECTIVE_NAME),
+ (GRAPHENE_FOLDER, "pristine", RELAXED_NAME),
+ (GRAPHENE_FOLDER, "relaxed", RELAXED_NAME),
+ (GRAPHENE_FOLDER, "Graphene", GRAPHENE_STANDATA_NAME),
+ (GRAPHENE_FOLDER, "Germanium", None),
+ (RENAMED_FILE_FOLDER, PRISTINE_NAME, RELAXED_NAME),
+ (RENAMED_FILE_FOLDER, "GRAPHENE 4X4", RELAXED_NAME),
+ (RECASED_NAME_FOLDER, PRISTINE_NAME, PRISTINE_NAME),
+ ],
+)
+def test_load_material_from_folder(tmp_path, files, requested_name, expected_name):
+ material = load_material_from_folder(_material_folder(tmp_path, files), requested_name, verbose=False)
+
+ assert (material.name if material else None) == expected_name
diff --git a/tests/py/unit/core/entity/test_property_api.py b/tests/py/unit/core/entity/test_property_api.py
index 8dc5585c8..f1693bc32 100644
--- a/tests/py/unit/core/entity/test_property_api.py
+++ b/tests/py/unit/core/entity/test_property_api.py
@@ -9,6 +9,8 @@
EXABYTE_ID = "exabyte-a"
MATERIAL = {"_id": MATERIAL_ID, "exabyteId": EXABYTE_ID}
OWNER_ACCOUNT_ID = "account-a"
+GROUP = "qe:dft:lda:pz"
+PRECISION_VALUE = 1116
TOTAL_ENERGY_PROPERTY = {"data": {"value": -12.34}, "precision": {"value": 0.001}}
@@ -84,6 +86,23 @@ def test_find_total_energy_for_material_public_scope_has_no_owner_filter():
)
+def test_find_total_energy_for_material_constrains_group_and_precision():
+ client = _client()
+
+ find_total_energy_for_material(client, MATERIAL_ID, group=GROUP, precision_value=PRECISION_VALUE)
+
+ client.properties.list.assert_called_once_with(
+ query={
+ "exabyteId": EXABYTE_ID,
+ "slug": "total_energy",
+ "group": GROUP,
+ "precision.value": PRECISION_VALUE,
+ "owner._id": OWNER_ACCOUNT_ID,
+ },
+ projection={"sort": {"precision.value": -1}, "limit": 1},
+ )
+
+
def test_find_total_energy_for_material_rejects_invalid_source():
client = _client()