diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb index 4b0b8ecd5..f78e5f2b6 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene.ipynb @@ -222,7 +222,10 @@ "id": "16", "metadata": {}, "source": [ - "## 4. Write resulting material to the folder" + "## 4. Write resulting materials to the folder\n", + "\n", + "Both materials are saved to `uploads` under the names the simulation notebook loads by exact name:\n", + "`graphene 4x4` is the reference cell, `graphene 4x4 N3V pyridinic (C28N3)` is the defective one." ] }, { @@ -235,9 +238,10 @@ "from mat3ra.notebooks_utils.io import download_content_to_file\n", "from mat3ra.notebooks_utils.material import set_materials\n", "\n", - "material_with_defect.name = \"N-doped Graphene\"\n", - "set_materials(material_with_defect)\n", - "download_content_to_file(material_with_defect.to_json(), \"N-doped_Graphene.json\")" + "supercell.name = \"graphene 4x4\"\n", + "material_with_defect.name = \"graphene 4x4 N3V pyridinic (C28N3)\"\n", + "set_materials([supercell, material_with_defect])\n", + "download_content_to_file(material_with_defect.to_json(), \"graphene_4x4_N3V_pyridinic.json\")" ] } ], diff --git a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb index 4902ba679..f662193de 100644 --- a/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb +++ b/other/materials_designer/specific_examples/defect_point_substitution_graphene_SIMULATION.ipynb @@ -5,31 +5,35 @@ "id": "0", "metadata": {}, "source": [ - "# Calculating Band Structure of N-doped Graphene.\n", + "# Formation Energy and Band Structure of the Pyridinic N₃-Vacancy Defect in Graphene\n", "\n", - "> **Yoshitaka Fujimoto and Susumu Saito**, \"Formation, stabilities, and electronic properties of nitrogen defects in graphene\", Physical Review B, 2011. [DOI: 10.1103/PhysRevB.84.245446](https://journals.aps.org/prb/abstract/10.1103/PhysRevB.84.245446).\n", + "> **Yoshitaka Fujimoto and Susumu Saito**, \"Formation, stabilities, and electronic properties of\n", + "> nitrogen defects in graphene\", Physical Review B 84, 245446 (2011).\n", + "> [DOI:10.1103/PhysRevB.84.245446](https://doi.org/10.1103/PhysRevB.84.245446)\n", "\n", - "\n", - "Calculate the band structure of material created in the specialized notebook using Quantum ESPRESSO application and the band structure workflow available in Standata.\n", + "Computes the formation energy and the band structure of the trimerized pyridine-type C₂₈N₃ defect\n", + "created in the [structure notebook](defect_point_substitution_graphene.ipynb), both off the same\n", + "cell, and compares the formation energy with the manuscript's Table I entry, 2.51 eV.\n", "\n", "

Usage

\n", "\n", - "1. Create the material in the [defect creation notebook](defect_point_substitution_graphene.ipynb) or place the material file in `../uploads` folder.\n", - "2. Set material and calculation parameters in cell 1.2. below (or use the default values). Choose between \"debug\" and \"production\" modes using `RUN_PROFILE` variable.\n", + "1. Create the materials in the [structure notebook](defect_point_substitution_graphene.ipynb), which saves `graphene 4x4` and `graphene 4x4 N3V pyridinic (C28N3)` to the `uploads` folder.\n", + "1. Set the material names and parameters in cells 1.2-1.4 (or use the defaults); `RUN_TIER = \"production\"` runs the manuscript's k-grid and relaxes C₂₈N₃ once, saving it as ` relaxed` for reuse by this and other notebooks.\n", "1. Click \"Run\" > \"Run All\" to run all cells.\n", - "1. Wait for the job to complete.\n", - "1. Scroll down to view the result.\n", + "1. Wait for the jobs to complete.\n", + "1. Scroll down to view the results.\n", "\n", "## Summary\n", "\n", - "1. Set up the environment and parameters: install packages (JupyterLite only) and configure parameters for material, workflow, compute resources, and job.\n", + "1. Set up the environment and parameters: install packages (JupyterLite only) and configure the material names, model and compute parameters.\n", "1. Authenticate and initialize API client: authenticate via browser, initialize the client, then select account and project.\n", - "1. Create material: materials are read from the `../uploads` folder — place files there manually or run a [material creation notebook](defect_point_substitution_graphene.ipynb) first. If the material is not found by name, Standata is used as a fallback. The material is then saved to the platform.\n", - "1. Create workflow and set its parameters: select application, load total energy workflow from Standata, optionally add relaxation or adjust model/method parameters, and save the workflow to the platform.\n", - "1. Configure compute: get list of clusters and create compute configuration with selected cluster, queue, and number of processors.\n", - "1. Create the job with material and workflow configuration: assemble the job from material, workflow, project, and compute configuration.\n", - "1. Submit the job and monitor the status: submit the job and wait for completion.\n", - "1. Retrieve results: get and display the properties." + "1. Load the materials by name from the `uploads` folder, resolve the nitrogen reference, and save them to the platform.\n", + "1. Configure the shared DFT model and k-grid: one model and a per-material k-grid for every job below.\n", + "1. Configure compute: get the list of clusters and create a compute configuration.\n", + "1. Relax the defective cell on the `production` tier, reusing a saved relaxed structure when one already exists.\n", + "1. Create the missing Total Energy jobs for the defective, pristine and nitrogen cells and wait for them.\n", + "1. Configure the Band Structure workflow and run it on the defective cell.\n", + "1. Retrieve the band structure, assemble the formation energy and compare it with the manuscript." ] }, { @@ -50,7 +54,7 @@ "source": [ "from mat3ra.notebooks_utils.packages import install_packages\n", "\n", - "await install_packages(\"api_examples\")" + "await install_packages(\"made|specific_examples|api_examples\")" ] }, { @@ -58,14 +62,7 @@ "id": "3", "metadata": {}, "source": [ - "### 1.2. Set parameters and configurations for the workflow and job\n", - "\n", - "Use `RUN_PROFILE` in the next cell to switch between two predefined modes:\n", - "\n", - "- `\"debug\"`: quick validation run (small grids/path, no relaxation) to finish in a few minutes.\n", - "- `\"production\"`: paper-like settings (relaxation + denser sampling) for final-quality results; this can take much longer.\n", - "\n", - "For first-time use, start with `\"debug\"`. Once everything works, switch to `\"production\"` and rerun the notebook." + "### 1.2. Material names" ] }, { @@ -74,100 +71,130 @@ "id": "4", "metadata": {}, "outputs": [], + "source": [ + "# Names saved by defect_point_substitution_graphene.ipynb.\n", + "PRISTINE_NAME = \"graphene 4x4\"\n", + "DEFECTIVE_NAME = \"graphene 4x4 N3V pyridinic (C28N3)\"\n", + "# Standata's solid nitrogen -- the platform carries no free N2 molecule.\n", + "NITROGEN_NAME = \"N2, Nitrogen, FCC (P2_13) 3D (Bulk), mp-154\"" + ] + }, + { + "cell_type": "markdown", + "id": "5", + "metadata": {}, + "source": [ + "### 1.3. Parameters\n", + "\n", + "`RUN_TIER = \"fast\"` runs end to end in minutes; `\"production\"` takes about 1.5 hours from cold, which outlives the browser session — every job here is looked up by name before it is created, so signing in again and re-running attaches to the jobs already on the cluster instead of resubmitting them." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "6", + "metadata": {}, + "outputs": [], "source": [ "from datetime import datetime\n", "from mat3ra.ide.compute import QueueName\n", "\n", - "# 2. Auth and organization parameters\n", "ORGANIZATION_NAME = None # set to your organization name (full or partial); otherwise, your default one is used\n", - "\n", - "# 3. Material parameters\n", "FOLDER = \"./uploads\"\n", - "MATERIAL_NAME = \"N-doped Graphene\"\n", "\n", - "# 4. Workflow parameters\n", - "WORKFLOW_SEARCH_TERM = \"band_structure.json\"\n", - "RUN_PROFILE = \"debug\" # Change to \"production\" for paper-quality results\n", + "TOTAL_ENERGY_SEARCH_TERM = \"total_energy.json\"\n", + "RELAX_WORKFLOW_SEARCH_TERM = \"fixed_cell_relaxation.json\"\n", + "BAND_STRUCTURE_WORKFLOW_SEARCH_TERM = \"band_structure.json\"\n", "MY_WORKFLOW_NAME = \"Band Structure\"\n", "APPLICATION_NAME = \"espresso\"\n", "\n", - "# DFT model (same for both profiles)\n", - "MODEL_SUBTYPE = \"lda\"\n", - "\n", - "# 5. Compute parameters\n", - "CLUSTER_NAME = \"101\" # specify full or partial name i.e. \"cluster-101\" to select\n", - "QUEUE_NAME = QueueName.D\n", - "PPN = 1\n", + "RUN_TIER = \"fast\" # minutes; coarse k-grid, no relaxation — will NOT match the manuscript\n", + "# RUN_TIER = \"production\" # ~1.5 h; 6x6x1 and relaxation to 0.05 eV/Å — reproduces Table I\n", + "TIER_SETTINGS = {\n", + " \"fast\": {\"kpoint_density\": 3.5, \"relax\": False},\n", + " \"production\": {\"kpoint_density\": 7, \"relax\": True},\n", + "}\n", + "# Points per Å⁻¹: 3.5 gives 3 x 3 x 1 on the 4x4 cell, 7 gives the manuscript's 6 x 6 x 1.\n", + "KPOINT_DENSITY = TIER_SETTINGS[RUN_TIER][\"kpoint_density\"]\n", + "RELAX = TIER_SETTINGS[RUN_TIER][\"relax\"]\n", + "\n", + "CLUSTER_NAME = \"001\" # specify full or partial name i.e. \"cluster-001\" to select\n", + "QUEUE_NAME = QueueName.OR\n", + "PPN = 16 # queue OR on cluster-001 allows at most 16 cores per node\n", + "TIME_LIMIT = \"04:00:00\" # covers the optional relaxation (~1 h); walltime x ppn is reserved up front\n", "\n", - "# 6. Job parameters\n", "timestamp = datetime.now().strftime(\"%Y-%m-%d %H:%M\")\n", - "JOB_NAME = MY_WORKFLOW_NAME + \" \" + MATERIAL_NAME + f\" ({RUN_PROFILE}) \" + timestamp\n", - "POLL_INTERVAL = 60 # seconds\n" + "POLL_INTERVAL = 60 # seconds" ] }, { "cell_type": "markdown", - "id": "5", + "id": "7", "metadata": {}, "source": [ - "### 1.3. Set specific parameters" + "### 1.4. DFT model parameters" ] }, { "cell_type": "code", "execution_count": null, - "id": "6", + "id": "8", "metadata": {}, "outputs": [], "source": [ - "# Parameters for the Method Selected\n", - "PSEUDOPOTENTIAL_TYPE = \"nc\"\n", + "# NOTE: the manuscript's value needs its own settings — Troullier-Martins norm-conserving LDA;\n", + "# its 4x4 cell and 50 Ry cutoff are matched below, its 6x6x1 k-grid on the production tier,\n", + "# its pseudopotentials not at all.\n", + "MODEL_SUBTYPE = \"lda\"\n", "FUNCTIONAL = \"pz\"\n", - "\n", - "# Energy cutoffs (check per the ONCV guidelines - https://www.pseudo-dojo.org/)\n", - "ECUTWFC = 50\n", - "# Density cutoff in relation to wavefunction cutoff\n", - "ECUTRHO = 4 * ECUTWFC\n", - "\n", - "# Parameters for the specific property\n", - "if RUN_PROFILE == \"debug\":\n", - " # Quick validation run (finishes in a few minutes)\n", - " ADD_RELAXATION = False\n", - " RELAXATION_KGRID = [1, 1, 1]\n", - " SCF_KGRID = [1, 1, 1]\n", - " KPATH = [\n", - " {\"point\": \"K\", \"steps\": 6},\n", - " {\"point\": \"Γ\", \"steps\": 6},\n", - " {\"point\": \"M\", \"steps\": 6},\n", - " {\"point\": \"K\", \"steps\": 1},\n", - " ]\n", - "elif RUN_PROFILE == \"production\":\n", - " # Paper settings (Fujimoto & Saito 2011): relaxation + dense sampling\n", - " ADD_RELAXATION = True\n", - " RELAXATION_KGRID = [6, 6, 1]\n", - " SCF_KGRID = [6, 6, 1]\n", - " KPATH = [\n", - " {\"point\": \"K\", \"steps\": 20},\n", - " {\"point\": \"Γ\", \"steps\": 20},\n", - " {\"point\": \"M\", \"steps\": 20},\n", - " {\"point\": \"K\", \"steps\": 1},\n", - " ]" + "PSEUDOPOTENTIAL_TYPE = \"us\" # GBRV ultrasoft, the only LDA family the platform publishes for C and N\n", + "ECUTWFC = 50 # Ry, Fujimoto & Saito Sec. II\n", + "ECUTRHO = 200 # Ry, GBRV's recommended charge-density cutoff\n", + "\n", + "MODEL_TAG = f\"{FUNCTIONAL}-{PSEUDOPOTENTIAL_TYPE} {ECUTWFC}-{ECUTRHO}Ry k{KPOINT_DENSITY}\"\n", + "\n", + "SCF_UNIT = \"pw_scf\"\n", + "BANDS_UNIT = \"pw_bands\"\n", + "RELAX_UNIT = \"pw_relax\"\n", + "# C28N3 carries 127 valence electrons, so its ground state is a doublet.\n", + "SPIN_SETTINGS = {\n", + " DEFECTIVE_NAME: {\"nspin\": 2, \"tot_magnetization\": 1},\n", + " PRISTINE_NAME: {\"nspin\": 1},\n", + " NITROGEN_NAME: {\"nspin\": 1},\n", + "}\n", + "RELAXATION_SETTINGS = {\"forc_conv_thr\": 1.9e-3, \"nstep\": 100} # 0.05 eV/Å, Fujimoto & Saito Sec. II\n", + "# Names the relaxation job; the relaxed structure itself is found by content hash.\n", + "RELAX_TAG = f\"{MODEL_TAG} f{RELAXATION_SETTINGS['forc_conv_thr']}\"\n", + "\n", + "# Whose total_energy properties to reuse when this notebook has no job of its own for a material:\n", + "# \"public\" (any owner), \"curators\" (only curators'), or \"my_account\" (only your own).\n", + "TOTAL_ENERGY_SOURCE = \"public\" # so a first run reuses references someone has already computed\n", + "# The model identifier a reusable property must carry. It stops at the functional, so it pins\n", + "# neither the pseudopotential family nor the cutoffs; the k-point density is matched separately.\n", + "MODEL_GROUP = f\"qe:dft:{MODEL_SUBTYPE}:{FUNCTIONAL}\"\n", + "\n", + "KPATH_STEPS = 20\n", + "KPATH = [\n", + " {\"point\": \"K\", \"steps\": KPATH_STEPS},\n", + " {\"point\": \"Γ\", \"steps\": KPATH_STEPS},\n", + " {\"point\": \"M\", \"steps\": KPATH_STEPS},\n", + " {\"point\": \"K\", \"steps\": 1},\n", + "]" ] }, { "cell_type": "markdown", - "id": "7", + "id": "9", "metadata": {}, "source": [ "## 2. Authenticate and initialize API client\n", - "### 2.1. Authenticate\n", - "Authenticate in the browser and have credentials stored in environment variable \"OIDC_ACCESS_TOKEN\".\n" + "### 2.1. Authenticate" ] }, { "cell_type": "code", "execution_count": null, - "id": "8", + "id": "10", "metadata": {}, "outputs": [], "source": [ @@ -178,16 +205,16 @@ }, { "cell_type": "markdown", - "id": "9", + "id": "11", "metadata": {}, "source": [ - "### 2.2. Initialize API Client\n" + "### 2.2. Initialize API client" ] }, { "cell_type": "code", "execution_count": null, - "id": "10", + "id": "12", "metadata": {}, "outputs": [], "source": [ @@ -199,16 +226,16 @@ }, { "cell_type": "markdown", - "id": "11", + "id": "13", "metadata": {}, "source": [ - "### 2.3. Select account to work under" + "### 2.3. Select account" ] }, { "cell_type": "code", "execution_count": null, - "id": "12", + "id": "14", "metadata": {}, "outputs": [], "source": [ @@ -218,7 +245,7 @@ { "cell_type": "code", "execution_count": null, - "id": "13", + "id": "15", "metadata": {}, "outputs": [], "source": [ @@ -233,7 +260,7 @@ }, { "cell_type": "markdown", - "id": "14", + "id": "16", "metadata": {}, "source": [ "### 2.4. Select project" @@ -242,7 +269,7 @@ { "cell_type": "code", "execution_count": null, - "id": "15", + "id": "17", "metadata": {}, "outputs": [], "source": [ @@ -251,41 +278,13 @@ "print(f\"✅ Using project: {projects[0]['name']} ({project_id})\")" ] }, - { - "cell_type": "markdown", - "id": "16", - "metadata": {}, - "source": [ - "## 3. Create material\n", - "### 3.1. Load material from local file (or Standata)" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "id": "17", - "metadata": {}, - "outputs": [], - "source": [ - "from mat3ra.made.material import Material\n", - "from mat3ra.standata.materials import Materials\n", - "from mat3ra.notebooks_utils.ipython.entity.material.visualize import visualize_materials as visualize\n", - "from mat3ra.notebooks_utils.material import load_material_from_folder\n", - "\n", - "material = load_material_from_folder(FOLDER, MATERIAL_NAME) or Material.create(\n", - " Materials.get_by_name_first_match(MATERIAL_NAME))\n", - "\n", - "# set lattice for the correct display of Point Path for this material\n", - "material.lattice.type = \"HEX\"\n", - "visualize(material)" - ] - }, { "cell_type": "markdown", "id": "18", "metadata": {}, "source": [ - "### 3.2. Save material to the platform" + "## 3. Load the materials\n", + "### 3.1. Load from the uploads folder or the platform, and print provenance" ] }, { @@ -295,10 +294,15 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.core.entity.material.api import get_or_create_material\n", + "from collections import Counter\n", + "from mat3ra.notebooks_utils.core.entity.material.api import load_material\n", "\n", - "saved_material_response = get_or_create_material(client, material, ACCOUNT_ID)\n", - "saved_material = Material.create(saved_material_response)" + "materials_by_name = {name: load_material(client, FOLDER, name, ACCOUNT_ID) for name in (PRISTINE_NAME, DEFECTIVE_NAME)}\n", + "pristine, defective = materials_by_name[PRISTINE_NAME], materials_by_name[DEFECTIVE_NAME]\n", + "for name, material in materials_by_name.items():\n", + " composition = \"\".join(f\"{e}{n}\" for e, n in sorted(Counter(material.basis.elements.values).items()))\n", + " print(f\"{name}: {composition}, {material.basis.number_of_atoms} atoms, \"\n", + " f\"cell {material.lattice.a:.2f} x {material.lattice.b:.2f} x {material.lattice.c:.2f} Å\")" ] }, { @@ -306,8 +310,7 @@ "id": "20", "metadata": {}, "source": [ - "## 4. Create workflow and set its parameters\n", - "### 4.1. Get list of applications and select one" + "### 3.2. Resolve the nitrogen reference material" ] }, { @@ -317,12 +320,11 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.standata.applications import ApplicationStandata\n", - "from mat3ra.ade.application import Application\n", + "from mat3ra.made.material import Material\n", + "from mat3ra.standata.materials import Materials\n", "\n", - "app_config = ApplicationStandata.get_by_name_first_match(APPLICATION_NAME)\n", - "app = Application(**app_config)\n", - "print(f\"Using application: {app.name}\")" + "nitrogen = Material.create(Materials.get_by_name_first_match(NITROGEN_NAME))\n", + "print(f\"{nitrogen.name}: {nitrogen.basis.number_of_atoms} atoms, cell {nitrogen.lattice.a:.2f} Å\")" ] }, { @@ -330,7 +332,7 @@ "id": "22", "metadata": {}, "source": [ - "### 4.2. Create workflow from standard workflows and preview it" + "### 3.3. Save the materials to the platform" ] }, { @@ -340,15 +342,11 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.standata.workflows import WorkflowStandata\n", - "from mat3ra.wode.workflows import Workflow\n", - "from mat3ra.notebooks_utils.ipython.entity.workflow.visualize import visualize_workflow\n", - "\n", - "workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(WORKFLOW_SEARCH_TERM)\n", - "workflow = Workflow.create(workflow_config)\n", - "workflow.name = MY_WORKFLOW_NAME\n", + "from mat3ra.notebooks_utils.core.entity.material.api import get_or_create_material\n", "\n", - "visualize_workflow(workflow)" + "saved_pristine = get_or_create_material(client, pristine, ACCOUNT_ID)\n", + "saved_defective = get_or_create_material(client, defective, ACCOUNT_ID)\n", + "saved_nitrogen = get_or_create_material(client, nitrogen, ACCOUNT_ID)" ] }, { @@ -356,8 +354,8 @@ "id": "24", "metadata": {}, "source": [ - "### 4.3. Modify workflow (Optional)\n", - "#### 4.3.1. Add relaxation" + "## 4. Configure the shared DFT model and k-grid\n", + "### 4.1. DFT model" ] }, { @@ -367,8 +365,18 @@ "metadata": {}, "outputs": [], "source": [ - "if ADD_RELAXATION:\n", - " workflow.add_relaxation()\n" + "from mat3ra.standata.applications import ApplicationStandata\n", + "from mat3ra.standata.model_tree import ModelTreeStandata\n", + "from mat3ra.ade.application import Application\n", + "from mat3ra.mode import ModelFactory\n", + "\n", + "app_config = ApplicationStandata.get_by_name_first_match(APPLICATION_NAME)\n", + "app = Application(**app_config)\n", + "\n", + "model_config = ModelTreeStandata.get_model_by_parameters(type=\"dft\", subtype=MODEL_SUBTYPE, functional=FUNCTIONAL)\n", + "model_config[\"method\"] = {\"type\": \"pseudopotential\", \"subtype\": PSEUDOPOTENTIAL_TYPE}\n", + "model = ModelFactory.create(model_config)\n", + "print(f\"Using application: {app.name}, model: {MODEL_TAG}\")" ] }, { @@ -376,8 +384,7 @@ "id": "26", "metadata": {}, "source": [ - "#### 4.3.2. Modify model and method parameters (Optional)\n", - "Uncomment the code below and adjust selection of model parameters as needed." + "### 4.2. k-grid per material" ] }, { @@ -387,19 +394,19 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.standata.model_tree import ModelTreeStandata\n", - "from mat3ra.mode.model import Model\n", - "\n", - "model_config = ModelTreeStandata.get_model_by_parameters(\n", - " type=\"dft\", subtype=MODEL_SUBTYPE, functional=FUNCTIONAL\n", - ")\n", - "model_config[\"method\"] = {\"type\": \"pseudopotential\", \"subtype\": PSEUDOPOTENTIAL_TYPE}\n", - "model = Model.create(model_config)\n", + "import math\n", "\n", - "for subworkflow in workflow.subworkflows:\n", - " subworkflow.model = model\n", + "from mat3ra.notebooks_utils.workflow import kgrid_from_density\n", "\n", - "visualize_workflow(workflow)" + "material_objects = {PRISTINE_NAME: pristine, DEFECTIVE_NAME: defective, NITROGEN_NAME: nitrogen}\n", + "kgrid = {\n", + " PRISTINE_NAME: kgrid_from_density(pristine, KPOINT_DENSITY, periodic_dims=(0, 1)),\n", + " DEFECTIVE_NAME: kgrid_from_density(defective, KPOINT_DENSITY, periodic_dims=(0, 1)),\n", + " NITROGEN_NAME: kgrid_from_density(nitrogen, KPOINT_DENSITY),\n", + "}\n", + "kppra = {name: math.prod(grid) * material_objects[name].basis.number_of_atoms for name, grid in kgrid.items()}\n", + "for name, grid in kgrid.items():\n", + " print(f\"{name}: k-grid {grid}, KPPRA {kppra[name]}\")" ] }, { @@ -407,8 +414,8 @@ "id": "28", "metadata": {}, "source": [ - "#### 4.3.3. Modify important settings\n", - "Set k-grid, k-path and energy cutoffs, as well as other settings." + "## 5. Create the compute configuration\n", + "### 5.1. Select cluster" ] }, { @@ -418,48 +425,8 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.wode.context.providers import PlanewaveCutoffsContextProvider, PointsGridDataProvider, \\\n", - " PointsPathDataProvider\n", - "\n", - "bs_subworkflow = workflow.subworkflows[1 if ADD_RELAXATION else 0]\n", - "cutoffs_context = PlanewaveCutoffsContextProvider(wavefunction=ECUTWFC, density=ECUTRHO, isEdited=True).get_context_item_data()\n", - "\n", - "if RELAXATION_KGRID is not None and ADD_RELAXATION:\n", - " unit = workflow.subworkflows[0].get_unit_by_name(name_regex=\"relax\")\n", - " unit.add_context(PointsGridDataProvider(material=material, dimensions=RELAXATION_KGRID, isEdited=True).get_context_item_data())\n", - " workflow.subworkflows[0].set_unit(unit)\n", - "\n", - "if SCF_KGRID is not None:\n", - " unit = bs_subworkflow.get_unit_by_name(name=\"pw_scf\")\n", - " unit.add_context(PointsGridDataProvider(material=material, dimensions=SCF_KGRID, isEdited=True).get_context_item_data())\n", - " bs_subworkflow.set_unit(unit)\n", - "\n", - "if KPATH is not None:\n", - " unit = bs_subworkflow.get_unit_by_name(name=\"pw_bands\")\n", - " unit.add_context(PointsPathDataProvider(path=KPATH, isEdited=True).get_context_item_data())\n", - " bs_subworkflow.set_unit(unit)\n", - "\n", - "for unit_name in [\"pw_relax\", \"pw_vc-relax\", \"pw_scf\", \"pw_bands\"]:\n", - " for swf in workflow.subworkflows:\n", - " unit = swf.get_unit_by_name(name=unit_name)\n", - " if unit:\n", - " unit.add_context(cutoffs_context)\n", - " swf.set_unit(unit)\n", - "\n", - "# Below is the example of how to modify the input file content directly.\n", - "# bands_unit = bs_subworkflow.get_unit_by_name(name=\"bands\")\n", - "# bands_unit.input = [{\"name\": \"bands.in\", \"content\": (\n", - "# \"&BANDS\\n\"\n", - "# \" prefix = '__prefix__'\\n\"\n", - "# \" lsym = .false.\\n\"\n", - "# \" outdir = {% raw %}'{{ JOB_WORK_DIR }}/outdir'{% endraw %}\\n\"\n", - "# \" filband = {% raw %}'{{ JOB_WORK_DIR }}/bands.dat'{% endraw %}\\n\"\n", - "# \" no_overlap = .true.\\n\"\n", - "# \"/\"\n", - "# )}]\n", - "# bs_subworkflow.set_unit(bands_unit)\n", - "\n", - "visualize_workflow(workflow)" + "clusters = client.clusters.list()\n", + "print(f\"Available clusters: {[c['hostname'] for c in clusters]}\")" ] }, { @@ -467,7 +434,7 @@ "id": "30", "metadata": {}, "source": [ - "### 4.4. Save workflow to collection" + "### 5.2. Create the compute configuration for the jobs" ] }, { @@ -477,12 +444,17 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.utils.namespace import dict_to_namespace_recursive\n", - "from mat3ra.notebooks_utils.core.entity.workflow.api import get_or_create_workflow\n", + "from mat3ra.ide.compute import Compute\n", "\n", - "saved_workflow_response = get_or_create_workflow(client, workflow, ACCOUNT_ID)\n", - "saved_workflow = Workflow.create(saved_workflow_response)\n", - "print(f\"Workflow ID: {saved_workflow.id}\")" + "if CLUSTER_NAME:\n", + " cluster = next((c for c in clusters if CLUSTER_NAME in c[\"hostname\"]), None)\n", + " if cluster is None:\n", + " raise ValueError(f\"Cluster '{CLUSTER_NAME}' not found. Available: {[c['hostname'] for c in clusters]}\")\n", + "else:\n", + " cluster = clusters[0]\n", + "compute = Compute(cluster=cluster, queue=QUEUE_NAME, ppn=PPN, timeLimit=TIME_LIMIT)\n", + "print(f\"Using cluster: {compute.cluster.hostname}, queue: {QUEUE_NAME}, ppn: {PPN}, \"\n", + " f\"time limit: {TIME_LIMIT}\")" ] }, { @@ -490,8 +462,13 @@ "id": "32", "metadata": {}, "source": [ - "## 5. Create the compute configuration\n", - "### 5.1. Get list of clusters" + "## 6. Relax the defective cell (optional)\n", + "\n", + "Runs only on the `production` tier: finds any relaxed version of this structure already on the account,\n", + "regardless of who relaxed it or with what -- before spending any compute on a new one. The band\n", + "structure and the total energy are then both taken off that one geometry. In Fujimoto & Saito the\n", + "pyridinic C-N bond contracts from 1.41 Å to 1.33 Å and the acceptor-like states near E_F belong to\n", + "the relaxed geometry, so an unrelaxed run is a different system rather than a rougher answer." ] }, { @@ -501,8 +478,58 @@ "metadata": {}, "outputs": [], "source": [ - "clusters = client.clusters.list()\n", - "print(f\"Available clusters: {[c['hostname'] for c in clusters]}\")" + "from mat3ra.standata.workflows import WorkflowStandata\n", + "from mat3ra.wode.workflows import Workflow\n", + "from mat3ra.notebooks_utils.workflow import apply_planewave_cutoffs, apply_scf_kgrid, patch_workflow_qe_input\n", + "from mat3ra.notebooks_utils.core.entity.job.api import find_job_for_material\n", + "from mat3ra.notebooks_utils.core.entity.material.api import find_relaxed_material, get_final_structure_for_job\n", + "from mat3ra.notebooks_utils.core.entity.property.api import get_properties_for_job\n", + "from mat3ra.notebooks_utils.job import create_job\n", + "from mat3ra.notebooks_utils.api.job import submit_jobs, wait_for_jobs_to_finish_async\n", + "\n", + "relaxed_defective = None\n", + "if RELAX:\n", + " relax_workflow_name = f\"Fixed-cell Relaxation {DEFECTIVE_NAME} {RELAX_TAG}\"\n", + " relaxed_defective = find_relaxed_material(client, defective, ACCOUNT_ID)\n", + " if relaxed_defective is not None:\n", + " print(f\"♻️ Relaxed defective material: {relaxed_defective.name} ({relaxed_defective.id})\")\n", + " else:\n", + " relax_workflow = Workflow.create(\n", + " WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(RELAX_WORKFLOW_SEARCH_TERM)\n", + " )\n", + " relax_workflow.name = relax_workflow_name\n", + " relax_workflow.subworkflows[0].model = model\n", + " apply_planewave_cutoffs(relax_workflow, ECUTWFC, ECUTRHO, unit_name=RELAX_UNIT)\n", + " apply_scf_kgrid(relax_workflow, kgrid[DEFECTIVE_NAME], material=defective, unit_name=RELAX_UNIT)\n", + " patch_workflow_qe_input(relax_workflow, {\"system\": SPIN_SETTINGS[DEFECTIVE_NAME]}, [RELAX_UNIT])\n", + " patch_workflow_qe_input(relax_workflow, {\"control\": RELAXATION_SETTINGS}, [RELAX_UNIT])\n", + "\n", + " relax_job = find_job_for_material(\n", + " client, saved_defective[\"_id\"], relax_workflow_name, ACCOUNT_ID,\n", + " statuses=(\"submitted\", \"queued\", \"active\", \"finished\"),\n", + " )\n", + " if relax_job is None:\n", + " relax_job = create_job(\n", + " api_client=client, materials=[saved_defective], workflow=relax_workflow,\n", + " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n", + " prefix=f\"{relax_workflow_name} {timestamp}\",\n", + " )\n", + " submit_jobs(client.jobs, [relax_job[\"_id\"]])\n", + " await wait_for_jobs_to_finish_async(client.jobs, [relax_job[\"_id\"]], poll_interval=POLL_INTERVAL)\n", + "\n", + " relaxed_material = get_final_structure_for_job(client, relax_job[\"_id\"])\n", + " client.materials.update(relaxed_material.id, {\"name\": f\"{DEFECTIVE_NAME} relaxed\"})\n", + " relaxed_defective = Material.create(client.materials.get(relaxed_material.id))\n", + " print(f\"✅ Relaxed defective material: {relaxed_defective.name} ({relaxed_defective.id})\")\n", + "\n", + " total_force = get_properties_for_job(client, relax_job[\"_id\"], \"total_force\")[0]\n", + " print(f\"Residual force after relaxation (norm over all atoms): \"\n", + " f\"{total_force['value']:.4f} {total_force['units']}\")\n", + "\n", + "defective_for_jobs = relaxed_defective if RELAX else defective\n", + "defective_for_jobs.lattice.type = \"HEX\" # the symbolic K-Γ-M-K path needs a named lattice type\n", + "material_objects[DEFECTIVE_NAME] = defective_for_jobs\n", + "saved_defective_for_jobs = get_or_create_material(client, defective_for_jobs, ACCOUNT_ID)" ] }, { @@ -510,7 +537,12 @@ "id": "34", "metadata": {}, "source": [ - "### 5.2. Create compute configuration for the job\n" + "## 7. Total Energy jobs for the defective, pristine and nitrogen cells\n", + "\n", + "A total energy already on the platform is reused when it comes from this notebook's own job for the\n", + "material, or, failing that, from a `TOTAL_ENERGY_SOURCE` property computed with the same model and\n", + "the same k-point density. The jobs that remain are submitted one at a time: the account's\n", + "concurrency limit rejects a batch whenever another job of yours is already running." ] }, { @@ -520,122 +552,249 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.ide.compute import Compute\n", - "\n", - "# Select cluster: use specified name if provided, otherwise use first available\n", - "if CLUSTER_NAME:\n", - " cluster = next((c for c in clusters if CLUSTER_NAME in c[\"hostname\"]), None)\n", - "else:\n", - " cluster = clusters[0]\n", + "from mat3ra.notebooks_utils.core.entity.property.api import find_total_energy_for_material\n", "\n", - "compute = Compute(\n", - " cluster=cluster,\n", - " queue=QUEUE_NAME,\n", - " ppn=PPN\n", + "total_energy_workflow_config = WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(\n", + " TOTAL_ENERGY_SEARCH_TERM\n", ")\n", - "print(f\"Using cluster: {compute.cluster.hostname}, queue: {QUEUE_NAME}, ppn: {PPN}\")" + "saved_materials = {\n", + " DEFECTIVE_NAME: saved_defective_for_jobs,\n", + " PRISTINE_NAME: saved_pristine,\n", + " NITROGEN_NAME: saved_nitrogen,\n", + "}\n", + "total_energies = {}\n", + "total_energy_job_ids = {}\n", + "for name, saved_material in saved_materials.items():\n", + " workflow_name = f\"Total Energy {name} {MODEL_TAG}\"\n", + " existing_job = find_job_for_material(client, saved_material[\"_id\"], workflow_name, ACCOUNT_ID)\n", + " if existing_job is not None:\n", + " total_energies[name] = get_properties_for_job(client, existing_job[\"_id\"], \"total_energy\")[0][\"value\"]\n", + " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV\")\n", + " continue\n", + " existing_property = find_total_energy_for_material(\n", + " client, saved_material[\"_id\"], source=TOTAL_ENERGY_SOURCE,\n", + " group=MODEL_GROUP, precision_value=kppra[name],\n", + " )\n", + " if existing_property is not None:\n", + " total_energies[name] = existing_property[\"data\"][\"value\"]\n", + " print(f\"♻️ {name}: reusing total energy {total_energies[name]:.4f} eV \"\n", + " f\"from {existing_property['owner']['slug']}\")\n", + " continue\n", + " workflow = Workflow.create(total_energy_workflow_config)\n", + " workflow.name = workflow_name\n", + " workflow.subworkflows[0].model = model\n", + " apply_planewave_cutoffs(workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n", + " apply_scf_kgrid(workflow, kgrid[name], material=material_objects[name])\n", + " patch_workflow_qe_input(workflow, {\"system\": SPIN_SETTINGS[name]}, [SCF_UNIT])\n", + " job = create_job(\n", + " api_client=client, materials=[saved_material], workflow=workflow, project_id=project_id,\n", + " owner_id=ACCOUNT_ID, compute=compute.to_dict(), prefix=f\"{workflow.name} {timestamp}\",\n", + " )\n", + " total_energy_job_ids[name] = job[\"_id\"]\n", + " print(f\"✅ {name}: created Total Energy job {job['_id']}\")" ] }, { - "cell_type": "markdown", + "cell_type": "code", + "execution_count": null, "id": "36", "metadata": {}, + "outputs": [], + "source": [ + "for name, job_id in total_energy_job_ids.items():\n", + " submit_jobs(client.jobs, [job_id])\n", + " print(f\"✅ {name}: submitted Total Energy job {job_id}\")\n", + " await wait_for_jobs_to_finish_async(client.jobs, [job_id], poll_interval=POLL_INTERVAL)\n", + " total_energies[name] = get_properties_for_job(client, job_id, \"total_energy\")[0][\"value\"]\n", + " print(f\"{name}: {total_energies[name]:.4f} eV\")" + ] + }, + { + "cell_type": "markdown", + "id": "37", + "metadata": {}, "source": [ - "## 6. Create the job with material and workflow configuration\n", - "### 6.1. Create job" + "## 8. Band structure of the defective cell\n", + "### 8.1. Configure the workflow" ] }, { "cell_type": "code", "execution_count": null, - "id": "37", + "id": "38", "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.api.job import create_job\n", - "from mat3ra.notebooks_utils.ui import display_JSON\n", - "\n", - "print(f\"Material: {saved_material.id}\")\n", - "print(f\"Workflow: {saved_workflow.id}\")\n", - "print(f\"Project: {project_id}\")\n", - "\n", - "job_response = create_job(\n", - " api_client=client,\n", - " materials=[saved_material],\n", - " workflow=workflow,\n", - " project_id=project_id,\n", - " owner_id=ACCOUNT_ID,\n", - " prefix=JOB_NAME,\n", - " compute=compute.to_dict(),\n", + "from mat3ra.wode.context.providers import PointsPathDataProvider\n", + "from mat3ra.notebooks_utils.ipython.entity.workflow.visualize import visualize_workflow\n", + "\n", + "band_structure_workflow_name = f\"{MY_WORKFLOW_NAME} {DEFECTIVE_NAME} {MODEL_TAG}\"\n", + "band_structure_workflow = Workflow.create(\n", + " WorkflowStandata.filter_by_application(app.name).get_by_name_first_match(BAND_STRUCTURE_WORKFLOW_SEARCH_TERM)\n", ")\n", + "band_structure_workflow.name = band_structure_workflow_name\n", + "band_structure_workflow.subworkflows[0].model = model\n", + "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=SCF_UNIT)\n", + "apply_planewave_cutoffs(band_structure_workflow, ECUTWFC, ECUTRHO, unit_name=BANDS_UNIT)\n", + "apply_scf_kgrid(band_structure_workflow, kgrid[DEFECTIVE_NAME], material=defective_for_jobs)\n", + "patch_workflow_qe_input(band_structure_workflow, {\"system\": SPIN_SETTINGS[DEFECTIVE_NAME]}, [SCF_UNIT, BANDS_UNIT])\n", + "\n", + "bands_subworkflow = band_structure_workflow.subworkflows[0]\n", + "bands_unit = bands_subworkflow.get_unit_by_name(name=BANDS_UNIT)\n", + "bands_unit.add_context(PointsPathDataProvider(path=KPATH, isEdited=True).get_context_item_data())\n", + "bands_subworkflow.set_unit(bands_unit)\n", "\n", - "job = dict_to_namespace_recursive(job_response)\n", - "job_id = job._id\n", - "print(\"✅ Job created successfully!\")\n", - "print(f\"Job ID: {job_id}\")\n", - "display_JSON(job_response)" + "visualize_workflow(band_structure_workflow)" ] }, { "cell_type": "markdown", - "id": "38", + "id": "39", "metadata": {}, "source": [ - "## 7. Submit the job and monitor the status" + "### 8.2. Create the job" ] }, { "cell_type": "code", "execution_count": null, - "id": "39", + "id": "40", "metadata": {}, "outputs": [], "source": [ - "client.jobs.submit(job_id)\n", - "print(f\"✅ Job {job_id} submitted successfully!\")" + "band_structure_job = find_job_for_material(\n", + " client, saved_defective_for_jobs[\"_id\"], band_structure_workflow_name, ACCOUNT_ID,\n", + " statuses=(\"submitted\", \"queued\", \"active\", \"finished\"),\n", + ")\n", + "if band_structure_job is None:\n", + " band_structure_job = create_job(\n", + " api_client=client, materials=[saved_defective_for_jobs], workflow=band_structure_workflow,\n", + " project_id=project_id, owner_id=ACCOUNT_ID, compute=compute.to_dict(),\n", + " prefix=f\"{band_structure_workflow.name} {timestamp}\",\n", + " )\n", + " new_band_structure_job_id = band_structure_job[\"_id\"]\n", + " print(f\"✅ Band Structure job created: {new_band_structure_job_id}\")\n", + "else:\n", + " new_band_structure_job_id = None\n", + " print(f\"♻️ Reusing Band Structure job {band_structure_job['_id']}\")\n", + "band_structure_job_id = band_structure_job[\"_id\"]" + ] + }, + { + "cell_type": "markdown", + "id": "41", + "metadata": {}, + "source": [ + "### 8.3. Submit and monitor the job" ] }, { "cell_type": "code", "execution_count": null, - "id": "40", + "id": "42", "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.api.job import wait_for_jobs_to_finish_async\n", - "\n", - "await wait_for_jobs_to_finish_async(client.jobs, [job_id], poll_interval=POLL_INTERVAL)" + "if new_band_structure_job_id:\n", + " client.jobs.submit(new_band_structure_job_id)\n", + " print(f\"✅ Job {new_band_structure_job_id} submitted successfully!\")\n", + "await wait_for_jobs_to_finish_async(client.jobs, [band_structure_job_id], poll_interval=POLL_INTERVAL)" ] }, { "cell_type": "markdown", - "id": "41", + "id": "43", "metadata": {}, "source": [ - "## 8. Retrieve results" + "## 9. Retrieve the results\n", + "### 9.1. Band structure" ] }, { "cell_type": "code", "execution_count": null, - "id": "42", + "id": "44", "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.core.entity.property.api import get_properties_for_job\n", "from mat3ra.notebooks_utils.ipython.entity.property.visualize import visualize_properties\n", "\n", - "property_data = get_properties_for_job(client, job_id, property_name=\"band_structure\")\n", - "visualize_properties(property_data, title=\"Band Structure\", extra_config={\"material\": material.to_dict()})" + "band_structure_data = get_properties_for_job(client, band_structure_job_id, property_name=\"band_structure\")\n", + "visualize_properties(band_structure_data, title=\"Band Structure\",\n", + " extra_config={\"material\": defective_for_jobs.to_dict()})" + ] + }, + { + "cell_type": "markdown", + "id": "45", + "metadata": {}, + "source": [ + "### 9.2. Formation energy" ] }, { "cell_type": "code", "execution_count": null, - "id": "43", + "id": "46", "metadata": {}, "outputs": [], - "source": [] + "source": [ + "defective_composition = Counter(defective_for_jobs.basis.elements.values)\n", + "mu_carbon = total_energies[PRISTINE_NAME] / pristine.basis.number_of_atoms\n", + "mu_nitrogen = total_energies[NITROGEN_NAME] / nitrogen.basis.number_of_atoms\n", + "e_formation = (\n", + " total_energies[DEFECTIVE_NAME]\n", + " - defective_composition[\"C\"] * mu_carbon\n", + " - defective_composition[\"N\"] * mu_nitrogen\n", + ")\n", + "\n", + "print(f\"μ_C = {mu_carbon:.4f} eV/atom ({pristine.basis.number_of_atoms} atoms)\")\n", + "print(f\"μ_N = {mu_nitrogen:.4f} eV/atom ({nitrogen.basis.number_of_atoms} atoms)\")\n", + "print(f\"E_f = {total_energies[DEFECTIVE_NAME]:.4f} eV - {defective_composition['C']} μ_C \"\n", + " f\"- {defective_composition['N']} μ_N = {e_formation:.3f} eV\")" + ] + }, + { + "cell_type": "markdown", + "id": "47", + "metadata": {}, + "source": [ + "### 9.3. Compare with Fujimoto & Saito (2011)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "48", + "metadata": {}, + "outputs": [], + "source": [ + "FUJIMOTO_FORMATION_ENERGY = 2.51 # eV, Table I, trimerized pyridine-type C28N3\n", + "TOLERANCE_FRACTION = 0.15\n", + "\n", + "deviation = e_formation - FUJIMOTO_FORMATION_ENERGY\n", + "verdict = \"yes\" if RELAX and abs(deviation) <= TOLERANCE_FRACTION * FUJIMOTO_FORMATION_ENERGY else \"no\"\n", + "print(f\"Regime: {RUN_TIER} tier\")\n", + "print(f\"E_f (this notebook): {e_formation:.3f} eV\")\n", + "print(f\"E_f (Fujimoto & Saito 2011, Table I): {FUJIMOTO_FORMATION_ENERGY:.3f} eV\")\n", + "print(f\"Deviation: {deviation:+.3f} eV ({100 * deviation / FUJIMOTO_FORMATION_ENERGY:+.1f} %, \"\n", + " f\"tolerance {100 * TOLERANCE_FRACTION:.0f} %)\")\n", + "print(\"Offsets: μ_N comes from solid N2 (mp-154) rather than the free molecule the manuscript \"\n", + " \"uses, and GBRV ultrasoft LDA stands in for its Troullier-Martins norm-conserving set.\")\n", + "print(f\"Reproduces Fujimoto & Saito (2011): {verdict} ({RUN_TIER} tier)\")" + ] + }, + { + "cell_type": "markdown", + "id": "49", + "metadata": {}, + "source": [ + "## References\n", + "\n", + "[1] Y. Fujimoto and S. Saito, \"Formation, stabilities, and electronic properties of nitrogen defects in graphene\", Phys. Rev. B 84, 245446 (2011). https://doi.org/10.1103/PhysRevB.84.245446\n", + "\n", + "[2] K. F. Garrity, J. W. Bennett, K. M. Rabe and D. Vanderbilt, \"Pseudopotentials for high-throughput DFT calculations\", Comput. Mater. Sci. 81, 446 (2014). https://doi.org/10.1016/j.commatsci.2013.08.053" + ] } ], "metadata": { diff --git a/src/py/mat3ra/notebooks_utils/core/entity/material/io.py b/src/py/mat3ra/notebooks_utils/core/entity/material/io.py index 0a4cbe3f1..ef7136aa2 100644 --- a/src/py/mat3ra/notebooks_utils/core/entity/material/io.py +++ b/src/py/mat3ra/notebooks_utils/core/entity/material/io.py @@ -122,35 +122,52 @@ def load_materials_from_folder(folder_path: Optional[str] = None, verbose: bool return materials +def _get_name_match_rank(requested_name: str, material_name: str, filename_without_extension: str) -> Optional[int]: + """ + Rank how well one folder entry matches a requested name: lower is better, None is no match. + + An exact match wins over a substring match, and a case-sensitive match over a case-insensitive + one. Among exact matches the material name wins; among substring matches the filename wins. + """ + requested_name_lower = requested_name.lower() + ranked_candidates = [ + material_name == requested_name, + filename_without_extension == requested_name, + material_name.lower() == requested_name_lower, + filename_without_extension.lower() == requested_name_lower, + requested_name_lower in filename_without_extension.lower(), + requested_name_lower in material_name.lower(), + ] + return next((rank for rank, matches in enumerate(ranked_candidates) if matches), None) + + def load_material_from_folder(folder_path: str, name: str, verbose: bool = True) -> Optional[Any]: """ - Load a single material from the specified folder by matching a substring of the name or filename. + Load a single material from the specified folder by matching the name or filename. Args: folder_path (str): The path to the folder containing material files. - name (str): The substring to match against material names or filenames (case-insensitive). + name (str): The name to match against material names or filenames. An exact match is + preferred; otherwise a case-insensitive substring match is accepted. verbose (bool): Whether to log verbose messages. Returns: - Optional[Material]: The first Material object that matches, or None if not found. + Optional[Material]: The best matching Material object, or None if not found. """ - name_lower = name.lower() + best_rank = None resulting_material = None for filename in sorted(os.listdir(folder_path)): - if filename.endswith(".json") and name_lower in os.path.splitext(filename)[0].lower(): - with open(os.path.join(folder_path, filename), "r") as file: - data = json.load(file) - if MaterialWithBuildMetadata.is_valid(data): - resulting_material = MaterialWithBuildMetadata.create(data) - break - - if not resulting_material: - materials = load_materials_from_folder(folder_path, verbose=verbose) - for material in materials: - if name_lower in material.name.lower(): - resulting_material = material - break + if not filename.endswith(".json"): + continue + with open(os.path.join(folder_path, filename), "r") as file: + data = json.load(file) + if not MaterialWithBuildMetadata.is_valid(data): + continue + material = MaterialWithBuildMetadata.create(data) + rank = _get_name_match_rank(name, material.name, os.path.splitext(filename)[0]) + if rank is not None and (best_rank is None or rank < best_rank): + best_rank, resulting_material = rank, material if resulting_material: log(f"Found: '{resulting_material.name}'", SeverityLevelEnum.INFO, force_verbose=verbose) diff --git a/src/py/mat3ra/notebooks_utils/core/entity/property/api.py b/src/py/mat3ra/notebooks_utils/core/entity/property/api.py index 050c16f4f..65e59feb7 100644 --- a/src/py/mat3ra/notebooks_utils/core/entity/property/api.py +++ b/src/py/mat3ra/notebooks_utils/core/entity/property/api.py @@ -71,7 +71,13 @@ def update_property_holder_value(client: APIClient, property_holder_id: str, val return client.properties.update(property_holder_id, {"$set": {"data.value": value}}) -def find_total_energy_for_material(client: APIClient, material_id: str, source: str = "my_account") -> Optional[dict]: +def find_total_energy_for_material( + client: APIClient, + material_id: str, + source: str = "my_account", + group: Optional[str] = None, + precision_value: Optional[float] = None, +) -> Optional[dict]: """ Find the best-precision total_energy property for a material. Mirrors the platform's "Resolve Total Energies for Elemental Materials" subworkflow, @@ -86,6 +92,9 @@ def find_total_energy_for_material(client: APIClient, material_id: str, source: material_id (str): Material _id to look up the total_energy property for. source (str): Source of the total energy property: `my_account` (default), `curators` or `public`. + group (str, optional): Model identifier the property must carry, e.g. `qe:dft:lda:pz`. + Without it a property computed with a different model is returned. + precision_value (float, optional): `precision.value` (KPPRA) the property must carry. Returns: The best-precision total_energy property, or None if none exists. @@ -95,6 +104,10 @@ def find_total_energy_for_material(client: APIClient, material_id: str, source: if not exabyte_id: return None query = {"exabyteId": exabyte_id, "slug": "total_energy"} + if group: + query["group"] = group + if precision_value is not None: + query["precision.value"] = precision_value if source == "curators": query["owner.slug"] = "curators" elif source == "my_account": diff --git a/tests/py/unit/core/entity/test_material_io.py b/tests/py/unit/core/entity/test_material_io.py index 5771ca68d..51ebe827a 100644 --- a/tests/py/unit/core/entity/test_material_io.py +++ b/tests/py/unit/core/entity/test_material_io.py @@ -1,10 +1,33 @@ import json +import pytest from mat3ra.notebooks_utils.core.entity.material.io import load_material_from_folder, load_materials_from_folder from mat3ra.standata.materials import Materials RADII = {"Si": 1.11, "C": 0.76} +GRAPHENE_STANDATA_NAME = "C, Graphene, HEX (P6/mmm) 2D (Monolayer), 2dm-3993" +PRISTINE_NAME = "graphene 4x4" +DEFECTIVE_NAME = "graphene 4x4 N3V pyridinic (C28N3)" +RELAXED_NAME = "graphene 4x4 relaxed" + +# (filename without extension, material name) pairs; the relaxed material is stored under a +# filename that is not its name. +GRAPHENE_FOLDER = [ + (GRAPHENE_STANDATA_NAME.replace("/", "-"), GRAPHENE_STANDATA_NAME), + (PRISTINE_NAME, PRISTINE_NAME), + (DEFECTIVE_NAME, DEFECTIVE_NAME), + ("pristine graphene", RELAXED_NAME), +] +RENAMED_FILE_FOLDER = [ + (PRISTINE_NAME, RELAXED_NAME), + (DEFECTIVE_NAME, DEFECTIVE_NAME), +] +RECASED_NAME_FOLDER = [ + (PRISTINE_NAME, PRISTINE_NAME.upper()), + ("n3v defect", PRISTINE_NAME), +] + def _uploads_folder(tmp_path): (tmp_path / "silicon.json").write_text(json.dumps(Materials.get_by_name_first_match("Silicon"))) @@ -18,3 +41,32 @@ def test_data_file_beside_a_material_is_skipped(tmp_path): def test_lookup_by_name_works_alongside_a_data_file(tmp_path): assert load_material_from_folder(_uploads_folder(tmp_path), "Silicon", verbose=False) is not None + + +def _material_folder(tmp_path, files): + for filename, name in files: + graphene = {**Materials.get_by_name_first_match("Graphene"), "name": name} + (tmp_path / f"{filename}.json").write_text(json.dumps(graphene)) + return str(tmp_path) + + +@pytest.mark.parametrize( + ("files", "requested_name", "expected_name"), + [ + (GRAPHENE_FOLDER, PRISTINE_NAME, PRISTINE_NAME), + (GRAPHENE_FOLDER, DEFECTIVE_NAME, DEFECTIVE_NAME), + (GRAPHENE_FOLDER, "GRAPHENE 4X4", PRISTINE_NAME), + (GRAPHENE_FOLDER, "N3V", DEFECTIVE_NAME), + (GRAPHENE_FOLDER, "pristine", RELAXED_NAME), + (GRAPHENE_FOLDER, "relaxed", RELAXED_NAME), + (GRAPHENE_FOLDER, "Graphene", GRAPHENE_STANDATA_NAME), + (GRAPHENE_FOLDER, "Germanium", None), + (RENAMED_FILE_FOLDER, PRISTINE_NAME, RELAXED_NAME), + (RENAMED_FILE_FOLDER, "GRAPHENE 4X4", RELAXED_NAME), + (RECASED_NAME_FOLDER, PRISTINE_NAME, PRISTINE_NAME), + ], +) +def test_load_material_from_folder(tmp_path, files, requested_name, expected_name): + material = load_material_from_folder(_material_folder(tmp_path, files), requested_name, verbose=False) + + assert (material.name if material else None) == expected_name diff --git a/tests/py/unit/core/entity/test_property_api.py b/tests/py/unit/core/entity/test_property_api.py index 8dc5585c8..f1693bc32 100644 --- a/tests/py/unit/core/entity/test_property_api.py +++ b/tests/py/unit/core/entity/test_property_api.py @@ -9,6 +9,8 @@ EXABYTE_ID = "exabyte-a" MATERIAL = {"_id": MATERIAL_ID, "exabyteId": EXABYTE_ID} OWNER_ACCOUNT_ID = "account-a" +GROUP = "qe:dft:lda:pz" +PRECISION_VALUE = 1116 TOTAL_ENERGY_PROPERTY = {"data": {"value": -12.34}, "precision": {"value": 0.001}} @@ -84,6 +86,23 @@ def test_find_total_energy_for_material_public_scope_has_no_owner_filter(): ) +def test_find_total_energy_for_material_constrains_group_and_precision(): + client = _client() + + find_total_energy_for_material(client, MATERIAL_ID, group=GROUP, precision_value=PRECISION_VALUE) + + client.properties.list.assert_called_once_with( + query={ + "exabyteId": EXABYTE_ID, + "slug": "total_energy", + "group": GROUP, + "precision.value": PRECISION_VALUE, + "owner._id": OWNER_ACCOUNT_ID, + }, + projection={"sort": {"precision.value": -1}, "limit": 1}, + ) + + def test_find_total_energy_for_material_rejects_invalid_source(): client = _client()