diff --git a/.DS_Store b/.DS_Store
deleted file mode 100644
index a701106b..00000000
Binary files a/.DS_Store and /dev/null differ
diff --git a/.continue/agents/new-config.yaml b/.continue/agents/new-config.yaml
new file mode 100644
index 00000000..2d363778
--- /dev/null
+++ b/.continue/agents/new-config.yaml
@@ -0,0 +1,23 @@
+# This is an example configuration file
+# To learn more, see the full config.yaml reference: https://docs.continue.dev/reference
+
+name: Example Config
+version: 1.0.0
+schema: v1
+
+# Define which models can be used
+# https://docs.continue.dev/customization/models
+models:
+ - name: my gpt-5
+ provider: openai
+ model: gpt-5
+ apiKey: YOUR_OPENAI_API_KEY_HERE
+ - uses: ollama/qwen2.5-coder-7b
+ - uses: anthropic/claude-4-sonnet
+ with:
+ ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
+
+# MCP Servers that Continue can access
+# https://docs.continue.dev/customization/mcp-tools
+mcpServers:
+ - uses: anthropic/memory-mcp
\ No newline at end of file
diff --git a/.dockerignore b/.dockerignore
new file mode 100644
index 00000000..76db108a
--- /dev/null
+++ b/.dockerignore
@@ -0,0 +1,77 @@
+# Git and version control
+.git
+.gitignore
+.gitattributes
+
+# IDE and editor files
+.vscode
+.cursor
+.continue
+.claude
+.omc
+.playwright-cli
+.research
+.env
+.idea
+*.swp
+*.swo
+*~
+.DS_Store
+
+# Python
+__pycache__
+*.py[cod]
+*$py.class
+*.egg-info
+dist
+build
+.egg
+.pytest_cache
+.mypy_cache
+.ruff_cache
+*.egg
+venv
+env
+.venv
+
+# Test and coverage
+coverage
+htmlcov
+.coverage
+.coverage.*
+
+# Data and artifacts
+*.log
+.tmp
+tmp
+*.pth
+*.pt
+*.pickle
+*.pkl
+checkpoint*
+dataset
+output
+training_*.log
+
+# Docker
+.dockerignore
+Dockerfile
+
+# CI/CD
+.github
+.gitlab-ci.yml
+
+# Documentation build
+docs/_build
+site
+
+# Misc
+*.md
+!README.md
+!docs/DOCKER_RESEARCH.md
+!START_HERE.md
+!QUICK_START.md
+!docs/*.md
+!build_heterojunctions/README.md
+.env.local
+.env.*.local
diff --git a/.env.example b/.env.example
new file mode 100644
index 00000000..da67be6e
--- /dev/null
+++ b/.env.example
@@ -0,0 +1,4 @@
+INTERFACEML_PORT=8001
+JUPYTER_PORT=8888
+INTERFACEML_WORKSPACE=./.research
+INTERFACEML_UID=1000
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
new file mode 100644
index 00000000..a98fa16e
--- /dev/null
+++ b/.github/workflows/ci.yml
@@ -0,0 +1,111 @@
+name: CI
+
+on:
+ push:
+ branches: [main, devel]
+ pull_request:
+ branches: [main, devel]
+
+concurrency:
+ group: ${{ github.workflow }}-${{ github.ref }}
+ cancel-in-progress: true
+
+jobs:
+ lint:
+ name: Lint (ruff)
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+
+ - uses: actions/setup-python@v5
+ with:
+ python-version: "3.11"
+
+ - name: Install ruff
+ run: pip install ruff
+
+ - name: Ruff check
+ run: ruff check interfaceml/ tests/
+
+ - name: Ruff format check
+ run: ruff format --check interfaceml/ tests/
+
+ test-core:
+ name: Test (Python ${{ matrix.python-version }})
+ runs-on: ubuntu-latest
+ strategy:
+ fail-fast: false
+ matrix:
+ python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
+ steps:
+ - uses: actions/checkout@v4
+
+ - uses: actions/setup-python@v5
+ with:
+ python-version: ${{ matrix.python-version }}
+
+ - name: Cache pip
+ uses: actions/cache@v4
+ with:
+ path: ~/.cache/pip
+ key: ${{ runner.os }}-pip-${{ matrix.python-version }}-${{ hashFiles('pyproject.toml', 'requirements.txt') }}
+ restore-keys: |
+ ${{ runner.os }}-pip-${{ matrix.python-version }}-
+
+ - name: Install package
+ run: |
+ pip install --upgrade pip
+ pip install -e ".[dev]"
+
+ - name: Run core tests
+ run: |
+ python -m pytest tests/ -v --tb=short \
+ --ignore=tests/test_model_and_diffusion.py \
+ --ignore=tests/test_generate_utils.py \
+ -x
+
+ test-ml:
+ name: Test ML (PyTorch)
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+
+ - uses: actions/setup-python@v5
+ with:
+ python-version: "3.11"
+
+ - name: Cache pip
+ uses: actions/cache@v4
+ with:
+ path: ~/.cache/pip
+ key: ${{ runner.os }}-pip-ml-${{ hashFiles('pyproject.toml', 'fullerene_e3gen/requirements.txt') }}
+ restore-keys: |
+ ${{ runner.os }}-pip-ml-
+
+ - name: Install dependencies
+ run: |
+ pip install --upgrade pip
+ pip install -e ".[dev,ai]"
+
+ - name: Run ML tests
+ run: |
+ python -m pytest tests/test_model_and_diffusion.py tests/test_generate_utils.py -v --tb=short -x
+
+ typecheck:
+ name: Type check (mypy)
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+
+ - uses: actions/setup-python@v5
+ with:
+ python-version: "3.11"
+
+ - name: Install dependencies
+ run: |
+ pip install --upgrade pip
+ pip install -e ".[dev]"
+
+ - name: Run mypy
+ run: mypy interfaceml/ --ignore-missing-imports
+ continue-on-error: true # informational for now
diff --git a/.gitignore b/.gitignore
index 9148f239..22d07062 100644
--- a/.gitignore
+++ b/.gitignore
@@ -65,6 +65,10 @@ dmypy.json
*.rej
nohup.out
webserver.log
+.env
+.research/
+output/playwright/
+.playwright-cli/
# Output files (keep examples but ignore test outputs)
examples/output/*.vasp
diff --git a/Dockerfile b/Dockerfile
new file mode 100644
index 00000000..1520e395
--- /dev/null
+++ b/Dockerfile
@@ -0,0 +1,31 @@
+FROM python:3.11-slim@sha256:e41613d42d4891e4930f79523f93f81bbc7632584ec65e36ab055f41a800b41e
+
+WORKDIR /app
+
+ENV PYTHONUNBUFFERED=1 \
+ PYTHONDONTWRITEBYTECODE=1 \
+ MPLCONFIGDIR=/tmp/matplotlib \
+ OMP_NUM_THREADS=2
+
+RUN apt-get update && apt-get install -y --no-install-recommends libgomp1 git \
+ && rm -rf /var/lib/apt/lists/*
+
+COPY requirements-research.lock ./
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip install --require-hashes -r requirements-research.lock
+
+COPY . .
+RUN pip install --no-cache-dir --no-deps .
+
+ARG USER_ID=1000
+RUN useradd -m -u "${USER_ID}" interfaceml \
+ && mkdir -p /workspace \
+ && chown interfaceml:interfaceml /workspace
+USER interfaceml
+
+HEALTHCHECK --interval=15s --timeout=10s --start-period=60s --retries=5 \
+ CMD python -c "import json, urllib.request; data = json.load(urllib.request.urlopen('http://localhost:5000/api/health')); assert data['status'] == 'ok' and data['core_available']"
+
+EXPOSE 5000 8888
+
+CMD ["gunicorn", "--bind", "0.0.0.0:5000", "--workers", "1", "--threads", "4", "--timeout", "300", "interfaceml.web.app:app"]
diff --git a/INTEGRATION_GUIDE.md b/INTEGRATION_GUIDE.md
new file mode 100644
index 00000000..62a80416
--- /dev/null
+++ b/INTEGRATION_GUIDE.md
@@ -0,0 +1,152 @@
+# InterfaceML and AI Structure Generation Integration Guide
+
+## Overview
+
+This guide describes the integration of the AI structure generation module with
+InterfaceML. The integration adds fullerene generation and evaluation to the web
+interface and exposes new API endpoints for batch workflows.
+
+## Features
+
+- AI-based fullerene generation for C60, C70, C80, C84, C100, and custom sizes
+- Batch generation with metadata export (XYZ and JSON)
+- Web UI integration with status checks and validation
+- Unified REST API endpoints for traditional and AI features
+
+## API Endpoints
+
+Traditional features:
+- `/api/upload` - File uploads
+- `/api/build-interface` - Interface builder
+- `/api/fix-layers` - Layer management
+- `/api/compute-density` - Density analysis
+- `/api/plot-dos` - DOS visualization
+
+AI features:
+- `/api/ai/generate` - Structure generation
+- `/api/ai/evaluate` - Quality metrics
+- `/api/ai/info` - Model information
+
+## Installation and Setup
+
+### Prerequisites
+
+- Python 3.8 or higher
+- InterfaceML dependencies
+- PyTorch 2.0 or higher
+- PyTorch Geometric 2.0 or higher
+
+### Quick Start
+
+1. Ensure the AI module is present:
+
+```bash
+cd /path/to/InterfaceML
+ls fullerene_e3gen/checkpoints/best_model.pt
+```
+
+2. Optional environment overrides:
+
+```bash
+# Override the AI module location
+export INTERFACEML_FULLERENE_PATH=/path/to/fullerene_e3gen
+
+# Override the checkpoint path
+export INTERFACEML_FULLERENE_CHECKPOINT=/path/to/best_model.pt
+```
+
+3. Start the integrated server:
+
+```bash
+cd interfaceml/web
+python app.py --port 5000
+```
+
+Expected output includes the upload folder, module availability, and server URL.
+
+4. Open a browser at `http://localhost:5000` and select the AI Structure
+ Generation tab.
+
+## Usage Examples
+
+### 1. Generate C60 Fullerenes (Web)
+
+1. Select "C60 (Buckminsterfullerene)" from the dropdown
+2. Set the number of samples
+3. Click "Generate Structures"
+4. Download results
+
+### 2. Generate C60 Fullerenes (API)
+
+```bash
+curl -X POST http://localhost:5000/api/ai/generate \
+ -H "Content-Type: application/json" \
+ -d '{
+ "num_atoms": 60,
+ "num_samples": 5,
+ "output_format": "xyz"
+ }'
+```
+
+### 3. Evaluate Generated Structures (API)
+
+```bash
+curl -X POST http://localhost:5000/api/ai/evaluate \
+ -H "Content-Type: application/json" \
+ -d '{
+ "generated_dir": "ai_generated_abc123"
+ }'
+```
+
+### 4. Batch Generation (Web)
+
+1. Select multiple carbon counts
+2. Set samples per size
+3. Click "Batch Generate"
+
+### 5. Custom Fullerenes (Web)
+
+1. Select "Custom" from the dropdown
+2. Enter an even atom count between 20 and 240
+3. Generate
+
+## Architecture
+
+### Backend Integration
+
+Key files:
+- `interfaceml/web/app.py` - app factory and blueprint registration
+- `interfaceml/web/routes/` - Flask blueprints by domain
+- `interfaceml/web/ai.py` - AI module discovery and loading
+- `interfaceml/web/core.py` - core module availability
+- `interfaceml/web/dos.py` - DOS and PDOS helpers
+- `interfaceml/web/utils.py` - shared utilities
+
+```python
+from interfaceml.web.ai import load_fullerene_api
+ai_status = load_fullerene_api()
+
+app.register_blueprint(common_bp)
+app.register_blueprint(interface_bp)
+app.register_blueprint(dos_bp)
+app.register_blueprint(ai_bp)
+```
+
+### Frontend Integration
+
+Key files:
+- `templates/index.html` - AI tab UI
+- `static/css/style.css` - AI styles
+- `static/js/app.js` - AI interactions
+
+## Configuration
+
+The AI module loads from the following layout:
+
+```
+fullerene_e3gen/
+|-- checkpoints/
+| `-- best_model.pt # Main checkpoint (required)
+|-- api.py # Python API
+`-- config.yaml # Model config
+```
diff --git a/QUICK_START.md b/QUICK_START.md
index 48a33b3c..70bcb4e7 100644
--- a/QUICK_START.md
+++ b/QUICK_START.md
@@ -1,42 +1,35 @@
-# Quick Start Guide
+# Quick Start
-Get InterfaceML up and running in 5 minutes!
+This guide covers a minimal setup. For a full tutorial and troubleshooting, see
+`docs/GETTING_STARTED.md`.
-## 🚀 Installation
+## Install
```bash
-# 1. Navigate to the project directory
+# From the repository root
cd /path/to/InterfaceML
-# 2. Install dependencies
pip install -r requirements.txt
-# 3. (Optional) Install package for easy access
+# Optional: install the package in editable mode
pip install -e .
```
----
-
-## ✅ Verify Installation
+## Verify
```bash
# Test Python import
-python -c "from interfaceml.core import io, layering; print('✓ InterfaceML ready!')"
+python -c "from interfaceml.core import io, layering; print('InterfaceML import: OK')"
-# Check CLI tools
+# Check CLI help
python build_heterojunctions/interface_builder.py --help
```
----
-
-## 🎯 Choose Your Interface
-
-### Option 1: Command-Line (Traditional)
+## Run a Workflow
-Best for: Automation, scripting, HPC workflows
+### CLI (adsorbate example)
```bash
-# Build adsorbate model
python build_heterojunctions/interface_builder.py \
--adsorbate_mode \
--base structures/perovskites/fapbi3-1.cif \
@@ -45,7 +38,6 @@ python build_heterojunctions/interface_builder.py \
--distance 2.5 \
--output my_interface.vasp
-# Fix layers for relaxation
python build_heterojunctions/fix_interface_layers.py \
--by_z_layers \
--input my_interface.vasp \
@@ -53,174 +45,36 @@ python build_heterojunctions/fix_interface_layers.py \
--output my_interface_fixed.vasp
```
-### Option 2: Web Interface (Interactive)
-
-Best for: Visualization, parameter exploration, quick testing
+### Web Interface
```bash
-# Start web server
python -m interfaceml.web.app
-
-# Open browser to http://localhost:5000
-# Use drag-and-drop interface!
```
-### Option 3: Python API (Programmatic)
+Then open:
+- Docs homepage: `http://localhost:5000/`
+- Web app UI: `http://localhost:5000/app`
-Best for: Custom workflows, Jupyter notebooks, integration
+### Python API
```python
from interfaceml.core import io, layering
-# Load structure
structure = io.load_structure("my_structure.vasp")
+layers, _ = layering.split_layers_by_z(structure)
+fixed = [i for layer in layers[:3] for i in layer]
+fixed = layering.include_whole_molecules(structure, fixed)
-# Detect and fix bottom 3 layers
-layers, tol = layering.split_layers_by_z(structure)
-fixed_indices = []
-for layer in layers[:3]:
- fixed_indices.extend(layer)
-
-# Include whole molecules
-fixed_indices = layering.include_whole_molecules(structure, fixed_indices)
-
-# Write output
-selective_dynamics = [
- (False, False, False) if i in fixed_indices else (True, True, True)
+selective = [
+ (False, False, False) if i in fixed else (True, True, True)
for i in range(len(structure))
]
-io.write_poscar(structure, "output.vasp", selective_dynamics=selective_dynamics)
-```
-
----
-
-## 📖 Next Steps
-
-1. **Read the main README**: [`README.md`](README.md)
-2. **Try the example workflow**: `bash examples/example_workflow.sh`
-3. **Explore detailed docs**: [`docs/GETTING_STARTED.md`](docs/GETTING_STARTED.md)
-4. **Check technical details**: [`build_heterojunctions/README.md`](build_heterojunctions/README.md)
-
----
-
-## 🆘 Common Issues
-
-### "No module named 'pymatgen'"
-
-```bash
-pip install pymatgen
-# or
-conda install -c conda-forge pymatgen
-```
-
-### "No module named 'flask'"
-
-```bash
-pip install flask flask-cors
-```
-
-### Web interface won't start
-
-```bash
-# Try different port
-python -m interfaceml.web.app --port 8000
-
-# Check firewall settings
-```
-
-### Import errors
-
-```bash
-# Make sure you're in the project directory
-cd /path/to/InterfaceML
-
-# Or install package
-pip install -e .
-```
-
----
-
-## 💡 Quick Examples
-
-### Build FAPbI3/C60 Interface
-
-```bash
-python build_heterojunctions/interface_builder.py \
- --adsorbate_mode \
- --base structures/perovskites/fapbi3-1.cif \
- --adsorbate structures/etl/C60-Ih.cif \
- --termination FAI \
- --distance 2.5 \
- --supercell auto \
- --output fapbi3_c60.vasp
-```
-
-### Build Multi-Layer Stack
-
-```bash
-python build_heterojunctions/interface_builder.py \
- --adsorbate_mode \
- --base structures/perovskites/fapbi3-1.cif \
- --adsorbate structures/etl/C70-D5h.cif structures/etl/C60-Ih.cif \
- --termination AI \
- --layer_gaps 2.0 2.0 \
- --supercell auto \
- --output trilayer.vasp
-```
-
-### Fix Layers with Debug Info
-
-```bash
-python build_heterojunctions/fix_interface_layers.py \
- --by_z_layers \
- --input structure.vasp \
- --n_fix_layers 3 \
- --debug_layers \
- --print_only
-```
-
----
-
-## 🎨 Web Interface Features
-
-After starting `python -m interfaceml.web.app`:
-
-1. **Upload Tab** 📁
- - Drag and drop structure files
- - Automatic format detection
- - Structure info display
-
-2. **Build Tab** 🔨
- - Configure parameters
- - Real-time validation
- - One-click generation
-
-3. **Download Tab** ⬇️
- - Download results
- - View atom counts
- - Fixed indices list
-
----
-
-## 📚 Documentation Structure
-
+io.write_poscar(structure, "output.vasp", selective_dynamics=selective)
```
-README.md ← Start here!
-├── QUICK_START.md ← You are here
-├── docs/GETTING_STARTED.md ← Detailed tutorial
-├── build_heterojunctions/README.md ← Technical reference
-└── examples/example_workflow.sh ← Complete example
-```
-
----
-
-## 🤝 Getting Help
-
-- 📖 Read the [full documentation](README.md)
-- 🐛 [Report bugs](https://github.com/yourusername/InterfaceML/issues)
-- 💬 [Ask questions](https://github.com/yourusername/InterfaceML/discussions)
-- 📧 Email: interface@example.com
----
+## Next Steps
-**Ready to build amazing interfaces? Let's go! 🚀**
+- `docs/GETTING_STARTED.md` for the full tutorial and troubleshooting
+- `build_heterojunctions/README.md` for CLI reference
+- `docs/MATHEMATICAL_OVERVIEW.md` for algorithm details
+- `README.md` for the project overview
diff --git a/README.md b/README.md
index f9f0ecd0..38246649 100644
--- a/README.md
+++ b/README.md
@@ -2,39 +2,39 @@
-⚛️ **Professional Heterojunction Modeling Platform**
+**Professional Heterojunction Modeling Platform**
-[](https://www.python.org/)
+[](https://www.python.org/)
[](LICENSE)
[](https://pymatgen.org/)
*A comprehensive toolkit for building, analyzing, and optimizing heterostructure interfaces for DFT calculations*
-[Features](#features) • [Installation](#installation) • [Quick Start](#quick-start) • [Documentation](#documentation) • [Web Interface](#web-interface)
+[Features](#features) | [Installation](#installation) | [Quick Start](#quick-start) | [Documentation](#documentation) | [Web Interface](#web-interface)
---
-## 🌟 Features
+## Features
InterfaceML provides a complete suite of tools for heterojunction modeling with a focus on:
-### 🔬 **Adsorbate Modeling**
+### Adsorbate Modeling
- Build perovskite/fullerene interfaces for solar cell applications
- Automatic surface termination selection (PbI, FAI, MAI, AI)
- Multi-layer stacking (e.g., Perovskite/C70/C60)
- Intelligent supercell generation with minimal adsorbate interactions
- Support for symmetric slabs with consistent terminations
-### ⚡ **Interface Builder**
+### Interface Builder
- Coherent interface matching with automatic strain optimization
- ZSL algorithm for finding commensurate supercells
- Bidirectional or unidirectional strain application
- Customizable Miller indices for both materials
- Comprehensive strain analysis and reporting
-### 📐 **Layer Management**
+### Layer Management
- **Selective Dynamics**: Fix bottom N layers for relaxation calculations
- **Layer Splitting**: Separate multi-layer stacks into individual files
- Automatic layer detection for interface structures
@@ -43,17 +43,20 @@ InterfaceML provides a complete suite of tools for heterojunction modeling with
- Preserves original lattice parameters when splitting
- Compatible with VASP and CP2K
-### 📊 **Density Analysis**
-- Compute charge density differences: Δρ = ρ(interface) - ρ(A) - ρ(B)
+### Density Analysis
+- Compute charge density differences: Delta rho = rho(interface) - rho(A) - rho(B)
- Streaming cube file processing (memory-efficient)
- Direct VESTA visualization support
---
-## 📦 Installation
+## Installation
+
+For an ARM64 CPU environment with the web application and JupyterLab, see
+[Research Containers](docs/DOCKER_RESEARCH.md).
### Prerequisites
-- Python 3.8 or higher
+- Python 3.9 or higher
- pip or conda package manager
### Install from source
@@ -71,14 +74,14 @@ pip install -r requirements.txt
```
### Core Dependencies
-- `numpy >= 1.20.0`
+- `numpy >= 1.22.0`
- `pymatgen >= 2022.0.0`
- `flask >= 2.0.0` (for web interface)
- `flask-cors >= 3.0.0` (for web interface)
---
-## 🚀 Quick Start
+## Quick Start
### Command-Line Interface
@@ -186,9 +189,9 @@ io.write_poscar(structure, "output_fixed.vasp", selective_dynamics=selective_dyn
---
-## 🌐 Web Interface
+## Web Interface
-InterfaceML includes a beautiful, modern web interface for interactive modeling:
+InterfaceML includes a web interface for interactive modeling:
```bash
# Start the web server
@@ -198,66 +201,126 @@ python -m interfaceml.web.app
interfaceml-web
```
-Then open your browser to http://localhost:5000
+### AI Interface Generation (EGNN + Local GNN)
+
+If the fullerene diffusion checkpoint is available, you can generate a
+perovskite/fullerene interface directly through the web API:
+
+#### AI Setup (Required for the AI tab)
+
+The AI module depends on PyTorch + PyTorch Geometric and is not installed
+with the default `requirements.txt`. Recommended Python: 3.10 or 3.11.
+
+```bash
+# Option A: install optional AI dependencies
+pip install -e ".[ai]"
+
+# Option B: install fullerene module requirements directly
+pip install -r requirements.txt
+pip install -r fullerene_e3gen/requirements.txt
+```
+
+If PyTorch/PyG installation fails for your platform, follow their official
+installation guidance for your CUDA/CPU stack.
+
+You can override the AI asset locations with:
+- `INTERFACEML_FULLERENE_PATH`
+- `INTERFACEML_FULLERENE_CHECKPOINT`
+
+To start the web app with a specific Python environment:
+
+```bash
+INTERFACEML_PYTHON=/path/to/python ./start_web.sh
+```
+
+```json
+POST /api/ai/generate-interface
+{
+ "base_filename": "my_base_slab.cif",
+ "num_atoms": 60,
+ "miller": [0, 0, 1],
+ "slab_thickness": 18.0,
+ "vacuum": 20.0,
+ "separation": 3.2,
+ "xy_frac": [0.5, 0.5],
+ "local_refine": true,
+ "refine_scope": "adsorbate"
+}
+```
+
+The response includes a downloadable `POSCAR` (`.vasp`) for the combined interface.
+
+Then open your browser:
+- Documentation homepage: http://localhost:5000/
+- Web app interface: http://localhost:5000/app
-### Web Features:
-- 📁 **Drag-and-drop file upload** for structure files (.cif, .vasp, .xyz)
-- 🎨 **Interactive parameter configuration** with real-time validation
-- 📊 **Visual feedback** with structure information display
-- ⬇️ **Direct download** of generated structure files
-- 🔄 **Multi-tab interface** for different modeling workflows
-- ✂️ **Layer splitting** - Separate multi-layer structures (preserves lattice)
-- 🔧 **Selective dynamics** - Fix layers for relaxation
+### Web Features
+- **Drag-and-drop file upload** for structure files (.cif, .vasp, .xyz)
+- **Interactive parameter configuration** with real-time validation
+- **Visual feedback** with structure information display
+- **Direct download** of generated structure files
+- **Multi-tab interface** for different modeling workflows
+- **Layer splitting** - Separate multi-layer structures (preserves lattice)
+- **Selective dynamics** - Fix layers for relaxation
---
-## 📚 Documentation
+## Documentation
### Directory Structure
```
InterfaceML/
-├── interfaceml/ # Main Python package
-│ ├── core/ # Core functionality modules
-│ │ ├── io.py # Structure I/O (VASP, CIF, XYZ)
-│ │ ├── layering.py # Layer detection and manipulation
-│ │ └── splitting.py # Smart layer splitting
-│ ├── cli/ # Command-line interface tools
-│ │ ├── build_interface.py
-│ │ ├── fix_layers.py
-│ │ └── delta_density.py
-│ ├── web/ # Flask web application
-│ │ ├── app.py # Main Flask app
-│ │ ├── templates/ # HTML templates
-│ │ └── static/ # CSS and JavaScript
-│ └── utils/ # Utility functions
-│ └── geometry.py # Geometric calculations
-├── build_heterojunctions/ # Original scripts (maintained for compatibility)
-│ ├── interface_builder.py
-│ ├── fix_interface_layers.py
-│ ├── delta_density_cube.py
-│ └── README.md # Detailed technical documentation
-├── structures/ # Example structure files
-│ ├── perovskites/
-│ ├── etl/
-│ └── heterojunctions/
-├── examples/ # Tutorial notebooks and scripts
-├── docs/ # Additional documentation
-├── setup.py # Package installation script
-├── requirements.txt # Python dependencies
-└── README.md # This file
+|-- interfaceml/ # Main Python package
+| |-- core/ # Core functionality modules
+| | |-- io.py # Structure I/O (VASP, CIF, XYZ)
+| | |-- layering.py # Layer detection and manipulation
+| | `-- splitting.py # Smart layer splitting
+| |-- cli/ # Command-line interface tools
+| | |-- build_interface.py
+| | |-- fix_layers.py
+| | `-- delta_density.py
+| |-- web/ # Flask web application
+| | |-- app.py # App factory + blueprint registration
+| | |-- routes/ # API blueprints (common, interface, dos, ai)
+| | |-- ai.py # AI module discovery/loading
+| | |-- core.py # Core module availability
+| | |-- dos.py # DOS/PDOS helpers
+| | |-- utils.py # Shared web utilities
+| | |-- templates/ # HTML templates
+| | `-- static/ # CSS and JavaScript
+| `-- utils/ # Utility functions
+| `-- geometry.py # Geometric calculations
+|-- build_heterojunctions/ # Original scripts (maintained for compatibility)
+| |-- interface_builder.py
+| |-- fix_interface_layers.py
+| |-- delta_density_cube.py
+| `-- README.md # Detailed technical documentation
+|-- structures/ # Example structure files
+| |-- perovskites/
+| |-- etl/
+| `-- heterojunctions/
+|-- examples/ # Tutorial notebooks and scripts
+|-- docs/ # Additional documentation
+|-- setup.py # Package installation script
+|-- requirements.txt # Python dependencies
+`-- README.md # This file
```
+Note: `interfaceml/core` is the canonical API surface. The legacy scripts in `build_heterojunctions/`
+delegate to core utilities when available to keep behavior consistent.
+
### Detailed Documentation
- **[build_heterojunctions/README.md](build_heterojunctions/README.md)** - Comprehensive CLI usage guide
- **[docs/GETTING_STARTED.md](docs/GETTING_STARTED.md)** - Tutorial and workflows
- **[docs/LAYER_SPLITTING.md](docs/LAYER_SPLITTING.md)** - Layer splitting guide
- **[docs/SMART_SPLITTING_ALGORITHM.md](docs/SMART_SPLITTING_ALGORITHM.md)** - Smart splitting algorithm details
+- **[docs/MATHEMATICAL_OVERVIEW.md](docs/MATHEMATICAL_OVERVIEW.md)** - Mathematical and algorithmic overview
- **[Tutorials](examples/)** - Example workflows
---
-## 🎯 Use Cases
+## Use Cases
### Solar Cell Modeling
- Perovskite/ETL (C60, TiO2) interfaces
@@ -276,7 +339,7 @@ InterfaceML/
---
-## 🛠️ Advanced Usage
+## Advanced Usage
### Building from Pre-relaxed Structures
@@ -318,7 +381,7 @@ python build_heterojunctions/fix_interface_layers.py \
---
-## 📊 Examples
+## Examples
### Example 1: FAPbI3/C60 Solar Cell Interface
@@ -353,7 +416,7 @@ python build_heterojunctions/delta_density_cube.py \
---
-## 🤝 Contributing
+## Contributing
Contributions are welcome! Please feel free to submit issues, feature requests, or pull requests.
@@ -373,13 +436,13 @@ python -c "from interfaceml.core import io, layering, splitting; print('OK')"
---
-## 📄 License
+## License
This project is licensed under the MIT License - see the LICENSE file for details.
---
-## 📧 Contact
+## Contact
For questions, bug reports, or feature requests:
- Open an issue on [GitHub](https://github.com/yourusername/InterfaceML/issues)
@@ -387,7 +450,7 @@ For questions, bug reports, or feature requests:
---
-## 🙏 Acknowledgments
+## Acknowledgments
- Built with [Pymatgen](https://pymatgen.org/) - Materials analysis library
- Interface matching based on ZSL algorithm
@@ -395,7 +458,7 @@ For questions, bug reports, or feature requests:
---
-## 📖 Citation
+## Citation
If you use InterfaceML in your research, please cite:
@@ -412,8 +475,6 @@ If you use InterfaceML in your research, please cite:
-**Made with ❤️ for the computational materials science community**
-
-⭐ Star us on GitHub if InterfaceML helps your research!
+If InterfaceML supports your research, citations are appreciated.
diff --git a/START_HERE.md b/START_HERE.md
index 43496fcd..cd420aff 100644
--- a/START_HERE.md
+++ b/START_HERE.md
@@ -9,6 +9,7 @@ Minimal navigation for the core package docs.
- Project overview and installation: [`README.md`](README.md)
- 5-minute usage examples (CLI / Web / API): [`QUICK_START.md`](QUICK_START.md)
- End-to-end tutorial: [`docs/GETTING_STARTED.md`](docs/GETTING_STARTED.md)
+- Canonical core APIs live in `interfaceml/core`; legacy CLI scripts in `build_heterojunctions/` delegate to core when available.
---
@@ -19,6 +20,12 @@ Minimal navigation for the core package docs.
---
+## Math and Algorithms
+
+- Mathematical and algorithmic overview: [`docs/MATHEMATICAL_OVERVIEW.md`](docs/MATHEMATICAL_OVERVIEW.md)
+
+---
+
## CLI reference
- Full CLI/options reference: [`build_heterojunctions/README.md`](build_heterojunctions/README.md)
@@ -36,17 +43,18 @@ Minimal navigation for the core package docs.
```
InterfaceML/
-├── README.md
-├── QUICK_START.md
-├── START_HERE.md
-├── docs/
-│ ├── GETTING_STARTED.md
-│ ├── LAYER_SPLITTING.md
-│ └── SMART_SPLITTING_ALGORITHM.md
-├── build_heterojunctions/
-│ └── README.md
-└── templates/
- ├── new_cli_tool_template.py
- ├── new_core_module_template.py
- └── README_templates.md
+|-- README.md
+|-- QUICK_START.md
+|-- START_HERE.md
+|-- docs/
+| |-- GETTING_STARTED.md
+| |-- LAYER_SPLITTING.md
+| |-- MATHEMATICAL_OVERVIEW.md
+| `-- SMART_SPLITTING_ALGORITHM.md
+|-- build_heterojunctions/
+| `-- README.md
+`-- templates/
+ |-- new_cli_tool_template.py
+ |-- new_core_module_template.py
+ `-- README_templates.md
```
diff --git a/_tmp_poscar_write_check/layer1.vasp b/_tmp_poscar_write_check/layer1.vasp
deleted file mode 100644
index a2711ebe..00000000
--- a/_tmp_poscar_write_check/layer1.vasp
+++ /dev/null
@@ -1,188 +0,0 @@
-layer1
-1.0
- 12.7047439999999998 0.0000000000000000 0.0000000000000000
- 0.0000000000000000 12.9768600000000003 0.0000000000000000
- 0.0000000000000000 0.0000000000000000 57.1962619999999973
-H C N I Pb
-80 16 32 40 12
-direct
- 0.7380540001435685 0.8943269997518660 0.1719620000691654 H
- 0.1068449997890552 0.1280770001371672 0.1749900000458072 H
- 0.5742700002455775 0.8245450001001784 0.1760040000516118 H
- 0.5455860000012593 0.3245140003051586 0.1765500000681863 H
- 0.1408470001441981 0.7467599997225831 0.1768259999228621 H
- 0.1485099998866565 0.6107129999090689 0.1782229999575846 H
- 0.1145999998110942 0.2640040001972743 0.1786410000010140 H
- 0.7050909998658769 0.4009210001494969 0.1799390000346526 H
- 0.8889059999949626 0.8043220000832252 0.1813419999369889 H
- 0.5915009999414391 0.1983109997333715 0.1819839999334222 H
- 0.6199709998092051 0.6998429997703606 0.1828059999795091 H
- 0.8188170001693856 0.6914769998289263 0.1885249999379330 H
- 0.2688610002688759 0.0980270003683480 0.1914089999447866 H
- 0.8531959998564316 0.3113109997333716 0.1918900000842712 H
- 0.7860839998035379 0.1934699996763470 0.1945049999945801 H
- 0.2748129997739427 0.3246769996748057 0.2007120000254562 H
- 0.2617260001460872 0.8451259996640174 0.2011869999476539 H
- 0.2885859998438379 0.6168679996547701 0.2047460000095810 H
- 0.3641390003608101 0.2301869997826901 0.2089790000612278 H
- 0.3486250002361322 0.7741850000693542 0.2189609999338768 H
- 0.7990260000516343 0.7067149996224048 0.2638279999836353 H
- 0.2474660001020091 0.2617490001433321 0.2683870000105951 H
- 0.8056280000604499 0.2955970003529360 0.2701839999613961 H
- 0.3107440000365218 0.6291310001032607 0.2763620000202111 H
- 0.6759999996851570 0.7242819996516877 0.2770870000210853 H
- 0.6959599996662663 0.3580279998397147 0.2835279999591581 H
- 0.1955089996303743 0.1604989997580309 0.2852749999641585 H
- 0.1916520002292057 0.6043519996362756 0.2909649999505213 H
- 0.3340059996486351 0.7878030001094255 0.2935199999608366 H
- 0.8919700003400304 0.7243559998335499 0.2979840000732915 H
- 0.2764100000755623 0.3671109998874920 0.2999710000279389 H
- 0.8354080003501054 0.1927370003221118 0.3027519999471294 H
- 0.1350849997449771 0.7430589996347345 0.3200250000603186 H
- 0.2147849999968516 0.8518260002804994 0.3210580000839915 H
- 0.6828549996757116 0.7551869997826901 0.3216049999910833 H
- 0.6531110001114544 0.3075219999291046 0.3236040000306314 H
- 0.2023679997015288 0.1892720003144058 0.3291790000192670 H
- 0.8110899999244376 0.7523869996285697 0.3333539999519549 H
- 0.7294370000686359 0.2099989997580308 0.3362429999359049 H
- 0.2376349999653673 0.3139280003020762 0.3380249999554167 H
- 0.7173819999836282 0.6693860001572028 0.3823849999148546 H
- 0.1585680002682462 0.7587650001618266 0.3856450000176585 H
- 0.2593860002216495 0.8495599998767037 0.3858119999170575 H
- 0.6240360002531338 0.7658260002804994 0.3871049999735997 H
- 0.2380759997997598 0.1285889999583875 0.3929389999297506 H
- 0.1332149998457269 0.2151030002635461 0.3940549999578644 H
- 0.6240309997588303 0.2928730000940135 0.3962969999682846 H
- 0.7208579999722938 0.3872889998042670 0.3977599999454510 H
- 0.8164339997720536 0.7170780003791365 0.4151960000462968 H
- 0.6612410002122041 0.1279650000077060 0.4186449999127565 H
- 0.1531389998885456 0.6358309999491403 0.4200970000452127 H
- 0.3304000001889058 0.7888670001834035 0.4209740000141967 H
- 0.3357560002783212 0.2173220000832251 0.4211329999152741 H
- 0.1544010001303450 0.3657870000909311 0.4217770000424153 H
- 0.8298629999943329 0.2891880000246593 0.4220200000132876 H
- 0.6459869998167613 0.8736410002111451 0.4231019999174072 H
- 0.7807109997651270 0.1245399996609349 0.4336660000613327 H
- 0.2704400002077964 0.3667390000354477 0.4380960000497934 H
- 0.7583099997922036 0.8477140001510380 0.4399159999302052 H
- 0.2517749999527735 0.6531449996378169 0.4409619999292961 H
- 0.6957980003375117 0.6636209999953765 0.4950479999899294 H
- 0.1924179999219189 0.1622900000462361 0.4971970000417160 H
- 0.5911219997821286 0.7474410003652656 0.4996859999347509 H
- 0.0914500000944529 0.2484329999707171 0.5030199999783203 H
- 0.6980229999124736 0.1489780000708954 0.5049260000592346 H
- 0.1867790000333734 0.6353990002203923 0.5118090000357016 H
- 0.8321839999294751 0.2615629998320087 0.5147200000587451 H
- 0.3270610002059073 0.7476150000847663 0.5161789999493324 H
- 0.6056400003022493 0.2096640003822188 0.5230139999358699 H
- 0.7979679999848875 0.7352690003591008 0.5250220000390934 H
- 0.3052520003551429 0.2336790001587441 0.5251400000580457 H
- 0.0999450000724139 0.7136030002635461 0.5277350000599689 H
- 0.6191949999149924 0.8841359997719018 0.5316879999955242 H
- 0.1302210001240481 0.3843420002989937 0.5340890000818584 H
- 0.3073540002065370 0.9062690003591009 0.5347920000436392 H
- 0.8050060001208996 0.4011239999506815 0.5389319999618156 H
- 0.1719010001303450 0.8826579997010062 0.5406729999243657 H
- 0.6695610002059073 0.3716099996455229 0.5429590000129729 H
- 0.7441040000491156 0.8786239999506815 0.5446180000364360 H
- 0.2599930002525041 0.3804180001941919 0.5448000000419607 H
- 0.7318699998992503 0.8133320001911094 0.1767359999854536 C
- 0.7003439998476160 0.3177379997934786 0.1825500000332189 C
- 0.2400469997663865 0.1770699999845879 0.1901169999535983 C
- 0.2552579996889351 0.6881980001325436 0.1979179999560111 C
- 0.2657419999962219 0.7418939997811489 0.2984319999792993 C
- 0.8063999998740629 0.7304579998551268 0.2986289999161134 C
- 0.2447709997147522 0.2901429999244809 0.3029700000325196 C
- 0.7713259999571813 0.2489549998998217 0.3030560000581856 C
- 0.7440079996889352 0.7584050001310024 0.4110900000772778 C
- 0.2607849996820085 0.7518649996994651 0.4136109999635990 C
- 0.2587079999408095 0.2460979998243027 0.4158659999494372 C
- 0.7546129996794898 0.2578489996809706 0.4162360000029373 C
- 0.7189830003658475 0.7651030002635461 0.5217269999217782 C
- 0.7550149999086956 0.2719950003313590 0.5228269999532487 C
- 0.2258510002247979 0.2646999998458796 0.5230779999224425 C
- 0.2516290001593106 0.7670650003159470 0.5241960000812641 C
- 0.6362129996480055 0.7743269997518660 0.1779190000213650 N
- 0.6073049996127431 0.2749959998027258 0.1791930000600389 N
- 0.1495539996713039 0.1923589997888549 0.1793800000426601 N
- 0.8201830001454574 0.7625939996270286 0.1806089999377931 N
- 0.1804279999659969 0.6814219996208636 0.1819759999351007 N
- 0.7867059997430881 0.2690629998320087 0.1888020000327993 N
- 0.3001039997342725 0.2502069999984588 0.1989099999227222 N
- 0.2935430001580512 0.7756969998905745 0.2058300000444085 N
- 0.7559949999779610 0.7186049999768819 0.2786870000350722 N
- 0.7567959999823688 0.3057750002697108 0.2842979999287366 N
- 0.2310519999458470 0.2313649996994651 0.2844559999392968 N
- 0.2564860000327437 0.6506329998165967 0.2885899999898595 N
- 0.1994079998778409 0.7815100001078844 0.3136440000222392 N
- 0.7627719999710344 0.7494340002126864 0.3190240000299320 N
- 0.7134529999187704 0.2558809997179595 0.3221700000255261 N
- 0.2249159998816190 0.2625499997688193 0.3246980000196516 N
- 0.6934880002304651 0.7324560001417909 0.3919130000138820 N
- 0.2251399996725632 0.7870070001525794 0.3935190000703193 N
- 0.2059389996366711 0.1932279996855942 0.4001690000301069 N
- 0.6939390002663572 0.3173659999414343 0.4034209999597526 N
- 0.7308709998406894 0.1633190001279200 0.4228520000135673 N
- 0.2183970003645882 0.6757439997040887 0.4255600000573464 N
- 0.7131680000793404 0.8308000001541206 0.4256919999422339 N
- 0.2250849997449771 0.3315450001001783 0.4257690000790611 N
- 0.6625319998576910 0.7195589996347345 0.5054960000707739 N
- 0.1630050003368820 0.2186489998350911 0.5081060000039862 N
- 0.6809769996152618 0.2062059997564896 0.5167930000390585 N
- 0.1748370002575416 0.6994329999707172 0.5220379999658019 N
- 0.6876720003173618 0.8435040001972742 0.5347449999442271 N
- 0.2438219998765815 0.8564539996578525 0.5349799999167778 N
- 0.1998520001662371 0.3443179998859509 0.5362370000682911 N
- 0.7433879997896848 0.3495330002789581 0.5375299999499967 N
- 0.0293360000012594 0.4461069996902178 0.1800490000552833 I
- 0.4826539999546626 0.0090770001371672 0.1813129999649278 I
- 0.4857209999666267 0.5246389997272067 0.1834499999318137 I
- 0.0602520003551429 0.9305460003421476 0.1836660000613327 I
- 0.7809949999779610 0.0199099997996434 0.2148830000464017 I
- 0.7703700003715147 0.5608480002096039 0.2219220000775575 I
- 0.9566909998343925 0.2571220002373455 0.2401309999943703 I
- 0.2666500001889058 0.4460909996717233 0.2420420000523811 I
- 0.5089640003765523 0.2296600001849446 0.2422859999137706 I
- 0.2803929996543024 0.9887889998042670 0.2504839999858732 I
- 0.4847479996448570 0.7263950000231181 0.2512489999783552 I
- 0.0209449997575709 0.7452699998304676 0.2517470000050003 I
- 0.6627380000730435 0.9898440000123296 0.2870070000378696 I
- 0.0198480000856373 0.4894029996470641 0.2936550000417860 I
- 0.5260459998249473 0.4641899997379952 0.2942970000382192 I
- 0.0080209998721737 0.0390480000554834 0.2945890000643748 I
- 0.4932820000151125 0.7846459998797861 0.3401709999859781 I
- 0.2333369999426986 0.0071470001217552 0.3522059999305549 I
- 0.9912989998066865 0.7261459998797861 0.3564610000212951 I
- 0.7322069999993703 0.5070170002604637 0.3582359999330026 I
- 0.7099629996480055 0.0217319998828684 0.3630560000581856 I
- 0.4761820002040182 0.2616440001664501 0.3638500000227288 I
- 0.2169249998268363 0.4784250003467711 0.3674279999276875 I
- 0.9615269996782304 0.2458999996917590 0.3685789999703127 I
- 0.4779729996920836 0.5436150000847664 0.4053590000339533 I
- 0.9810720003488460 0.9726799996301110 0.4055589999919925 I
- 0.4814839999924437 0.9972279996855943 0.4174220000600739 I
- 0.9863230002902852 0.4969410003652656 0.4182060000704242 I
- 0.7136650002550229 0.5402259999722584 0.4595910000202461 I
- 0.4935640001876464 0.2607750002697109 0.4596560000721726 I
- 0.9756999999370315 0.7554650000077061 0.4607669999483533 I
- 0.2114189998633581 0.0426299998612916 0.4613239999844745 I
- 0.9398819999836281 0.2524980002866641 0.4711570000850755 I
- 0.4301539999546626 0.7591980001325437 0.4718810000205957 I
- 0.7389869996593398 0.9955160000184945 0.4771399999881111 I
- 0.2369370000686358 0.4997319998828685 0.4784870000770330 I
- 0.4626479998337629 0.0238019998674564 0.5295179999699980 I
- 0.9647729997550522 0.5090770001371673 0.5301660000088817 I
- 0.4738849999653673 0.4231549997457013 0.5362509999342264 I
- 0.9676360003790709 0.9065909996717234 0.5375519999890902 I
- 0.5556309997273460 0.0031589999429754 0.2341749999326879 Pb
- 0.0204890000144828 0.4775930001556617 0.2344059999934961 Pb
- 0.5370369997223085 0.5018960002650873 0.2375029999338069 Pb
- 0.0415190003041384 0.9716020000215769 0.2417070000133925 Pb
- 0.5245500003778116 0.0128269997518660 0.3299740000491641 Pb
- 0.9918389996681555 0.4849620000524010 0.3469199999468497 Pb
- 0.9800529998872861 0.0139259998181378 0.3544310000538147 Pb
- 0.4789819999521439 0.4934760003575596 0.3544340000750398 Pb
- 0.4448500001259372 0.9974259998181380 0.4712950000124134 Pb
- 0.9469320003614399 0.4905270003683480 0.4719569999172323 Pb
- 0.4819289999074362 0.4660079996239461 0.4832089999168127 Pb
- 0.9791460001083060 0.9583499999229398 0.4849840000033569 Pb
diff --git a/_tmp_poscar_write_check/layer2.vasp b/_tmp_poscar_write_check/layer2.vasp
deleted file mode 100644
index 5d0afcea..00000000
--- a/_tmp_poscar_write_check/layer2.vasp
+++ /dev/null
@@ -1,78 +0,0 @@
-layer2
-1.0
- 12.7047439999999998 0.0000000000000000 0.0000000000000000
- 0.0000000000000000 12.9768600000000003 0.0000000000000000
- 0.0000000000000000 0.0000000000000000 57.1962619999999973
-C
-70
-direct
- 0.4999229996291149 0.4523269997518660 0.5797669999833206 C
- 0.5000169999489954 0.5655110003498536 0.5806130000243722 C
- 0.5949479997393100 0.3990179997318304 0.5849399999251699 C
- 0.4048030003595507 0.3991939999352694 0.5849459999676202 C
- 0.5951680002367620 0.6169850002234748 0.5865680000207008 C
- 0.4049800003841085 0.6171489998350911 0.5865779999748935 C
- 0.6921399998299848 0.4536459998797861 0.5893940000834320 C
- 0.3077140003765522 0.4539750001155904 0.5894120000359464 C
- 0.6922460003916647 0.5610019997133360 0.5902029999792643 C
- 0.3078050002424291 0.5613279999938352 0.5902180000853902 C
- 0.5948840000239281 0.3071700002928290 0.5990510000461219 C
- 0.4047069999993703 0.3073429997703605 0.5990549999578644 C
- 0.5952659998501347 0.7045429996162400 0.6020230000345127 C
- 0.4050710002499854 0.7047069999984589 0.6020299999674803 C
- 0.7533410000232983 0.3952810001803210 0.6061609999968179 C
- 0.2464010002877665 0.3957110001957330 0.6061780000588151 C
- 0.7535530003595508 0.6142659996331933 0.6078029999582839 C
- 0.2466180003312149 0.6147040000431537 0.6078210000856350 C
- 0.4997570002197604 0.2663540001202140 0.6083429999324081 C
- 0.5002100002959524 0.7427099999537639 0.6119110000580108 C
- 0.6920320000151124 0.3050620003606420 0.6122270000791310 C
- 0.3075829997046772 0.3053990002203923 0.6122420000104202 C
- 0.6923909997714240 0.7026180000400714 0.6152090000217146 C
- 0.3079539997027882 0.7029519999445167 0.6152219999971327 C
- 0.8122119997065663 0.4463950000231180 0.6232970000731866 C
- 0.1876390000459671 0.4469039998890332 0.6233219999586687 C
- 0.8123180002682463 0.5581999998458795 0.6241409999835303 C
- 0.1877480002745431 0.5587189998196791 0.6241559999148196 C
- 0.4997559998060567 0.2278040003513947 0.6324990000220644 C
- 0.6920179997330131 0.2685249998844096 0.6351439999697882 C
- 0.3075659997556818 0.2688480002096039 0.6351580000105601 C
- 0.5002409997399396 0.7741340000585658 0.6365939999715365 C
- 0.6924380003249180 0.7324219996208637 0.6386239999390170 C
- 0.3079950001353825 0.7327469996593938 0.6386360000239176 C
- 0.5948659996612289 0.2329639997657369 0.6455710000768931 C
- 0.4046739997279756 0.2331280001479557 0.6455769999445068 C
- 0.8121839999294751 0.4083060000647306 0.6471629999177219 C
- 0.1876289998444675 0.4088480002096039 0.6471909999992657 C
- 0.8123489997122334 0.5892040000431538 0.6485290000245121 C
- 0.1878070002827290 0.5897659996331932 0.6485409999345760 C
- 0.5953269998986205 0.7649810000262005 0.6495590000269598 C
- 0.4051680002367619 0.7651530000323653 0.6495689999811526 C
- 0.7533119998325035 0.3207029998011846 0.6529129999789147 C
- 0.2463749997638677 0.3211410002111451 0.6529289999755578 C
- 0.7536199997418287 0.6750450001001784 0.6555689999461853 C
- 0.2467189996114837 0.6754859996948414 0.6555840000523111 C
- 0.8122849999968516 0.4965640000739779 0.6627590000549336 C
- 0.1877199997103444 0.4971440001664502 0.6627780000728020 C
- 0.5948860000642280 0.2637340003668068 0.6697520000520314 C
- 0.4047250003620695 0.2639110000416126 0.6697550000732566 C
- 0.5953030002021292 0.7272319998828686 0.6732270000441638 C
- 0.4051629997424585 0.7273990002203924 0.6732350000424853 C
- 0.6920690003671069 0.3182829998936569 0.6742650000449331 C
- 0.3076419997128632 0.3186190002820405 0.6742780000203510 C
- 0.6923880001045279 0.6713359996177812 0.6769140000792360 C
- 0.3079700000251874 0.6716820001140491 0.6769259999892999 C
- 0.4998290000963419 0.2900800000924723 0.6814580000000700 C
- 0.7534920003110649 0.4935949998689976 0.6834520000625215 C
- 0.2465490001215294 0.4940450001001783 0.6834650000379395 C
- 0.5002159996297446 0.6976739997195007 0.6845149999487730 C
- 0.6921629998998798 0.4030260001263788 0.6892419999404856 C
- 0.3077319999521438 0.4033749998073494 0.6892499999388071 C
- 0.6923489997122335 0.5824029996470641 0.6905829999869572 C
- 0.3078669999175112 0.5827439997040886 0.6905940000065039 C
- 0.4999300001637184 0.3794519999445167 0.6972389999891950 C
- 0.5001249997638677 0.6039359999260222 0.6989310000712983 C
- 0.5950610000484858 0.4358260002804993 0.7001469999910134 C
- 0.4049009999729235 0.4359859996948414 0.7001549999893349 C
- 0.5951810001051576 0.5467119996670998 0.7009789999912932 C
- 0.4049800003841085 0.5468659999414343 0.7009830000778722 C
diff --git a/_tmp_poscar_write_check/layer3.vasp b/_tmp_poscar_write_check/layer3.vasp
deleted file mode 100644
index 5d06d3c1..00000000
--- a/_tmp_poscar_write_check/layer3.vasp
+++ /dev/null
@@ -1,68 +0,0 @@
-layer3
-1.0
- 12.7047439999999998 0.0000000000000000 0.0000000000000000
- 0.0000000000000000 12.9768600000000003 0.0000000000000000
- 0.0000000000000000 0.0000000000000000 57.1962619999999973
-C
-60
-direct
- 0.5019459998564315 0.4939389998813273 0.7359500000192321 C
- 0.5698699997418287 0.5814560001417909 0.7408010000723474 C
- 0.3941569999364017 0.5245909996717234 0.7408250000673121 C
- 0.5356239999798500 0.3940339997503248 0.7413060000319601 C
- 0.3954650003179915 0.6310539999660936 0.7486939999680398 C
- 0.5040660000705248 0.6661739997195006 0.7486939999680398 C
- 0.6689810003255476 0.5657750002697108 0.7508600000818235 C
- 0.3240459996675258 0.4542639999198573 0.7508920000751098 C
- 0.6385620001473465 0.3778090000200356 0.7517530000474506 C
- 0.4628309999792203 0.3209950003313590 0.7517669999133859 C
- 0.7040159998501347 0.4620450001001783 0.7564359999609764 C
- 0.3590470000812295 0.3505450001001783 0.7564770000179383 C
- 0.5398759998627285 0.7321890002666285 0.7663280000710535 C
- 0.3266989999955922 0.6632279996855943 0.7663520000660182 C
- 0.6294209997462366 0.2947089997117947 0.7686589999535284 C
- 0.5208079997519036 0.2596039997349128 0.7686639999306248 C
- 0.7061500003463274 0.6343160001726149 0.7691740000421706 C
- 0.2526030001076763 0.4876709997641957 0.7692259999438424 C
- 0.6428030002021292 0.7159229998628328 0.7767720000653190 C
- 0.2538970002071667 0.5901920002219335 0.7768060000144764 C
- 0.7627500003148430 0.4664650000077061 0.7782099999821666 C
- 0.3091999996221884 0.3198179998859509 0.7782529999949996 C
- 0.4684290002222792 0.7655710002265571 0.7846640000704941 C
- 0.3638469999867766 0.7317439997040887 0.7846729999593330 C
- 0.7640979999282158 0.5729370001679913 0.7860779999923772 C
- 0.2434250001416794 0.4045939996270284 0.7861339999806281 C
- 0.6859889998570613 0.2989739998736212 0.7896290000559826 C
- 0.4728350000598202 0.2300320000369889 0.7896420000314006 C
- 0.7539670000434483 0.3864359999260222 0.7944870000420656 C
- 0.3650670001693856 0.2607259999722583 0.7945220000565771 C
- 0.6349839996776008 0.7393109997333716 0.8015559999358000 C
- 0.2460569996530430 0.6135510000107884 0.8015879999290862 C
- 0.5271920000906748 0.7699430000786015 0.8064350000704591 C
- 0.3140000003148430 0.7010490002974527 0.8064510000671022 C
- 0.7565660000705249 0.5953920000678130 0.8099479999234914 C
- 0.2358749999212892 0.4270789998505031 0.8100030000212252 C
- 0.6361809997903146 0.2682530003406063 0.8114100000101405 C
- 0.5315959998879158 0.2344460000339065 0.8114149999872370 C
- 0.6907779999345126 0.6801570002296395 0.8178249999973775 C
- 0.2372449999779610 0.5335429996162400 0.8178699999660817 C
- 0.7461109999540330 0.4098139996886766 0.8192700000220293 C
- 0.3572090000396702 0.2840740001818622 0.8193080000577659 C
- 0.7473819999836282 0.5123179998859508 0.8268600000468562 C
- 0.2938619998954721 0.3656890002666285 0.8269000000384640 C
- 0.4791880001674965 0.7404150002388867 0.8274129999963983 C
- 0.3705920001221591 0.7052829998936568 0.8274170000829775 C
- 0.6733159999131033 0.3367880003329003 0.8297289999475840 C
- 0.4601399996725632 0.2678179998859508 0.8297499999213235 C
- 0.6409389996366711 0.6494850002234748 0.8396080000822430 C
- 0.2959949999779610 0.5379280003020762 0.8396450000526258 C
- 0.5371339997090850 0.6789850002234747 0.8443080000577660 C
- 0.3613930001265669 0.6222029998011845 0.8443299999220228 C
- 0.6759259997682756 0.5457550000539422 0.8451880000479752 C
- 0.3310160000075562 0.4341990003745128 0.8452209999317788 C
- 0.6045049998646175 0.3689099997996434 0.8473849999148546 C
- 0.4958899998299847 0.3338190001279200 0.8473920000226588 C
- 0.4643330003343633 0.6059500002311807 0.8547700000045457 C
- 0.6058760003349930 0.4753689998967393 0.8552550000557728 C
- 0.4301410000862670 0.4185930001556617 0.8552790000507376 C
- 0.4981020003236586 0.5060740001818622 0.8601309999943703 C
diff --git a/active_learning/__init__.py b/active_learning/__init__.py
new file mode 100644
index 00000000..81d8630c
--- /dev/null
+++ b/active_learning/__init__.py
@@ -0,0 +1 @@
+"""Active learning pipeline: AI generation -> Interface building -> DFT -> Retrain."""
diff --git a/active_learning/__main__.py b/active_learning/__main__.py
new file mode 100644
index 00000000..992daaec
--- /dev/null
+++ b/active_learning/__main__.py
@@ -0,0 +1,13 @@
+"""Allow running as: python -m active_learning"""
+
+import logging
+
+logging.basicConfig(
+ level=logging.INFO,
+ format="%(asctime)s [%(levelname)s] %(name)s: %(message)s",
+ datefmt="%Y-%m-%d %H:%M:%S",
+)
+
+from .pipeline import main
+
+main()
diff --git a/active_learning/build_interfaces.py b/active_learning/build_interfaces.py
new file mode 100644
index 00000000..2e983f41
--- /dev/null
+++ b/active_learning/build_interfaces.py
@@ -0,0 +1,168 @@
+"""Step 2: Build fullerene/perovskite interface structures."""
+
+import json
+import logging
+import subprocess
+import sys
+from pathlib import Path
+from typing import List
+
+logger = logging.getLogger(__name__)
+
+
+def build_interfaces(
+ fullerene_dir: Path,
+ perovskite_dir: Path,
+ output_dir: Path,
+ config: dict,
+) -> List[dict]:
+ """Build interfaces using batch_adsorbate_builder.
+
+ Args:
+ fullerene_dir: Directory containing generated fullerene .xyz files.
+ perovskite_dir: Directory containing generated perovskite .cif files.
+ output_dir: Directory for output interface POSCAR files.
+ config: Full pipeline configuration dict.
+
+ Returns:
+ List of metadata dicts from metadata.jsonl.
+ """
+ output_dir.mkdir(parents=True, exist_ok=True)
+ iface_cfg = config["interface"]
+
+ miller = ",".join(str(m) for m in iface_cfg["miller"])
+ xy_grid = ",".join(str(g) for g in iface_cfg.get("xy_grid", [2, 2]))
+
+ # Build command for batch_adsorbate_builder
+ builder_script = (
+ Path(__file__).resolve().parent.parent
+ / "build_heterojunctions"
+ / "batch_adsorbate_builder.py"
+ )
+
+ cmd = [
+ sys.executable,
+ str(builder_script),
+ "--perovskite_dir", str(perovskite_dir),
+ "--fullerene_dir", str(fullerene_dir),
+ "--output_dir", str(output_dir),
+ "--miller", miller,
+ "--slab_thickness", str(iface_cfg.get("slab_thickness", 18.0)),
+ "--vacuum", str(iface_cfg.get("vacuum", 20.0)),
+ "--distance", str(iface_cfg.get("distance", 3.2)),
+ "--orientation_samples", str(iface_cfg.get("orientation_samples", 4)),
+ "--xy_grid", xy_grid,
+ "--seed", str(config["pipeline"].get("seed", 42)),
+ ]
+
+ max_interfaces = iface_cfg.get("max_interfaces")
+ if max_interfaces:
+ cmd.extend(["--max_pairs", str(max_interfaces)])
+
+ logger.info("Running batch_adsorbate_builder: %s", " ".join(cmd))
+
+ result = subprocess.run(
+ cmd,
+ capture_output=True,
+ text=True,
+ cwd=str(Path(__file__).resolve().parent.parent),
+ )
+
+ if result.returncode != 0:
+ logger.error("batch_adsorbate_builder failed:\n%s", result.stderr)
+ raise RuntimeError(
+ f"batch_adsorbate_builder exited with code {result.returncode}: "
+ f"{result.stderr[:500]}"
+ )
+
+ if result.stdout:
+ logger.info("Builder output:\n%s", result.stdout[:1000])
+
+ # Read metadata
+ metadata_path = output_dir / "metadata.jsonl"
+ records = _read_metadata(metadata_path)
+
+ # Apply selective dynamics (fix bottom layers)
+ fix_layers = config.get("dft", {}).get("fix_bottom_layers", 2)
+ if fix_layers and fix_layers > 0:
+ records = _apply_selective_dynamics(records, fix_layers)
+
+ successful = [r for r in records if r.get("status") == "success"]
+ logger.info(
+ "Built %d interfaces (%d successful, %d filtered/failed)",
+ len(records),
+ len(successful),
+ len(records) - len(successful),
+ )
+
+ return records
+
+
+def _read_metadata(metadata_path: Path) -> List[dict]:
+ """Read JSONL metadata file."""
+ records = []
+ if not metadata_path.exists():
+ logger.warning("No metadata file found at %s", metadata_path)
+ return records
+
+ with metadata_path.open("r", encoding="utf-8") as f:
+ for line in f:
+ line = line.strip()
+ if line:
+ records.append(json.loads(line))
+ return records
+
+
+def _apply_selective_dynamics(records: List[dict], n_fix_layers: int) -> List[dict]:
+ """Apply selective dynamics to fix bottom layers of perovskite slab."""
+ sys.path.insert(
+ 0,
+ str(Path(__file__).resolve().parent.parent / "build_heterojunctions"),
+ )
+ from fix_interface_layers import print_fixed_atoms_by_z_layers
+
+ from interfaceml.core.io import load_structure, write_poscar
+
+ for record in records:
+ if record.get("status") != "success":
+ continue
+
+ poscar_path = record.get("output_path")
+ if not poscar_path or not Path(poscar_path).exists():
+ continue
+
+ try:
+ structure = load_structure(poscar_path)
+
+ fixed_indices = print_fixed_atoms_by_z_layers(
+ str(poscar_path),
+ n_fix_layers=n_fix_layers,
+ tol=None,
+ one_based_output=False,
+ gap_cut=True,
+ )
+
+ # Build selective dynamics flags: fixed atoms get (F,F,F)
+ fixed_set = set(fixed_indices)
+ selective_dynamics = [
+ (False, False, False) if i in fixed_set else (True, True, True)
+ for i in range(len(structure))
+ ]
+
+ write_poscar(
+ structure,
+ poscar_path,
+ selective_dynamics=selective_dynamics,
+ comment=f"Interface with {len(fixed_indices)} fixed atoms",
+ )
+
+ record["n_fixed_atoms"] = len(fixed_indices)
+
+ except Exception as exc:
+ logger.warning(
+ "Failed to apply selective dynamics to %s: %s",
+ poscar_path,
+ exc,
+ )
+
+ return records
diff --git a/active_learning/config.yaml b/active_learning/config.yaml
new file mode 100644
index 00000000..0186a0ba
--- /dev/null
+++ b/active_learning/config.yaml
@@ -0,0 +1,56 @@
+pipeline:
+ round: 0 # Current active learning round
+ base_dir: "active_learning_runs"
+ seed: 42
+
+generation:
+ fullerene:
+ checkpoint: "fullerene_e3gen/checkpoints/best_model.pt"
+ C_values: [60, 70, 80] # Carbon counts to generate
+ num_samples_per_C: 5
+ perovskite:
+ checkpoint: "perovskite_e3gen/checkpoints/best_model.pt"
+ compositions: # A/B/X combinations
+ - {A: Ba, B: Ti, X: O}
+ - {A: Cs, B: Pb, X: I}
+ - {A: Cs, B: Pb, X: Br}
+ num_atoms: 5
+ num_samples_per_comp: 5
+
+interface:
+ miller: [0, 0, 1]
+ slab_thickness: 18.0
+ vacuum: 20.0
+ distance: 3.2
+ orientation_samples: 4
+ xy_grid: [2, 2]
+ max_interfaces: 500 # Cap total interfaces per round
+
+dft:
+ code: cp2k
+ basis_set: DZVP-MOLOPT-SR-GTH
+ potential: GTH-PBE
+ functional: PBE
+ cutoff: 400 # Ry
+ rel_cutoff: 60 # Ry
+ max_scf: 300
+ geo_opt_max_iter: 200
+ geo_opt_convergence: 3.0e-3 # Ha/Bohr
+ dispersion: DFT-D3 # vdW correction (important for interfaces)
+ fix_bottom_layers: 2
+
+slurm:
+ partition: "gpu"
+ nodes: 1
+ ntasks_per_node: 48
+ time: "24:00:00"
+ account: null # Set if needed
+ cp2k_module: "cp2k/2024.1" # module load command
+ cp2k_binary: "cp2k.psmp"
+
+retraining:
+ energy_filter: -1.0 # Keep structures with binding energy < this (eV)
+ force_threshold: 0.5 # Max force for "converged" (Ha/Bohr)
+ min_structures: 50 # Min structures before retraining
+ retrain_fullerene: false # Usually not needed
+ retrain_perovskite: true # Retrain with DFT-relaxed slab geometries
diff --git a/active_learning/dft_inputs.py b/active_learning/dft_inputs.py
new file mode 100644
index 00000000..1676ded0
--- /dev/null
+++ b/active_learning/dft_inputs.py
@@ -0,0 +1,362 @@
+"""Step 3: Generate CP2K input files and SLURM submission scripts."""
+
+import json
+import logging
+from pathlib import Path
+from typing import Dict, List, Optional, Tuple
+
+import numpy as np
+
+logger = logging.getLogger(__name__)
+
+# CP2K basis set + potential mappings for common interface elements
+# Format: element -> (basis_set, potential)
+_DEFAULT_BASIS_POTENTIAL = {
+ # Fullerene
+ "C": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q4"),
+ "H": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q1"),
+ # Perovskite A-site
+ "Ba": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q10"),
+ "Cs": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q9"),
+ "Rb": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q9"),
+ "K": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q9"),
+ "Na": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q9"),
+ "Ma": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q1"), # Methylammonium proxy
+ # Perovskite B-site
+ "Ti": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q12"),
+ "Pb": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q4"),
+ "Sn": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q4"),
+ "Ge": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q4"),
+ "Zr": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q12"),
+ # Perovskite X-site
+ "O": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q6"),
+ "I": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q7"),
+ "Br": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q7"),
+ "Cl": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q7"),
+ "F": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q7"),
+ "N": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q5"),
+ "S": ("DZVP-MOLOPT-SR-GTH", "GTH-PBE-q6"),
+}
+
+# Hartree to eV conversion
+HA_TO_EV = 27.211386245988
+
+
+def prepare_dft_calculations(
+ interface_records: List[dict],
+ output_base: Path,
+ config: dict,
+) -> List[dict]:
+ """Generate CP2K input + SLURM script for each interface.
+
+ Args:
+ interface_records: List of metadata dicts from build_interfaces step.
+ output_base: Base directory for DFT jobs (e.g., round_dir/dft_jobs/).
+ config: Full pipeline configuration dict.
+
+ Returns:
+ List of job dicts with paths to generated files.
+ """
+ output_base.mkdir(parents=True, exist_ok=True)
+ dft_cfg = config["dft"]
+ slurm_cfg = config["slurm"]
+
+ # Load templates
+ template_dir = Path(__file__).resolve().parent / "templates"
+ geo_opt_template = (template_dir / "cp2k_geo_opt.inp").read_text()
+ slurm_template = (template_dir / "slurm_cp2k.sh").read_text()
+
+ # Filter to successful interfaces only
+ successful = [r for r in interface_records if r.get("status") == "success"]
+ if not successful:
+ logger.warning("No successful interfaces to prepare DFT for")
+ return []
+
+ jobs = []
+ for idx, record in enumerate(successful, 1):
+ poscar_path = Path(record["output_path"])
+ if not poscar_path.exists():
+ logger.warning("POSCAR not found: %s", poscar_path)
+ continue
+
+ # Create job directory
+ perovskite_name = Path(record.get("perovskite_path", "unknown")).stem
+ fullerene_name = Path(record.get("fullerene_path", "unknown")).stem
+ job_name = f"{idx:04d}_{perovskite_name}_{fullerene_name}"
+ # Truncate long names
+ if len(job_name) > 80:
+ job_name = job_name[:80]
+ job_dir = output_base / job_name
+ job_dir.mkdir(parents=True, exist_ok=True)
+
+ try:
+ job_info = _prepare_single_job(
+ poscar_path=poscar_path,
+ job_dir=job_dir,
+ job_name=job_name,
+ geo_opt_template=geo_opt_template,
+ slurm_template=slurm_template,
+ dft_cfg=dft_cfg,
+ slurm_cfg=slurm_cfg,
+ record=record,
+ )
+ jobs.append(job_info)
+ except Exception as exc:
+ logger.error("Failed to prepare job %s: %s", job_name, exc)
+ jobs.append({
+ "job_id": job_name,
+ "status": "prep_failed",
+ "error": str(exc),
+ })
+
+ # Generate submit_all.sh
+ _write_submit_all(output_base, jobs)
+
+ logger.info("Prepared %d DFT jobs in %s", len(jobs), output_base)
+ return jobs
+
+
+def _prepare_single_job(
+ poscar_path: Path,
+ job_dir: Path,
+ job_name: str,
+ geo_opt_template: str,
+ slurm_template: str,
+ dft_cfg: dict,
+ slurm_cfg: dict,
+ record: dict,
+) -> dict:
+ """Prepare CP2K input and SLURM script for a single interface."""
+ from interfaceml.core.io import load_structure
+
+ structure = load_structure(poscar_path)
+
+ # Write structure.xyz for CP2K
+ xyz_path = job_dir / "structure.xyz"
+ _write_cp2k_xyz(structure, xyz_path)
+
+ # Get unique elements
+ elements = sorted(set(str(site.specie) for site in structure))
+
+ # Build KIND blocks
+ kind_blocks = _build_kind_blocks(elements, dft_cfg)
+
+ # Build cell vectors
+ lattice = structure.lattice
+ cell_a = f"{lattice.matrix[0][0]:.10f} {lattice.matrix[0][1]:.10f} {lattice.matrix[0][2]:.10f}"
+ cell_b = f"{lattice.matrix[1][0]:.10f} {lattice.matrix[1][1]:.10f} {lattice.matrix[1][2]:.10f}"
+ cell_c = f"{lattice.matrix[2][0]:.10f} {lattice.matrix[2][1]:.10f} {lattice.matrix[2][2]:.10f}"
+
+ # Build dispersion block
+ dispersion_block = _build_dispersion_block(dft_cfg.get("dispersion", "DFT-D3"))
+
+ # Build constraint block for fixed atoms
+ constraint_block = _build_constraint_block(poscar_path, structure)
+
+ # Fill CP2K template
+ cp2k_input = geo_opt_template.format(
+ project_name=job_name.replace("-", "_"),
+ cutoff=dft_cfg.get("cutoff", 400),
+ rel_cutoff=dft_cfg.get("rel_cutoff", 60),
+ max_scf=dft_cfg.get("max_scf", 300),
+ functional=dft_cfg.get("functional", "PBE"),
+ dispersion_block=dispersion_block,
+ cell_a=cell_a,
+ cell_b=cell_b,
+ cell_c=cell_c,
+ kind_blocks=kind_blocks,
+ geo_opt_max_iter=dft_cfg.get("geo_opt_max_iter", 200),
+ geo_opt_convergence=dft_cfg.get("geo_opt_convergence", 3.0e-3),
+ constraint_block=constraint_block,
+ )
+
+ cp2k_path = job_dir / "cp2k.inp"
+ cp2k_path.write_text(cp2k_input)
+
+ # Fill SLURM template
+ account_line = ""
+ if slurm_cfg.get("account"):
+ account_line = f"#SBATCH --account={slurm_cfg['account']}"
+
+ slurm_input = slurm_template.format(
+ job_name=job_name,
+ partition=slurm_cfg.get("partition", "gpu"),
+ nodes=slurm_cfg.get("nodes", 1),
+ ntasks_per_node=slurm_cfg.get("ntasks_per_node", 48),
+ time=slurm_cfg.get("time", "24:00:00"),
+ account_line=account_line,
+ cp2k_module=slurm_cfg.get("cp2k_module", "cp2k/2024.1"),
+ cp2k_binary=slurm_cfg.get("cp2k_binary", "cp2k.psmp"),
+ )
+
+ slurm_path = job_dir / "submit.sh"
+ slurm_path.write_text(slurm_input)
+
+ # Save metadata
+ metadata = {
+ "job_id": job_name,
+ "source_poscar": str(poscar_path),
+ "n_atoms": len(structure),
+ "elements": elements,
+ "cell_params": {
+ "a": lattice.a,
+ "b": lattice.b,
+ "c": lattice.c,
+ "alpha": lattice.alpha,
+ "beta": lattice.beta,
+ "gamma": lattice.gamma,
+ },
+ "perovskite_path": record.get("perovskite_path"),
+ "fullerene_path": record.get("fullerene_path"),
+ "interface_record": record,
+ }
+ metadata_path = job_dir / "metadata.json"
+ metadata_path.write_text(json.dumps(metadata, indent=2, default=str))
+
+ return {
+ "job_id": job_name,
+ "job_dir": str(job_dir),
+ "cp2k_input": str(cp2k_path),
+ "slurm_script": str(slurm_path),
+ "xyz_file": str(xyz_path),
+ "n_atoms": len(structure),
+ "elements": elements,
+ "status": "prepared",
+ }
+
+
+def _write_cp2k_xyz(structure, filepath: Path) -> None:
+ """Write structure in XYZ format for CP2K."""
+ n_atoms = len(structure)
+ lines = [str(n_atoms), ""]
+ for site in structure:
+ elem = str(site.specie)
+ x, y, z = site.coords
+ lines.append(f"{elem:>4s} {x:16.10f} {y:16.10f} {z:16.10f}")
+ filepath.write_text("\n".join(lines) + "\n")
+
+
+def _build_kind_blocks(elements: List[str], dft_cfg: dict) -> str:
+ """Build CP2K &KIND blocks for each element."""
+ basis_set = dft_cfg.get("basis_set", "DZVP-MOLOPT-SR-GTH")
+ potential_prefix = dft_cfg.get("potential", "GTH-PBE")
+
+ blocks = []
+ for elem in elements:
+ if elem in _DEFAULT_BASIS_POTENTIAL:
+ bs, pot = _DEFAULT_BASIS_POTENTIAL[elem]
+ else:
+ bs = basis_set
+ pot = f"{potential_prefix}-q4" # Fallback
+ logger.warning(
+ "Element %s not in default mapping, using %s / %s",
+ elem, bs, pot,
+ )
+
+ blocks.append(
+ f" &KIND {elem}\n"
+ f" BASIS_SET {bs}\n"
+ f" POTENTIAL {pot}\n"
+ f" &END KIND"
+ )
+ return "\n".join(blocks)
+
+
+def _build_dispersion_block(dispersion: Optional[str]) -> str:
+ """Build CP2K dispersion correction block."""
+ if not dispersion:
+ return ""
+
+ dispersion = dispersion.upper()
+ if dispersion == "DFT-D3":
+ return (
+ " &VDW_POTENTIAL\n"
+ " POTENTIAL_TYPE PAIR_POTENTIAL\n"
+ " &PAIR_POTENTIAL\n"
+ " TYPE DFTD3\n"
+ " PARAMETER_FILE_NAME dftd3.dat\n"
+ " REFERENCE_FUNCTIONAL PBE\n"
+ " &END PAIR_POTENTIAL\n"
+ " &END VDW_POTENTIAL"
+ )
+ elif dispersion == "DFT-D3(BJ)":
+ return (
+ " &VDW_POTENTIAL\n"
+ " POTENTIAL_TYPE PAIR_POTENTIAL\n"
+ " &PAIR_POTENTIAL\n"
+ " TYPE DFTD3(BJ)\n"
+ " PARAMETER_FILE_NAME dftd3.dat\n"
+ " REFERENCE_FUNCTIONAL PBE\n"
+ " &END PAIR_POTENTIAL\n"
+ " &END VDW_POTENTIAL"
+ )
+ else:
+ logger.warning("Unknown dispersion type: %s", dispersion)
+ return ""
+
+
+def _build_constraint_block(
+ poscar_path: Path,
+ structure,
+) -> str:
+ """Build CP2K CONSTRAINT block for fixed atoms from selective dynamics."""
+ fixed_indices = _get_fixed_atoms_from_poscar(poscar_path)
+ if not fixed_indices:
+ return ""
+
+ # CP2K uses 1-based atom indices in LIST
+ fixed_1based = [i + 1 for i in fixed_indices]
+
+ # Build fixed atoms list string (CP2K format)
+ list_str = " ".join(str(i) for i in fixed_1based)
+
+ return (
+ " &CONSTRAINT\n"
+ " &FIXED_ATOMS\n"
+ f" LIST {list_str}\n"
+ " &END FIXED_ATOMS\n"
+ " &END CONSTRAINT"
+ )
+
+
+def _get_fixed_atoms_from_poscar(poscar_path: Path) -> List[int]:
+ """Extract fixed atom indices (0-based) from POSCAR selective dynamics."""
+ from pymatgen.io.vasp import Poscar
+
+ try:
+ flags = Poscar.from_file(str(poscar_path), check_for_potcar=False).selective_dynamics
+ except (OSError, ValueError, IndexError):
+ return []
+ if flags is None:
+ return []
+ return [index for index, movable in enumerate(flags) if not np.any(movable)]
+
+
+def _write_submit_all(output_base: Path, jobs: List[dict]) -> None:
+ """Write a master script to submit all SLURM jobs."""
+ prepared = [j for j in jobs if j.get("status") == "prepared"]
+ if not prepared:
+ return
+
+ lines = [
+ "#!/bin/bash",
+ f"# Submit all {len(prepared)} DFT jobs",
+ f"# Generated for active learning round",
+ "",
+ "SCRIPT_DIR=$(cd \"$(dirname \"$0\")\" && pwd)",
+ "",
+ ]
+
+ for job in prepared:
+ job_dir = Path(job["job_dir"]).name
+ lines.append(f'echo "Submitting {job["job_id"]}..."')
+ lines.append(f'cd "$SCRIPT_DIR/{job_dir}" && sbatch submit.sh && cd "$SCRIPT_DIR"')
+ lines.append("")
+
+ lines.append(f'echo "Submitted {len(prepared)} jobs"')
+
+ submit_all_path = output_base / "submit_all.sh"
+ submit_all_path.write_text("\n".join(lines) + "\n")
+ submit_all_path.chmod(0o755)
+
+ logger.info("Wrote %s", submit_all_path)
diff --git a/active_learning/generate_components.py b/active_learning/generate_components.py
new file mode 100644
index 00000000..435bcf9e
--- /dev/null
+++ b/active_learning/generate_components.py
@@ -0,0 +1,117 @@
+"""Step 1: Generate fullerene + perovskite structures using AI models."""
+
+import logging
+import sys
+from pathlib import Path
+
+import numpy as np
+
+logger = logging.getLogger(__name__)
+
+
+def generate_round(config: dict, round_dir: Path) -> dict:
+ """Generate fullerene + perovskite structures for one AL round.
+
+ Args:
+ config: Full pipeline configuration dict.
+ round_dir: Directory for this round (e.g. active_learning_runs/round_000/).
+
+ Returns:
+ dict with keys: fullerene_dir, perovskite_dir, fullerene_count, perovskite_count.
+ """
+ seed = config["pipeline"].get("seed", 42)
+ gen_cfg = config["generation"]
+
+ fullerene_dir = round_dir / "generated" / "fullerenes"
+ perovskite_dir = round_dir / "generated" / "perovskites"
+ fullerene_dir.mkdir(parents=True, exist_ok=True)
+ perovskite_dir.mkdir(parents=True, exist_ok=True)
+
+ fullerene_count = _generate_fullerenes(gen_cfg["fullerene"], fullerene_dir, seed)
+ perovskite_count = _generate_perovskites(gen_cfg["perovskite"], perovskite_dir, seed)
+
+ logger.info(
+ "Generated %d fullerenes and %d perovskites",
+ fullerene_count,
+ perovskite_count,
+ )
+
+ return {
+ "fullerene_dir": fullerene_dir,
+ "perovskite_dir": perovskite_dir,
+ "fullerene_count": fullerene_count,
+ "perovskite_count": perovskite_count,
+ }
+
+
+def _generate_fullerenes(cfg: dict, output_dir: Path, seed: int) -> int:
+ """Generate fullerene XYZ files using FullereneAPI."""
+ # Import here to avoid hard dependency at module level
+ sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "fullerene_e3gen"))
+ from api import FullereneAPI
+
+ checkpoint = cfg["checkpoint"]
+ c_values = cfg["C_values"]
+ num_samples = cfg["num_samples_per_C"]
+
+ logger.info("Loading fullerene model from %s", checkpoint)
+ api = FullereneAPI(checkpoint_path=checkpoint)
+
+ count = 0
+ for c_val in c_values:
+ logger.info("Generating %d C%d fullerenes", num_samples, c_val)
+ np.random.seed(seed + c_val)
+
+ samples = api.generate(num_carbon=c_val, num_samples=num_samples)
+
+ for i, sample in enumerate(samples):
+ filename = f"C{c_val}_sample_{i:03d}.xyz"
+ filepath = output_dir / filename
+ api.save_structure(
+ positions=sample["positions"],
+ edges=sample.get("edges", []),
+ output_path=str(filepath),
+ format="xyz",
+ )
+ count += 1
+ logger.debug("Saved %s", filepath)
+
+ return count
+
+
+def _generate_perovskites(cfg: dict, output_dir: Path, seed: int) -> int:
+ """Generate perovskite CIF files using PerovskiteAPI."""
+ sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "perovskite_e3gen"))
+ from api import PerovskiteAPI
+
+ checkpoint = cfg["checkpoint"]
+ compositions = cfg["compositions"]
+ num_atoms = cfg.get("num_atoms", 5)
+ num_samples = cfg["num_samples_per_comp"]
+
+ logger.info("Loading perovskite model from %s", checkpoint)
+ api = PerovskiteAPI(checkpoint_path=checkpoint)
+
+ count = 0
+ for comp in compositions:
+ a, b, x = comp["A"], comp["B"], comp["X"]
+ formula = f"{a}{b}{x}3"
+ logger.info("Generating %d %s perovskites", num_samples, formula)
+
+ samples = api.generate(
+ A=a,
+ B=b,
+ X=x,
+ num_atoms=num_atoms,
+ num_samples=num_samples,
+ seed=seed,
+ )
+
+ for i, sample in enumerate(samples):
+ filename = f"{formula}_sample_{i:03d}.cif"
+ filepath = output_dir / filename
+ api.save_structure(sample, str(filepath), fmt="cif")
+ count += 1
+ logger.debug("Saved %s", filepath)
+
+ return count
diff --git a/active_learning/parse_results.py b/active_learning/parse_results.py
new file mode 100644
index 00000000..ca263ba9
--- /dev/null
+++ b/active_learning/parse_results.py
@@ -0,0 +1,295 @@
+"""Step 4: Parse CP2K output files to extract energies, forces, and optimized geometries."""
+
+import json
+import logging
+import re
+from pathlib import Path
+from typing import Dict, List, Optional, Tuple
+
+import numpy as np
+
+logger = logging.getLogger(__name__)
+
+HA_TO_EV = 27.211386245988
+BOHR_TO_ANG = 0.529177249
+
+
+def parse_dft_results(
+ dft_jobs_dir: Path,
+ output_path: Path,
+) -> List[dict]:
+ """Parse CP2K outputs and collect results.
+
+ Args:
+ dft_jobs_dir: Directory containing job subdirectories.
+ output_path: Path to write results.json.
+
+ Returns:
+ List of result dicts.
+ """
+ output_path.parent.mkdir(parents=True, exist_ok=True)
+
+ job_dirs = sorted(
+ d for d in dft_jobs_dir.iterdir()
+ if d.is_dir() and (d / "cp2k.inp").exists()
+ )
+
+ if not job_dirs:
+ logger.warning("No job directories found in %s", dft_jobs_dir)
+ return []
+
+ results = []
+ for job_dir in job_dirs:
+ job_id = job_dir.name
+ result = _parse_single_job(job_dir, job_id)
+ results.append(result)
+
+ # Write results
+ output_path.write_text(json.dumps(results, indent=2, default=str))
+ logger.info("Parsed %d jobs, wrote results to %s", len(results), output_path)
+
+ # Write human-readable summary
+ summary_path = output_path.parent / "summary.txt"
+ _write_summary(results, summary_path)
+
+ return results
+
+
+def _parse_single_job(job_dir: Path, job_id: str) -> dict:
+ """Parse a single CP2K job directory."""
+ result = {
+ "job_id": job_id,
+ "job_dir": str(job_dir),
+ "converged": False,
+ "total_energy_Ha": None,
+ "total_energy_eV": None,
+ "max_force_Ha_bohr": None,
+ "n_scf_steps": None,
+ "n_geo_steps": None,
+ "n_atoms": None,
+ "optimized_xyz": None,
+ "status": "not_started",
+ }
+
+ # Load metadata if available
+ metadata_path = job_dir / "metadata.json"
+ if metadata_path.exists():
+ try:
+ metadata = json.loads(metadata_path.read_text())
+ result["perovskite_formula"] = _extract_formula(
+ metadata.get("perovskite_path", "")
+ )
+ result["fullerene_C"] = _extract_carbon_count(
+ metadata.get("fullerene_path", "")
+ )
+ result["n_atoms"] = metadata.get("n_atoms")
+ result["metadata"] = metadata
+ except Exception:
+ pass
+
+ # Check if cp2k.out exists
+ cp2k_out = job_dir / "cp2k.out"
+ if not cp2k_out.exists():
+ result["status"] = "not_started"
+ return result
+
+ result["status"] = "started"
+
+ try:
+ out_text = cp2k_out.read_text(errors="replace")
+ except Exception as exc:
+ result["status"] = "read_error"
+ result["error"] = str(exc)
+ return result
+
+ # Parse total energy
+ energy = _parse_total_energy(out_text)
+ if energy is not None:
+ result["total_energy_Ha"] = energy
+ result["total_energy_eV"] = energy * HA_TO_EV
+
+ # Parse convergence
+ result["converged"] = _check_convergence(out_text)
+
+ # Parse SCF steps
+ result["n_scf_steps"] = _count_scf_steps(out_text)
+
+ # Parse geometry optimization steps
+ result["n_geo_steps"] = _count_geo_steps(out_text)
+
+ # Parse max force
+ max_force = _parse_max_force(out_text)
+ if max_force is not None:
+ result["max_force_Ha_bohr"] = max_force
+
+ # Look for optimized trajectory
+ traj_files = list(job_dir.glob("*-pos-1.xyz"))
+ if not traj_files:
+ traj_files = list(job_dir.glob("*-pos-*.xyz"))
+
+ if traj_files:
+ traj_file = traj_files[0]
+ opt_xyz = job_dir / "optimized.xyz"
+ _extract_last_frame(traj_file, opt_xyz)
+ if opt_xyz.exists():
+ result["optimized_xyz"] = str(opt_xyz)
+
+ # Set final status
+ if result["converged"]:
+ result["status"] = "converged"
+ elif energy is not None:
+ result["status"] = "completed_not_converged"
+ else:
+ result["status"] = "failed"
+
+ return result
+
+
+def _parse_total_energy(text: str) -> Optional[float]:
+ """Extract the final total energy from CP2K output."""
+ # CP2K prints: ENERGY| Total FORCE_EVAL ( QS ) energy [a.u.]: -1234.567890
+ matches = re.findall(
+ r"ENERGY\|\s+Total FORCE_EVAL.*?energy.*?:\s+([-\d.Ee+]+)",
+ text,
+ )
+ if matches:
+ return float(matches[-1]) # Last occurrence = final energy
+ return None
+
+
+def _check_convergence(text: str) -> bool:
+ """Check if geometry optimization converged."""
+ return "GEOMETRY OPTIMIZATION COMPLETED" in text
+
+
+def _count_scf_steps(text: str) -> Optional[int]:
+ """Count total SCF iterations."""
+ matches = re.findall(r"SCF\s+run\s+converged\s+in\s+(\d+)\s+steps", text)
+ if matches:
+ return sum(int(m) for m in matches)
+ return None
+
+
+def _count_geo_steps(text: str) -> Optional[int]:
+ """Count geometry optimization steps."""
+ matches = re.findall(r"OPTIMIZATION STEP:\s+(\d+)", text)
+ if matches:
+ return int(matches[-1])
+ return None
+
+
+def _parse_max_force(text: str) -> Optional[float]:
+ """Extract the final maximum force."""
+ # CP2K prints: Max. step size = 0.0012345678
+ # or: MAX GRADIENT = 0.0012345678
+ matches = re.findall(
+ r"Max(?:imum)?\s+(?:gradient|force)\s*[:=]\s*([-\d.Ee+]+)",
+ text,
+ re.IGNORECASE,
+ )
+ if matches:
+ return float(matches[-1])
+
+ # Alternative pattern
+ matches = re.findall(
+ r"Convergence.*?RMS\s+gradient\s*[:=]\s*([-\d.Ee+]+)",
+ text,
+ re.IGNORECASE,
+ )
+ if matches:
+ return float(matches[-1])
+
+ return None
+
+
+def _extract_last_frame(trajectory_path: Path, output_path: Path) -> None:
+ """Extract the last frame from a CP2K trajectory XYZ file."""
+ try:
+ text = trajectory_path.read_text()
+ except Exception:
+ return
+
+ lines = text.strip().splitlines()
+ if not lines:
+ return
+
+ # Find the last frame by scanning backwards for atom count line
+ frames = []
+ i = 0
+ while i < len(lines):
+ try:
+ n_atoms = int(lines[i].strip())
+ except (ValueError, IndexError):
+ i += 1
+ continue
+ frame_end = i + n_atoms + 2 # count + comment + atoms
+ if frame_end <= len(lines):
+ frames.append((i, frame_end))
+ i = frame_end
+
+ if frames:
+ start, end = frames[-1]
+ output_path.write_text("\n".join(lines[start:end]) + "\n")
+
+
+def _extract_formula(path: str) -> Optional[str]:
+ """Extract perovskite formula from filename."""
+ if not path:
+ return None
+ name = Path(path).stem
+ # Try patterns like BaTiO3_sample_001
+ match = re.match(r"([A-Z][a-z]?[A-Z][a-z]?[A-Z][a-z]?\d*)", name)
+ if match:
+ return match.group(1)
+ return name
+
+
+def _extract_carbon_count(path: str) -> Optional[int]:
+ """Extract carbon count from fullerene filename."""
+ if not path:
+ return None
+ match = re.search(r"C(\d+)", Path(path).stem)
+ if match:
+ return int(match.group(1))
+ return None
+
+
+def _write_summary(results: List[dict], summary_path: Path) -> None:
+ """Write a human-readable summary of DFT results."""
+ total = len(results)
+ converged = sum(1 for r in results if r["converged"])
+ failed = sum(1 for r in results if r["status"] == "failed")
+ not_started = sum(1 for r in results if r["status"] == "not_started")
+ energies = [r["total_energy_eV"] for r in results if r["total_energy_eV"] is not None]
+
+ lines = [
+ "=" * 60,
+ "DFT Results Summary",
+ "=" * 60,
+ f"Total jobs: {total}",
+ f"Converged: {converged}",
+ f"Failed: {failed}",
+ f"Not started: {not_started}",
+ "",
+ ]
+
+ if energies:
+ lines.extend([
+ f"Energy range: {min(energies):.4f} to {max(energies):.4f} eV",
+ f"Mean energy: {np.mean(energies):.4f} eV",
+ "",
+ ])
+
+ lines.append("-" * 60)
+ lines.append(f"{'Job ID':<30s} {'Status':<15s} {'Energy (eV)':>15s} {'Force':>10s}")
+ lines.append("-" * 60)
+
+ for r in results:
+ energy_str = f"{r['total_energy_eV']:.4f}" if r["total_energy_eV"] else "N/A"
+ force_str = f"{r['max_force_Ha_bohr']:.6f}" if r["max_force_Ha_bohr"] else "N/A"
+ lines.append(
+ f"{r['job_id']:<30s} {r['status']:<15s} {energy_str:>15s} {force_str:>10s}"
+ )
+
+ summary_path.write_text("\n".join(lines) + "\n")
+ logger.info("Wrote summary to %s", summary_path)
diff --git a/active_learning/pipeline.py b/active_learning/pipeline.py
new file mode 100644
index 00000000..cd084407
--- /dev/null
+++ b/active_learning/pipeline.py
@@ -0,0 +1,300 @@
+"""Main orchestrator for the active learning pipeline.
+
+Usage:
+ # Run phases 1-3 (generate → build → dft_prep), then stop for manual DFT
+ python -m active_learning.pipeline --config active_learning/config.yaml --round 0
+
+ # After DFT completes, parse results
+ python -m active_learning.pipeline --config active_learning/config.yaml --round 0 --phase parse
+
+ # Prepare retraining data
+ python -m active_learning.pipeline --config active_learning/config.yaml --round 0 --phase retrain
+
+ # Run a specific phase
+ python -m active_learning.pipeline --phase generate --round 1
+ python -m active_learning.pipeline --phase build --round 1
+ python -m active_learning.pipeline --phase dft_prep --round 1
+"""
+
+import argparse
+import json
+import logging
+import sys
+from pathlib import Path
+
+import yaml
+
+from .build_interfaces import build_interfaces
+from .dft_inputs import prepare_dft_calculations
+from .generate_components import generate_round
+from .parse_results import parse_dft_results
+from .prepare_retraining import prepare_retraining_data
+
+logger = logging.getLogger(__name__)
+
+PHASES = ["generate", "build", "dft_prep", "parse", "retrain"]
+# Phases that run before DFT (automated)
+PRE_DFT_PHASES = ["generate", "build", "dft_prep"]
+
+
+def load_config(config_path: str) -> dict:
+ """Load pipeline configuration from YAML."""
+ path = Path(config_path)
+ if not path.exists():
+ raise FileNotFoundError(f"Config not found: {config_path}")
+ with path.open() as f:
+ return yaml.safe_load(f)
+
+
+def get_round_dir(config: dict, round_num: int) -> Path:
+ """Get the directory for a specific round."""
+ base_dir = Path(config["pipeline"].get("base_dir", "active_learning_runs"))
+ return base_dir / f"round_{round_num:03d}"
+
+
+def run_phase(phase: str, config: dict, round_dir: Path) -> dict:
+ """Run a single phase of the pipeline.
+
+ Args:
+ phase: One of PHASES.
+ config: Full pipeline configuration.
+ round_dir: Directory for this round.
+
+ Returns:
+ Phase result dict.
+ """
+ round_dir.mkdir(parents=True, exist_ok=True)
+ state_path = round_dir / "pipeline_state.json"
+ state = _load_state(state_path)
+
+ logger.info("=" * 60)
+ logger.info("Running phase: %s", phase)
+ logger.info("Round directory: %s", round_dir)
+ logger.info("=" * 60)
+
+ if phase == "generate":
+ result = generate_round(config, round_dir)
+ state["generate"] = {
+ "status": "completed",
+ "fullerene_dir": str(result["fullerene_dir"]),
+ "perovskite_dir": str(result["perovskite_dir"]),
+ "fullerene_count": result["fullerene_count"],
+ "perovskite_count": result["perovskite_count"],
+ }
+
+ elif phase == "build":
+ gen_state = state.get("generate", {})
+ fullerene_dir = Path(
+ gen_state.get("fullerene_dir", round_dir / "generated" / "fullerenes")
+ )
+ perovskite_dir = Path(
+ gen_state.get("perovskite_dir", round_dir / "generated" / "perovskites")
+ )
+ interfaces_dir = round_dir / "interfaces"
+
+ records = build_interfaces(
+ fullerene_dir=fullerene_dir,
+ perovskite_dir=perovskite_dir,
+ output_dir=interfaces_dir,
+ config=config,
+ )
+ successful = [r for r in records if r.get("status") == "success"]
+ state["build"] = {
+ "status": "completed",
+ "interfaces_dir": str(interfaces_dir),
+ "total_records": len(records),
+ "successful": len(successful),
+ }
+ result = {"records": records, "interfaces_dir": interfaces_dir}
+
+ elif phase == "dft_prep":
+ build_state = state.get("build", {})
+ interfaces_dir = Path(
+ build_state.get("interfaces_dir", round_dir / "interfaces")
+ )
+
+ # Load metadata
+ metadata_path = interfaces_dir / "metadata.jsonl"
+ records = []
+ if metadata_path.exists():
+ with metadata_path.open() as f:
+ for line in f:
+ if line.strip():
+ records.append(json.loads(line))
+
+ dft_jobs_dir = round_dir / "dft_jobs"
+ jobs = prepare_dft_calculations(records, dft_jobs_dir, config)
+ prepared = [j for j in jobs if j.get("status") == "prepared"]
+
+ state["dft_prep"] = {
+ "status": "completed",
+ "dft_jobs_dir": str(dft_jobs_dir),
+ "total_jobs": len(jobs),
+ "prepared": len(prepared),
+ }
+ result = {"jobs": jobs, "dft_jobs_dir": dft_jobs_dir}
+
+ logger.info("")
+ logger.info("=" * 60)
+ logger.info("DFT preparation complete!")
+ logger.info(" Jobs directory: %s", dft_jobs_dir)
+ logger.info(" Prepared jobs: %d", len(prepared))
+ logger.info("")
+ logger.info("Next steps:")
+ logger.info(" 1. Review CP2K inputs in %s", dft_jobs_dir)
+ logger.info(" 2. Submit jobs: bash %s/submit_all.sh", dft_jobs_dir)
+ logger.info(" 3. After DFT completes, run:")
+ logger.info(
+ " python -m active_learning.pipeline --round %d --phase parse",
+ config["pipeline"].get("round", 0),
+ )
+ logger.info("=" * 60)
+
+ elif phase == "parse":
+ dft_state = state.get("dft_prep", {})
+ dft_jobs_dir = Path(
+ dft_state.get("dft_jobs_dir", round_dir / "dft_jobs")
+ )
+ results_dir = round_dir / "results"
+ results_path = results_dir / "results.json"
+
+ results = parse_dft_results(dft_jobs_dir, results_path)
+ converged = sum(1 for r in results if r.get("converged"))
+
+ state["parse"] = {
+ "status": "completed",
+ "results_path": str(results_path),
+ "total_results": len(results),
+ "converged": converged,
+ }
+ result = {"results": results, "results_path": results_path}
+
+ elif phase == "retrain":
+ parse_state = state.get("parse", {})
+ results_path = Path(
+ parse_state.get("results_path", round_dir / "results" / "results.json")
+ )
+ retrain_dir = round_dir / "retraining"
+
+ # Existing data directory (project-level)
+ existing_data = Path(
+ config.get("retraining", {}).get(
+ "existing_data_dir", "dataset"
+ )
+ )
+
+ summary = prepare_retraining_data(
+ results_path=results_path,
+ existing_data_dir=existing_data,
+ output_dir=retrain_dir,
+ config=config,
+ )
+
+ state["retrain"] = {
+ "status": "completed",
+ "retrain_dir": str(retrain_dir),
+ "num_accepted": summary["num_accepted"],
+ "num_rejected": summary["num_rejected"],
+ }
+ result = summary
+
+ else:
+ raise ValueError(f"Unknown phase: {phase}. Must be one of {PHASES}")
+
+ _save_state(state, state_path)
+ return result
+
+
+def run_round(config_path: str, round_num: int = None, phase: str = None):
+ """Execute one active learning round (or a specific phase).
+
+ Args:
+ config_path: Path to config.yaml.
+ round_num: Override round number (default: from config).
+ phase: Run only this phase (default: run pre-DFT phases).
+ """
+ config = load_config(config_path)
+
+ if round_num is None:
+ round_num = config["pipeline"].get("round", 0)
+ config["pipeline"]["round"] = round_num
+
+ round_dir = get_round_dir(config, round_num)
+
+ if phase:
+ return run_phase(phase, config, round_dir)
+
+ # Run all pre-DFT phases sequentially
+ results = {}
+ for p in PRE_DFT_PHASES:
+ results[p] = run_phase(p, config, round_dir)
+
+ return results
+
+
+def _load_state(state_path: Path) -> dict:
+ """Load pipeline state from JSON."""
+ if state_path.exists():
+ return json.loads(state_path.read_text())
+ return {}
+
+
+def _save_state(state: dict, state_path: Path) -> None:
+ """Save pipeline state to JSON."""
+ state_path.write_text(json.dumps(state, indent=2, default=str))
+
+
+def main():
+ parser = argparse.ArgumentParser(
+ description="Active learning pipeline: AI → Interface → DFT → Retrain",
+ formatter_class=argparse.RawDescriptionHelpFormatter,
+ epilog="""
+Examples:
+ # Run full round (stops after dft_prep for manual DFT submission)
+ python -m active_learning.pipeline --round 0
+
+ # After DFT completes, parse results
+ python -m active_learning.pipeline --round 0 --phase parse
+
+ # Prepare retraining data
+ python -m active_learning.pipeline --round 0 --phase retrain
+
+ # Run specific phases
+ python -m active_learning.pipeline --phase generate --round 1
+ python -m active_learning.pipeline --phase build --round 1
+ """,
+ )
+ parser.add_argument(
+ "--config",
+ default="active_learning/config.yaml",
+ help="Path to pipeline config (default: active_learning/config.yaml)",
+ )
+ parser.add_argument(
+ "--round",
+ type=int,
+ default=None,
+ help="Round number (default: from config)",
+ )
+ parser.add_argument(
+ "--phase",
+ choices=PHASES,
+ default=None,
+ help="Run only this phase (default: run generate→build→dft_prep)",
+ )
+
+ args = parser.parse_args()
+
+ run_round(
+ config_path=args.config,
+ round_num=args.round,
+ phase=args.phase,
+ )
+
+
+if __name__ == "__main__":
+ logging.basicConfig(
+ level=logging.INFO,
+ format="%(asctime)s [%(levelname)s] %(name)s: %(message)s",
+ datefmt="%Y-%m-%d %H:%M:%S",
+ )
+ main()
diff --git a/active_learning/prepare_retraining.py b/active_learning/prepare_retraining.py
new file mode 100644
index 00000000..efae6bf4
--- /dev/null
+++ b/active_learning/prepare_retraining.py
@@ -0,0 +1,267 @@
+"""Step 5: Filter DFT results and prepare retraining datasets."""
+
+import json
+import logging
+import shutil
+from pathlib import Path
+from typing import Dict, List, Optional
+
+import numpy as np
+import yaml
+
+logger = logging.getLogger(__name__)
+
+
+def prepare_retraining_data(
+ results_path: Path,
+ existing_data_dir: Path,
+ output_dir: Path,
+ config: dict,
+) -> dict:
+ """Filter + convert DFT results into retraining dataset.
+
+ Args:
+ results_path: Path to results.json from parse step.
+ existing_data_dir: Directory with existing training data.
+ output_dir: Directory for retraining outputs.
+ config: Full pipeline configuration dict.
+
+ Returns:
+ dict with keys: perovskite_cifs, interface_structures, num_accepted,
+ num_rejected, stats.
+ """
+ output_dir.mkdir(parents=True, exist_ok=True)
+ retrain_cfg = config.get("retraining", {})
+
+ energy_filter = retrain_cfg.get("energy_filter", -1.0)
+ force_threshold = retrain_cfg.get("force_threshold", 0.5)
+ min_structures = retrain_cfg.get("min_structures", 50)
+
+ # Load results
+ results = json.loads(results_path.read_text())
+
+ # Filter
+ accepted = []
+ rejected = []
+ for r in results:
+ reason = _check_acceptance(r, energy_filter, force_threshold)
+ if reason is None:
+ accepted.append(r)
+ else:
+ r["reject_reason"] = reason
+ rejected.append(r)
+
+ logger.info(
+ "Filtering: %d accepted, %d rejected out of %d total",
+ len(accepted),
+ len(rejected),
+ len(results),
+ )
+
+ # Warn if below minimum
+ if len(accepted) < min_structures:
+ logger.warning(
+ "Only %d accepted structures (minimum: %d). "
+ "Consider relaxing filters or running more DFT.",
+ len(accepted),
+ min_structures,
+ )
+
+ # Extract and save structures
+ perovskite_cif_dir = output_dir / "perovskite_cifs"
+ interface_data_dir = output_dir / "interface_data"
+ perovskite_cif_dir.mkdir(parents=True, exist_ok=True)
+ interface_data_dir.mkdir(parents=True, exist_ok=True)
+
+ perovskite_count = 0
+ interface_count = 0
+
+ for r in accepted:
+ opt_xyz = r.get("optimized_xyz")
+ if not opt_xyz or not Path(opt_xyz).exists():
+ continue
+
+ job_id = r["job_id"]
+
+ # Copy optimized interface structure
+ dst_xyz = interface_data_dir / f"{job_id}_optimized.xyz"
+ shutil.copy2(opt_xyz, dst_xyz)
+ interface_count += 1
+
+ # Extract perovskite slab from optimized interface
+ if retrain_cfg.get("retrain_perovskite", True):
+ try:
+ cif_path = perovskite_cif_dir / f"{job_id}_slab.cif"
+ _extract_perovskite_slab(opt_xyz, cif_path, r)
+ perovskite_count += 1
+ except Exception as exc:
+ logger.warning(
+ "Failed to extract perovskite slab from %s: %s",
+ job_id, exc,
+ )
+
+ # Compute statistics
+ stats = _compute_stats(accepted, rejected)
+
+ # Write retrain config
+ retrain_config = _build_retrain_config(
+ config,
+ perovskite_cif_dir,
+ interface_data_dir,
+ existing_data_dir,
+ stats,
+ )
+ retrain_config_path = output_dir / "retrain_config.yaml"
+ retrain_config_path.write_text(yaml.dump(retrain_config, default_flow_style=False))
+
+ # Save accepted/rejected lists
+ (output_dir / "accepted.json").write_text(
+ json.dumps(accepted, indent=2, default=str)
+ )
+ (output_dir / "rejected.json").write_text(
+ json.dumps(rejected, indent=2, default=str)
+ )
+
+ summary = {
+ "perovskite_cifs": str(perovskite_cif_dir),
+ "interface_structures": str(interface_data_dir),
+ "num_accepted": len(accepted),
+ "num_rejected": len(rejected),
+ "perovskite_cifs_extracted": perovskite_count,
+ "interface_structures_saved": interface_count,
+ "stats": stats,
+ }
+
+ logger.info(
+ "Prepared retraining data: %d perovskite CIFs, %d interface structures",
+ perovskite_count,
+ interface_count,
+ )
+
+ return summary
+
+
+def _check_acceptance(
+ result: dict,
+ energy_filter: float,
+ force_threshold: float,
+) -> Optional[str]:
+ """Check if a result passes quality filters.
+
+ Returns None if accepted, or a rejection reason string.
+ """
+ if not result.get("converged"):
+ return "not_converged"
+
+ if result.get("total_energy_eV") is None:
+ return "no_energy"
+
+ binding = result.get("binding_energy_eV")
+ if binding is not None and binding > energy_filter:
+ return f"binding_energy_too_high ({binding:.4f} > {energy_filter})"
+
+ max_force = result.get("max_force_Ha_bohr")
+ if max_force is not None and max_force > force_threshold:
+ return f"force_too_high ({max_force:.6f} > {force_threshold})"
+
+ return None
+
+
+def _extract_perovskite_slab(
+ optimized_xyz: str,
+ output_cif: Path,
+ result: dict,
+) -> None:
+ """Extract the perovskite slab portion from an optimized interface.
+
+ Uses layer splitting to separate the fullerene (molecular) layer
+ from the perovskite slab layers.
+ """
+ from interfaceml.core.io import load_structure
+
+ structure = load_structure(optimized_xyz)
+
+ # Identify carbon atoms (fullerene) vs non-carbon (perovskite slab)
+ # This is a heuristic — fullerene is pure carbon, perovskite has mixed elements
+ carbon_indices = []
+ slab_indices = []
+ for i, site in enumerate(structure):
+ if str(site.specie) == "C":
+ carbon_indices.append(i)
+ else:
+ slab_indices.append(i)
+
+ if not slab_indices:
+ raise ValueError("No non-carbon atoms found — cannot extract slab")
+
+ # Extract slab structure
+ slab_structure = structure.copy()
+ # Remove carbon atoms (in reverse order to preserve indices)
+ for idx in sorted(carbon_indices, reverse=True):
+ slab_structure.remove_sites([idx])
+
+ # Write as CIF
+ slab_structure.to(filename=str(output_cif))
+
+
+def _compute_stats(accepted: List[dict], rejected: List[dict]) -> dict:
+ """Compute summary statistics."""
+ stats = {
+ "total": len(accepted) + len(rejected),
+ "accepted": len(accepted),
+ "rejected": len(rejected),
+ "acceptance_rate": len(accepted) / max(len(accepted) + len(rejected), 1),
+ }
+
+ energies = [r["total_energy_eV"] for r in accepted if r.get("total_energy_eV")]
+ if energies:
+ stats["energy_mean_eV"] = float(np.mean(energies))
+ stats["energy_std_eV"] = float(np.std(energies))
+ stats["energy_min_eV"] = float(np.min(energies))
+ stats["energy_max_eV"] = float(np.max(energies))
+
+ forces = [r["max_force_Ha_bohr"] for r in accepted if r.get("max_force_Ha_bohr")]
+ if forces:
+ stats["max_force_mean"] = float(np.mean(forces))
+ stats["max_force_max"] = float(np.max(forces))
+
+ # Rejection reasons
+ reasons = {}
+ for r in rejected:
+ reason = r.get("reject_reason", "unknown")
+ # Group parametric reasons
+ base_reason = reason.split("(")[0].strip()
+ reasons[base_reason] = reasons.get(base_reason, 0) + 1
+ stats["rejection_reasons"] = reasons
+
+ return stats
+
+
+def _build_retrain_config(
+ config: dict,
+ perovskite_cif_dir: Path,
+ interface_data_dir: Path,
+ existing_data_dir: Path,
+ stats: dict,
+) -> dict:
+ """Build a retraining configuration file."""
+ retrain_cfg = config.get("retraining", {})
+
+ retrain_config = {
+ "data": {
+ "existing_data_dir": str(existing_data_dir),
+ "new_perovskite_cifs": str(perovskite_cif_dir),
+ "new_interface_data": str(interface_data_dir),
+ "num_new_structures": stats["accepted"],
+ },
+ "training": {
+ "retrain_perovskite": retrain_cfg.get("retrain_perovskite", True),
+ "retrain_fullerene": retrain_cfg.get("retrain_fullerene", False),
+ "fine_tune": True,
+ "learning_rate": 1e-4,
+ "epochs": 50,
+ },
+ "round": config["pipeline"].get("round", 0),
+ }
+
+ return retrain_config
diff --git a/active_learning/templates/cp2k_energy.inp b/active_learning/templates/cp2k_energy.inp
new file mode 100644
index 00000000..5f92b930
--- /dev/null
+++ b/active_learning/templates/cp2k_energy.inp
@@ -0,0 +1,67 @@
+&GLOBAL
+ PROJECT {project_name}
+ RUN_TYPE ENERGY_FORCE
+ PRINT_LEVEL MEDIUM
+&END GLOBAL
+
+&FORCE_EVAL
+ METHOD Quickstep
+
+ &DFT
+ BASIS_SET_FILE_NAME BASIS_MOLOPT
+ POTENTIAL_FILE_NAME POTENTIAL
+
+ &MGRID
+ CUTOFF {cutoff}
+ REL_CUTOFF {rel_cutoff}
+ NGRIDS 5
+ &END MGRID
+
+ &QS
+ METHOD GPW
+ EPS_DEFAULT 1.0E-12
+ &END QS
+
+ &SCF
+ MAX_SCF {max_scf}
+ EPS_SCF 1.0E-6
+ SCF_GUESS ATOMIC
+ &OT
+ MINIMIZER DIIS
+ PRECONDITIONER FULL_SINGLE_INVERSE
+ ENERGY_GAP 0.1
+ &END OT
+ &OUTER_SCF
+ MAX_SCF 20
+ EPS_SCF 1.0E-6
+ &END OUTER_SCF
+ &END SCF
+
+ &XC
+ &XC_FUNCTIONAL {functional}
+ &END XC_FUNCTIONAL
+{dispersion_block}
+ &END XC
+ &END DFT
+
+ &SUBSYS
+ &CELL
+ A {cell_a}
+ B {cell_b}
+ C {cell_c}
+ PERIODIC XYZ
+ &END CELL
+
+ &TOPOLOGY
+ COORD_FILE_NAME structure.xyz
+ COORD_FILE_FORMAT XYZ
+ &END TOPOLOGY
+
+{kind_blocks}
+ &END SUBSYS
+
+ &PRINT
+ &FORCES ON
+ &END FORCES
+ &END PRINT
+&END FORCE_EVAL
diff --git a/active_learning/templates/cp2k_geo_opt.inp b/active_learning/templates/cp2k_geo_opt.inp
new file mode 100644
index 00000000..e14348ff
--- /dev/null
+++ b/active_learning/templates/cp2k_geo_opt.inp
@@ -0,0 +1,108 @@
+&GLOBAL
+ PROJECT {project_name}
+ RUN_TYPE GEO_OPT
+ PRINT_LEVEL MEDIUM
+&END GLOBAL
+
+&FORCE_EVAL
+ METHOD Quickstep
+
+ &DFT
+ BASIS_SET_FILE_NAME BASIS_MOLOPT
+ POTENTIAL_FILE_NAME POTENTIAL
+
+ &MGRID
+ CUTOFF {cutoff}
+ REL_CUTOFF {rel_cutoff}
+ NGRIDS 5
+ &END MGRID
+
+ &QS
+ METHOD GPW
+ EPS_DEFAULT 1.0E-12
+ EXTRAPOLATION ASPC
+ EXTRAPOLATION_ORDER 3
+ &END QS
+
+ &SCF
+ MAX_SCF {max_scf}
+ EPS_SCF 1.0E-6
+ SCF_GUESS RESTART
+ &OT
+ MINIMIZER DIIS
+ PRECONDITIONER FULL_SINGLE_INVERSE
+ ENERGY_GAP 0.1
+ &END OT
+ &OUTER_SCF
+ MAX_SCF 20
+ EPS_SCF 1.0E-6
+ &END OUTER_SCF
+ &END SCF
+
+ &XC
+ &XC_FUNCTIONAL {functional}
+ &END XC_FUNCTIONAL
+{dispersion_block}
+ &END XC
+
+ &PRINT
+ &E_DENSITY_CUBE OFF
+ &END E_DENSITY_CUBE
+ &END PRINT
+ &END DFT
+
+ &SUBSYS
+ &CELL
+ A {cell_a}
+ B {cell_b}
+ C {cell_c}
+ PERIODIC XYZ
+ &END CELL
+
+ &TOPOLOGY
+ COORD_FILE_NAME structure.xyz
+ COORD_FILE_FORMAT XYZ
+ &END TOPOLOGY
+
+{kind_blocks}
+ &END SUBSYS
+
+ &PRINT
+ &FORCES ON
+ &END FORCES
+ &END PRINT
+&END FORCE_EVAL
+
+&MOTION
+ &GEO_OPT
+ TYPE MINIMIZATION
+ MAX_ITER {geo_opt_max_iter}
+ OPTIMIZER BFGS
+ &BFGS
+ TRUST_RADIUS 0.2
+ &END BFGS
+ MAX_FORCE {geo_opt_convergence}
+ &END GEO_OPT
+
+ &PRINT
+ &TRAJECTORY
+ FORMAT XYZ
+ &EACH
+ GEO_OPT 1
+ &END EACH
+ &END TRAJECTORY
+ &FORCES
+ FORMAT XYZ
+ &EACH
+ GEO_OPT 1
+ &END EACH
+ &END FORCES
+ &RESTART
+ &EACH
+ GEO_OPT 10
+ &END EACH
+ &END RESTART
+ &END PRINT
+
+{constraint_block}
+&END MOTION
diff --git a/active_learning/templates/slurm_cp2k.sh b/active_learning/templates/slurm_cp2k.sh
new file mode 100644
index 00000000..4dd1b19a
--- /dev/null
+++ b/active_learning/templates/slurm_cp2k.sh
@@ -0,0 +1,19 @@
+#!/bin/bash
+#SBATCH --job-name={job_name}
+#SBATCH --partition={partition}
+#SBATCH --nodes={nodes}
+#SBATCH --ntasks-per-node={ntasks_per_node}
+#SBATCH --time={time}
+#SBATCH --output=slurm-%j.out
+#SBATCH --error=slurm-%j.err
+{account_line}
+
+echo "Job started at $(date)"
+echo "Running on $(hostname)"
+echo "Working directory: $(pwd)"
+
+module load {cp2k_module}
+
+srun {cp2k_binary} -i cp2k.inp -o cp2k.out
+
+echo "Job finished at $(date)"
diff --git a/build_heterojunctions/README.md b/build_heterojunctions/README.md
index 3d7d53be..cb44055e 100644
--- a/build_heterojunctions/README.md
+++ b/build_heterojunctions/README.md
@@ -34,16 +34,32 @@ python build_heterojunctions/fix_interface_layers.py \
-t 2.0
```
+## Algorithms
+
+This folder implements interface construction and analysis pipelines. Key algorithms:
+
+1. In-plane commensurate matching using the ZSL search to find integer supercells
+ within length and angle tolerances.
+2. Coherent interface construction from matched slabs with controlled gap and vacuum.
+3. Layer detection for selective dynamics using height projection and gap clustering.
+4. Stack splitting by gap detection for multi-layer structures (legacy mode).
+5. Difference charge density computed as Delta rho = rho_interface - rho_A - rho_B
+ on identical cube grids using streaming IO.
+
+For math details and the composition-aware splitting algorithm, see:
+- `docs/MATHEMATICAL_OVERVIEW.md`
+- `docs/SMART_SPLITTING_ALGORITHM.md`
+
## 0) Enlarge a CIF unit cell (add vacuum) (`expand_cif_cell.py`)
-This helper script enlarges the **unit cell** to a target size (e.g., \(a=b=c=20\) Å) while keeping the
+This helper script enlarges the **unit cell** to a target size (e.g., \(a=b=c=20\) Angstrom) while keeping the
**molecular geometry** unchanged. This is useful when you want to add vacuum around an isolated molecule
(C60/C70, etc.) before building adsorption or interface models.
### Recommended usage (keep the structure centered)
```bash
-# Enlarge the cell to 20 Å and keep the molecule centered in the new unit cell.
+# Enlarge the cell to 20 Angstrom and keep the molecule centered in the new unit cell.
# --wrap keeps fractional coordinates within [0, 1) for nicer visualization.
python build_heterojunctions/expand_cif_cell.py \
structures/etl/C60-Ih.cif \
@@ -61,7 +77,7 @@ This will write:
### Notes
-- `--target`: Sets the new cell length (Å). The script sets \(a=b=c=\) target and keeps the original angles.
+- `--target`: Sets the new cell length (Angstrom). The script sets \(a=b=c=\) target and keeps the original angles.
- `--center`: Translates the structure so its **geometric center** is at the **new cell center**.
- `--wrap`: Wraps fractional coordinates back to \([0, 1)\) after transforming/translating.
- If your shell complains about parentheses in paths (e.g. zsh), wrap paths in quotes.
@@ -92,12 +108,12 @@ python build_heterojunctions/interface_builder.py \
### Common optional arguments
-- `--slab_thickness_a`: Slab thickness for A in Å (default: 20.0)
-- `--slab_thickness_b`: Slab thickness for B in Å (default: 12.0)
-- `--vacuum`: Vacuum size in Å (default: 20.0)
-- `--sep`: Initial separation between surfaces in Å (default: 3.2)
+- `--slab_thickness_a`: Slab thickness for A in Angstrom (default: 20.0)
+- `--slab_thickness_b`: Slab thickness for B in Angstrom (default: 12.0)
+- `--vacuum`: Vacuum size in Angstrom (default: 20.0)
+- `--sep`: Initial separation between surfaces in Angstrom (default: 3.2)
- `--tol`: Matching strain tolerance (default: 0.03)
-- `--max_area`: Maximum interface area in Ų for searching matches (default: 800.0)
+- `--max_area`: Maximum interface area in Angstrom^2 for searching matches (default: 800.0)
- `--strain_target`: Which material to strain: `A`, `B`, or `both` (default: `A`)
- `--use_builder_interface`: Use CoherentInterfaceBuilder method (recommended for ordered interfaces)
- `--max_atoms`: Maximum number of atoms in final structure (default: 400)
@@ -122,9 +138,9 @@ Plane-wave DFT uses **3D periodic boundary conditions**. That means the C60 mole
in the in-plane directions (x/y). You control whether this represents:
- a **dense periodic C60 overlayer** (small in-plane cell), or
-- an **isolated C60 adsorption** model (larger in-plane supercell to reduce image–image interactions).
+- an **isolated C60 adsorption** model (larger in-plane supercell to reduce image-image interactions).
-Use `--auto_supercell` or `--supercell_xy` to make the in-plane cell larger and avoid unphysical C60–C60
+Use `--auto_supercell` or `--supercell_xy` to make the in-plane cell larger and avoid unphysical C60-C60
overlap across periodic images.
### Basic usage (recommended)
@@ -146,7 +162,7 @@ python build_heterojunctions/interface_builder.py \
### Multi-adsorbate stack (3 layers / 2 interfaces)
To build a **three-layer** model (perovskite bottom / C70 middle / C60 top) with **two interfaces**
-and a fixed **2 Å gap between neighboring layers**, use `--adsorbates` + `--layer_gaps`:
+and a fixed **2 Angstrom gap between neighboring layers**, use `--adsorbates` + `--layer_gaps`:
```bash
python build_heterojunctions/interface_builder.py \
@@ -176,14 +192,14 @@ python build_heterojunctions/interface_builder.py \
- `AI`: generic A-site organic + I termination (accepts MA and/or FA; recommended for mixed-cation perovskites such as MA0.5FA0.5PbI3)
- **Default behavior**: the slab is built as a **symmetric slab**, i.e. the **top and bottom surfaces have the same termination** (recommended for slab DFT).
- `--no_symmetric_slab`: Disable symmetric-slab enforcement (top/bottom terminations may differ). **Not recommended**; only useful for debugging.
-- `--adsorbate_distance`: Target distance (Å) between **top-most slab atom** and the **lowest C atom in C60**
+- `--adsorbate_distance`: Target distance (Angstrom) between **top-most slab atom** and the **lowest C atom in C60**
along the surface normal
- `--adsorbates`: Comma-separated adsorbate CIFs for multilayer stacking (bottom->top)
-- `--layer_gaps`: Comma-separated gaps in Å, must match `--adsorbates` length
+- `--layer_gaps`: Comma-separated gaps in Angstrom, must match `--adsorbates` length
- `--auto_supercell`: Automatically choose a small in-plane supercell to reduce C60 periodic-image interactions
(heuristic target: `c60_diameter + c60_buffer`)
- `--supercell_xy nx,ny`: Explicitly set the in-plane supercell (e.g., `2,2` or `3,3`)
-- `--c60_diameter`, `--c60_buffer`: Heuristic controls for `--auto_supercell` (defaults: 7.1 Å and 3.0 Å). For other fullerenes (e.g., C70), set `--c60_diameter` accordingly.
+- `--c60_diameter`, `--c60_buffer`: Heuristic controls for `--auto_supercell` (defaults: 7.1 Angstrom and 3.0 Angstrom). For other fullerenes (e.g., C70), set `--c60_diameter` accordingly.
- `--adsorbate_xy fx,fy`: Optional lateral placement of the C60 geometric center in fractional coordinates
### Outputs
@@ -210,7 +226,7 @@ python build_heterojunctions/fix_interface_layers.py \
- `input_file`: Input POSCAR file (combined heterojunction from Step 1)
- `-o, --output`: Output POSCAR file (default: input_file with `_relaxed` suffix)
- `-n, --n_layers`: Number of layers to relax on each side of interface (default: 1)
-- `-t, --thickness`: Layer thickness threshold in Å (default: 2.0)
+- `-t, --thickness`: Layer thickness threshold in Angstrom (default: 2.0)
- `-m, --method`: Interface identification method: `density_gap`, `median`, or `max_gap` (default: `density_gap`)
### Mode B: split a stacked heterostructure into layers and fix selected layer(s)
@@ -219,11 +235,11 @@ For stacked models with multiple interfaces (e.g. perovskite/C70/C60), it is oft
**split the structure into (n_interfaces + 1) layers** and then fix one or more complete layers.
Rule:
-- 1 interface → 2 layers
-- 2 interfaces → 3 layers
+- 1 interface -> 2 layers
+- 2 interfaces -> 3 layers
- ...
-Example (2 interfaces → 3 layers; fix the bottom layer):
+Example (2 interfaces -> 3 layers; fix the bottom layer):
```bash
python build_heterojunctions/fix_interface_layers.py your_stacked_model.vasp \
@@ -254,7 +270,7 @@ python build_heterojunctions/fix_interface_layers.py structures/heterojunctions/
```
Notes:
-- `--tol` is the layer clustering tolerance in Å; if omitted, it is auto-estimated.
+- `--tol` is the layer clustering tolerance in Angstrom; if omitted, it is auto-estimated.
- For CP2K `.xyz`, include `Tv_1/Tv_2/Tv_3` in the comment line whenever possible.
- By default, the script ensures **whole organic molecules** (e.g., MA/FA) are not cut by the layer boundary:
if any atom of a molecule is selected in the fixed region, **all atoms of that molecule** are included.
@@ -317,7 +333,7 @@ For an input file `PROJECT-1.cif`, the script creates:
Compute difference charge density (best for heterojunctions):
-- **Formula**: Δρ(r) = ρ_interface − ρ_layerA − ρ_layerB
+- **Formula**: Delta rho(r) = rho_interface - rho_layerA - rho_layerB
- **Output**: `delta-density.cube` (Gaussian cube format; open directly in **VESTA**)
### Example (PROJECT-1)
@@ -355,7 +371,7 @@ python build_heterojunctions/stack_slabs_relax.py [gap
### Examples
-**Example 1**: Stack two pre-relaxed slabs with 3 Å gap and 20 Å vacuum
+**Example 1**: Stack two pre-relaxed slabs with 3 Angstrom gap and 20 Angstrom vacuum
```bash
python build_heterojunctions/stack_slabs_relax.py \
@@ -366,11 +382,11 @@ python build_heterojunctions/stack_slabs_relax.py \
```
This will create `relax_slab/fapbi3@tio2_fa_stacked.vasp` with structure:
-- Bottom vacuum (10 Å, half of total 20 Å)
+- Bottom vacuum (10 Angstrom, half of total 20 Angstrom)
- Slab 1 (fapbi3)
-- Gap (3 Å)
+- Gap (3 Angstrom)
- Slab 2 (tio2_fa)
-- Top vacuum (10 Å, half of total 20 Å)
+- Top vacuum (10 Angstrom, half of total 20 Angstrom)
**Example 2**: Stack slabs with custom gap and vacuum
@@ -427,8 +443,8 @@ The script displays:
- **Output location**: By default, stacked structures are saved to the `relax_slab/` folder
- **File naming**: Output files are named as `{slab1}@{slab2}_stacked.vasp` (e.g., `fapbi3@tio2_fa_stacked.vasp`)
- **Duplicate handling**: If a file with the same name already exists, a number suffix will be added (e.g., `fapbi3@tio2_fa_stacked_1.vasp`)
-- **Vacuum layers**: The structure includes vacuum layers at both bottom and top. The specified vacuum thickness is the total (default: 20 Å), which is split equally between bottom and top (10 Å each) to ensure proper isolation for DFT calculations
-- **Structure layout**: The final structure follows: [bottom vacuum] → [slab1] → [gap] → [slab2] → [top vacuum]
+- **Vacuum layers**: The structure includes vacuum layers at both bottom and top. The specified vacuum thickness is the total (default: 20 Angstrom), which is split equally between bottom and top (10 Angstrom each) to ensure proper isolation for DFT calculations
+- **Structure layout**: The final structure follows: [bottom vacuum] -> [slab1] -> [gap] -> [slab2] -> [top vacuum]
### Use Cases
@@ -443,7 +459,7 @@ This tool is useful when:
1. **Use `--use_builder_interface`**: Recommended for ordered, high-symmetry interfaces
2. **Atom count control**: Use `--max_atoms` to limit system size (the script may adjust thickness if needed)
3. **Selective dynamics**: `-n 2` (2 layers per side) is a good default for many DFT relaxations
-4. **Layer thickness**: Tune `-t` (often 2.0–3.0 Å) based on your material’s layer spacing
+4. **Layer thickness**: Tune `-t` (often 2.0-3.0 Angstrom) based on your material's layer spacing
5. **Strain target**:
- `A`: Strain material A to match B (good when A is more flexible)
- `B`: Strain material B to match A (good when B is more flexible)
diff --git a/build_heterojunctions/__pycache__/__init__.cpython-313.pyc b/build_heterojunctions/__pycache__/__init__.cpython-313.pyc
deleted file mode 100644
index 313cc001..00000000
Binary files a/build_heterojunctions/__pycache__/__init__.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/_utils_layering.cpython-313.pyc b/build_heterojunctions/__pycache__/_utils_layering.cpython-313.pyc
deleted file mode 100644
index 44e6cf8b..00000000
Binary files a/build_heterojunctions/__pycache__/_utils_layering.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/_utils_structures.cpython-313.pyc b/build_heterojunctions/__pycache__/_utils_structures.cpython-313.pyc
deleted file mode 100644
index 4e0892f9..00000000
Binary files a/build_heterojunctions/__pycache__/_utils_structures.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/_utils_xyz.cpython-313.pyc b/build_heterojunctions/__pycache__/_utils_xyz.cpython-313.pyc
deleted file mode 100644
index 131767b8..00000000
Binary files a/build_heterojunctions/__pycache__/_utils_xyz.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/auto_interface_builder.cpython-310.pyc b/build_heterojunctions/__pycache__/auto_interface_builder.cpython-310.pyc
deleted file mode 100644
index 96bb1f3e..00000000
Binary files a/build_heterojunctions/__pycache__/auto_interface_builder.cpython-310.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/auto_interface_builder.cpython-313.pyc b/build_heterojunctions/__pycache__/auto_interface_builder.cpython-313.pyc
deleted file mode 100644
index 66580981..00000000
Binary files a/build_heterojunctions/__pycache__/auto_interface_builder.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/delta_density_cube.cpython-313.pyc b/build_heterojunctions/__pycache__/delta_density_cube.cpython-313.pyc
deleted file mode 100644
index dce1c296..00000000
Binary files a/build_heterojunctions/__pycache__/delta_density_cube.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/fix_interface_layers.cpython-313.pyc b/build_heterojunctions/__pycache__/fix_interface_layers.cpython-313.pyc
deleted file mode 100644
index 6d6f13d6..00000000
Binary files a/build_heterojunctions/__pycache__/fix_interface_layers.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/fixed_atoms_from_xyz.cpython-313.pyc b/build_heterojunctions/__pycache__/fixed_atoms_from_xyz.cpython-313.pyc
deleted file mode 100644
index 9d8720ea..00000000
Binary files a/build_heterojunctions/__pycache__/fixed_atoms_from_xyz.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/__pycache__/interface_builder.cpython-313.pyc b/build_heterojunctions/__pycache__/interface_builder.cpython-313.pyc
deleted file mode 100644
index 107d3734..00000000
Binary files a/build_heterojunctions/__pycache__/interface_builder.cpython-313.pyc and /dev/null differ
diff --git a/build_heterojunctions/_utils_layering.py b/build_heterojunctions/_utils_layering.py
index c98bca1f..9f29cc11 100644
--- a/build_heterojunctions/_utils_layering.py
+++ b/build_heterojunctions/_utils_layering.py
@@ -262,3 +262,45 @@ def include_whole_molecules(
fixed_set.update(comp)
return sorted(fixed_set)
+
+# Prefer canonical core implementations when available
+try:
+ from interfaceml.core import layering as _core_layering
+except ImportError: # pragma: no cover - fallback for standalone usage
+ _core_layering = None # type: ignore[assignment]
+
+if _core_layering is not None:
+ interface_normal_unit = _core_layering.interface_normal_unit # type: ignore[assignment]
+ unwrap_periodic_1d = _core_layering.unwrap_periodic_1d # type: ignore[assignment]
+ split_stack_layers = _core_layering.split_stack_layers # type: ignore[assignment]
+ connected_components_by_distance = _core_layering.connected_components_by_distance # type: ignore[assignment]
+
+ def auto_layer_tol(diffs: np.ndarray) -> float: # type: ignore[override]
+ return _core_layering.auto_layer_tolerance(diffs)
+
+ def split_layers_by_z( # type: ignore[override]
+ structure: Structure,
+ *,
+ tol: float | None = None,
+ gap_cut: bool = True,
+ ) -> tuple[list[list[int]], float]:
+ return _core_layering.split_layers_by_z(structure, tolerance=tol, gap_cut=gap_cut)
+
+ def include_whole_molecules( # type: ignore[override]
+ structure: Structure,
+ fixed_indices_0based: list[int],
+ *,
+ molecule_elements: Set[str] | None = None,
+ ) -> list[int]:
+ return _core_layering.include_whole_molecules(
+ structure,
+ fixed_indices_0based,
+ molecule_elements=molecule_elements,
+ )
+
+ def layer_indices_to_string( # type: ignore[override]
+ indices_0based: list[int],
+ *,
+ one_based: bool = True,
+ ) -> str:
+ return _core_layering.format_layer_indices(indices_0based, one_based=one_based)
diff --git a/build_heterojunctions/_utils_structures.py b/build_heterojunctions/_utils_structures.py
index ac2a92f2..2fe5c434 100644
--- a/build_heterojunctions/_utils_structures.py
+++ b/build_heterojunctions/_utils_structures.py
@@ -7,17 +7,26 @@
from __future__ import annotations
from pathlib import Path
-from typing import Optional
import numpy as np
from pymatgen.core import Lattice, Structure
try:
- # When executed as a module: `python -m build_heterojunctions.
+
+ {% block extra_head %}{% endblock %}
+
+
+{% block body %}{% endblock %}
+{% block extra_scripts %}{% endblock %}
+
+