From d9cc9bdbe038c65f39f95d517292bafafd944b87 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 11:49:44 -0700 Subject: [PATCH 01/36] feat(SOF-8051): notebook that uploads an SPM run A run folder is what the microscope leaves behind; the platform's side of it is a Sample Set with one Sample per measured position, a Measurement Set with one Measurement per Sample, the records as files and one hysteresis-loop Property per Sample. The notebook fetches upload_run.py from the host it uploads to, which is the copy the web app serves, so the instructions in the app and the script a reader runs cannot drift apart; parse() then prints the script's own summary, and nothing is created until the upload cell below it. The Authenticate block is the sibling notebooks' verbatim. mat3ra-api-client is installed from its branch in the cell above it, because samples, measurements and files are not in a release yet. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_spm_run.ipynb | 248 ++++++++++++++++++++++ 1 file changed, 248 insertions(+) create mode 100644 examples/measurement/upload_spm_run.ipynb diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb new file mode 100644 index 000000000..342e10e73 --- /dev/null +++ b/examples/measurement/upload_spm_run.ipynb @@ -0,0 +1,248 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Overview\n", + "\n", + "A run folder is one scanning probe microscopy run as the instrument exports it: `summary.json` with the recipe, the session and one record per measured point, and `loops/` with the raw curves.\n", + "This example creates the specimen on the platform as a Sample Set holding one Sample per measured position, the run as a Measurement Set holding one Measurement per Sample with the Setup it was measured on, the run's records as files on each Measurement, and one hysteresis-loop Property per Sample.\n", + "Re-running it adds only what is missing." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Install the API client\n", + "\n", + "The samples, measurements and files endpoints are not released yet, so the client is installed from its branch until it merges." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "%pip install -q \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Authenticate and initialize API client\n", + "\n", + "### Authenticate\n", + "Authenticate in the browser (OIDC device flow) or via JupyterLite host injection. Credentials are stored in environment variables.\n", + "\n", + "### Initialize API client\n", + "Create an authenticated API client and resolve the owner account ID." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from mat3ra.notebooks_utils.packages import install_packages\n", + "\n", + "await install_packages(\"api\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from mat3ra.notebooks_utils.auth import authenticate\n", + "\n", + "await authenticate()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import os\n", + "\n", + "from mat3ra.api_client import APIClient\n", + "\n", + "client = APIClient.authenticate()\n", + "selected_account = client.my_account\n", + "OWNER_ID = os.getenv(\"ORGANIZATION_ID\") or selected_account.id" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Set Parameters\n", + "\n", + "- **HOST**: platform the run is uploaded to\n", + "- **RUN_DIR**: path to the run folder, relative to this notebook\n", + "- **ACCOUNT_SLUG**: account the data belongs to, empty for the default account\n", + "- **FILES**: which files to upload per measurement" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "HOST = os.environ.get(\"MAT3RA_HOST\", \"https://platform.mat3ra.com\")\n", + "RUN_DIR = \"20260901_171601_alscn_01448\"\n", + "ACCOUNT_SLUG = \"\"\n", + "FILES = \"records\" # \"records\": the record JSONs, \"all\": also the loop arrays and plots, \"none\": no files" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Fetch the uploader\n", + "\n", + "Download `upload_run.py`, the parser and uploader the platform serves, into the working directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from pathlib import Path\n", + "\n", + "from mat3ra.notebooks_utils.io import read_from_url\n", + "\n", + "if not Path(\"upload_run.py\").exists():\n", + " Path(\"upload_run.py\").write_text(await read_from_url(f\"{HOST}/upload_run.py\"))\n", + "\n", + "from upload_run import account_id, parse, upload" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Parse the run folder\n", + "\n", + "Read the run folder into the documents the platform stores. Nothing is uploaded yet." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "parsed = parse(Path(RUN_DIR))\n", + "file_count = sum(len(files) for files in parsed[\"files\"].values())\n", + "print(\n", + " f\"specimen {parsed['wafer']}: {len(parsed['samples'])} samples (ordered set) · run {parsed['run']}: \"\n", + " f\"{len(parsed['measurements'])} measurements (ordered set, one per sample) · {len(parsed['records'])} records \"\n", + " f\"-> {file_count} files · {len(parsed['images'])} image(s) · {len(parsed['properties'])} samples with a combined \"\n", + " f\"loop · no curves: {len(parsed['skipped'])} samples\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Select the account\n", + "\n", + "`ACCOUNT_SLUG` re-authenticates the client against that account, so the run is read and written there." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "if ACCOUNT_SLUG:\n", + " client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Upload the run\n", + "\n", + "Create the Sample Set and its Samples, the Measurement Set and one Measurement per Sample, the files and the loop Properties." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "upload(client, parsed, command=\"both\", files=FILES)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Print the link to the run\n", + "\n", + "The run is a folder in the account's Measurements tab, named after the run." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "account = next(item for item in client.list_accounts() if item[\"_id\"] == client.my_account.id)\n", + "print(f\"{HOST}/{account['slug']}/measurements\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## References\n", + "\n", + "- [Mat3ra REST API](https://docs.mat3ra.com/rest-api/overview/)\n", + "- [Samples and measurements in the web app](https://github.com/mat3ra/web-app/pull/2976)" + ] + } + ], + "metadata": { + "colab": { + "name": "upload_spm_run.ipynb", + "provenance": [] + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.6" + } + }, + "nbformat": 4, + "nbformat_minor": 1 +} From be4b3c6d8a3b35ad74c2344381fc4620a4d038bb Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 12:24:20 -0700 Subject: [PATCH 02/36] fix(SOF-8051): the upload notebook carries its script and one host Fetching upload_run.py at run time made the notebook depend on whatever a host happens to serve: production answers an unknown path with the SPA shell, so the default wrote 4 KB of HTML into upload_run.py and died on the import, and the deployed copy is three revisions behind - its upload() takes its arguments the other way round. The script now sits beside the notebook, copied from the canonical one the plan repo keeps, and is imported like any other module. HOST was a second knob for a fact the client already held: it named where the script came from and what link was printed, while APIClient.authenticate() read API_HOST, so exporting one and not the other uploads a lab run to production and prints a localhost link for it. It is parsed once and passed to both authentications, the way upload_run.py's main() does it. The rest is the review's list: OWNER_ID and selected_account are dead here, since upload() owns every write with client.my_account.id; RUN_DIR names a folder the reader supplies and the markdown says what one is; the default command="both" and the private PR link go; and the last cell prints where to look from what the run already says, rather than asking for an account slug through a call that needs an OIDC token the ACCOUNT_ID/AUTH_TOKEN flow does not have. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_run.py | 576 ++++++++++++++++++++++ examples/measurement/upload_spm_run.ipynb | 91 ++-- mkdocs.yml | 2 + 3 files changed, 621 insertions(+), 48 deletions(-) create mode 100755 examples/measurement/upload_run.py diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py new file mode 100755 index 000000000..be8fa5c5a --- /dev/null +++ b/examples/measurement/upload_run.py @@ -0,0 +1,576 @@ +#!/usr/bin/env python3 +"""UTK run dir -> platform documents (sample set, samples, measurement, one hysteresis-loop property per sample), +validated against the ESSE schemas, then uploaded through the REST API. + +Ad hoc parser for SOF-8050. Field kinds follow ONTOLOGY.md. The property model follows PLAN-S3-loop-property.md: +the property is the pad's hysteresis loop — the eight loops combined — with the loop parameters (mean, population +standard deviation, count over the loops) inside it. Individual loops stay in the measurement's metadata. + + upload_run.py --account # upload everything + upload_run.py --dry-run [--emit-example out.json] # parse + validate only; write one property as the ESSE example + +Requires Python 3.9+ and `pip install mat3ra-api-client`, which talks to the platform and takes OIDC_ACCESS_TOKEN, or +ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. Optional: +`pip install mat3ra-esse` (tested with 2026.8.27-0) turns on schema validation before anything is uploaded. + +Canonical copy: mat3ra/web-app `src/application/public/upload_run.py` (served by the web app at /upload_run.py, so the +instructions in the app always match). The planning repo keeps a working copy and the tests; scripts/publish.sh +copies web-app → plan by default and plan → web-app with --push. +""" +import argparse, ast, concurrent.futures, json, math, os, re, statistics, struct, sys, threading, time, urllib.parse, uuid +from datetime import datetime, timezone +from pathlib import Path + +import requests +from mat3ra.api_client import APIClient + +try: # optional: schema validation before anything is sent + from mat3ra.esse import ESSE + from mat3ra.esse.models.sample import SampleSchema +except ImportError: + ESSE = SampleSchema = None + +WAFER_ID = re.compile(r"(PDAC_COM\d+_\d+)") +FIELD = {"off_field": "off", "on_field": "on"} +# loop_params key -> parameters path (units: voltages in xAxis.units, responses in yAxis.units) +PARAMETERS = { + "imprint_v": ("imprint",), + "v_c_rising": ("coerciveVoltage", "rising"), + "v_c_falling": ("coerciveVoltage", "falling"), + "loop_width_v": ("loopWidth",), + "loop_height_m": ("loopHeight",), + "remnant_rising_m": ("remanentResponse", "rising"), + "remnant_falling_m": ("remanentResponse", "falling"), +} +INSTRUMENT = {"name": "asylum-spm", "shortName": "spm", "summary": "Asylum Research SPM driven by afm-lib (switching-spectroscopy PFM)", + "version": "1.0", "build": "afm-lib", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} +g = lambda v: float(f"{v:.6g}") + + +def load_npy(path): + """A one-dimensional float32/float64 .npy file as a list of floats (no numpy: the format is a header + raw values).""" + data = Path(path).read_bytes() + if data[:6] != b"\x93NUMPY": + raise ValueError(f"{path}: not a .npy file") + header_length = struct.unpack(" 1 else 0.0, "count": len(values)} + + +def combine_pad(label, records, run_dir): + """One hysteresis_loop property for a pad: point-wise mean of the response over its loops (on and off), and + each loop parameter as mean / spread / count over the loops. None when no loop has curves.""" + bias, series, params = None, {"on": [], "off": []}, {"on": {}, "off": {}} + for r in records: + loops_dir = run_dir / "loops" / (r.get("out_stem") or Path(r["file_path"]).stem) + for branch, field in FIELD.items(): + lp = r["loop_params"].get(branch) or {} + for key, path in PARAMETERS.items(): + if lp.get(key) is not None: + params[field].setdefault(path, []).append(lp[key]) + bias_p = loops_dir / f"bias_{field}.npy" + if not bias_p.exists() or lp.get("phase_offset_deg") is None: + continue + b = load_npy(bias_p) + bias = bias or b + curve = response_curve(loops_dir, field, lp["phase_offset_deg"], b) + if curve is not None and len(curve) == len(bias): + series[field].append(curve) + if bias is None or not series["on"] or not series["off"]: + return None + parameters = {} + for field in ("on", "off"): + block = {} + for path, vals in params[field].items(): + node = block + for k in path[:-1]: + node = node.setdefault(k, {}) + node[path[-1]] = parameter_statistics(vals) + parameters[field] = block + return {"name": "hysteresis_loop", "legend": ["on", "off"], + "xAxis": {"label": "bias", "units": "V"}, "yAxis": {"label": "response", "units": "m"}, + "xDataArray": [g(v) for v in bias], "yDataSeries": [pointwise_mean(series["on"]), pointwise_mean(series["off"])], + "parameters": parameters} + + +WORKFLOW_NAMESPACE = uuid.UUID("6f3b0b0e-8c1e-4b7a-9f21-3a5f0e2d1c44") # stable ids: the same workflow every upload + + +def build_workflow(recipe, labels): + """The procedure that runs on UTK's Asylum SPM, the same shape as standata's `asylum-spm/ss_pfm` workflow: ONE + workflow "SS-PFM Hysteresis Loop", one subworkflow `ss_pfm`, one execution unit `run_loop` declaring + `hysteresis_loop`. It is the same workflow for every sample in the set — the measurement is not re-planned per + sample — and its ids are stable (uuid5 of the unit names), so a property's `source.info.unitId` means the same + thing across uploads. Application / executable / flavor are the standata registry entries (asylum-spm / loop / + ss_pfm) so the platform resolves the unit exactly as it does a job's. The recipe travels in metadata.""" + result = [{"name": "hysteresis_loop"}] + monitors = [{"name": "standard_output"}] + executable = {"name": "loop", "applicationName": INSTRUMENT["name"], "applicationVersion": "*", "isDefault": True, + "monitors": monitors, "results": result, "preProcessors": [], "postProcessors": []} + flavor = {"name": "ss_pfm", "executableName": "loop", "applicationName": INSTRUMENT["name"], "applicationVersion": "*", + "isDefault": True, "input": [], "monitors": monitors, "results": result, "preProcessors": [], "postProcessors": []} + unit = {"type": "execution", "name": "run_loop", "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, "run_loop").hex[:24], "head": True, "status": "finished", + "application": INSTRUMENT, "executable": executable, "flavor": flavor, "input": [], "context": [], + "monitors": monitors, "results": result, "preProcessors": [], "postProcessors": []} + # ESSE requires a model on every subworkflow; an experiment has none, so the legacy "unknown" model. + model = {"type": "unknown", "subtype": "unknown", "method": {"type": "unknown", "subtype": "unknown"}} + sw_id = uuid.uuid5(WORKFLOW_NAMESPACE, "ss_pfm").hex[:17] + subworkflow = {"_id": sw_id, "name": "ss_pfm", "application": INSTRUMENT, "model": model, + "properties": ["hysteresis_loop"], "units": [unit]} + # A subworkflow unit carries the subworkflow's own `_id` — that is how the platform pairs them. + sw_unit = {"_id": sw_id, "type": "subworkflow", "name": subworkflow["name"], "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, "ss_pfm/unit").hex[:24], + "head": True, "status": "finished", "preProcessors": [], "postProcessors": [], "monitors": [], "results": []} + return {"name": "SS-PFM Hysteresis Loop", "isDefault": False, "tags": ["experimental", "afm"], "properties": ["hysteresis_loop"], + "application": INSTRUMENT, "subworkflows": [subworkflow], "units": [sw_unit], "workflows": [], + "metadata": {"recipe": recipe, "loop_settings": recipe["per_site"][0]["loop_settings"], "sites": list(labels)}} + + + +RECORD_GROUPS = ("labels", "file_path", "requested_params", "instrument_params", "loop_params", "channel_stats") +SAMPLE_KEY = "labels.site_label" # a record's reference to its sample — stays on every record + + +def _flatten(d, prefix=""): + """Nested dict → one level, keys joined with dots.""" + out = {} + for k, v in (d or {}).items(): + key = f"{prefix}.{k}" if prefix else k + if isinstance(v, dict): + out.update(_flatten(v, key)) + else: + out[key] = v + return out + + +def _unflatten(flat): + """The inverse of _flatten.""" + out = {} + for key, v in flat.items(): + node = out + parts = key.split(".") + for part in parts[:-1]: + node = node.setdefault(part, {}) + node[parts[-1]] = v + return out + + +def _constant_keys(flat_rows): + """Keys present in every row with the same value everywhere. A key one row lacks is not constant, even if the + rows that have it agree — otherwise a group that is null on one record and absent on another looks shared.""" + if not flat_rows: + return set() + keys = set.intersection(*(set(row) for row in flat_rows)) + return {k for k in keys if all(row[k] == flat_rows[0][k] for row in flat_rows)} + + +def factor_records(records): + """Each record keeps only what is unique to it. Fields identical across the whole run move to the measurement + (`common`); fields identical across one sample's records move to that sample's metadata; the record keeps its + sample reference and the values that actually vary per loop (measured 128 → 43 / 4 / 81 on the 704-record run).""" + if not records: + return {}, {}, [] + flat = [_flatten({k: r.get(k) for k in RECORD_GROUPS}) for r in records] + common_keys = _constant_keys(flat) - {SAMPLE_KEY} + by_sample = {} + for row in flat: + by_sample.setdefault(row[SAMPLE_KEY], []).append(row) + sample_keys = set.intersection(*(_constant_keys(rows) for rows in by_sample.values())) - common_keys - {SAMPLE_KEY} + # a group can be a dict on one record and null on another: read with .get, never index + common = _unflatten({k: flat[0].get(k) for k in sorted(common_keys)}) + per_sample = {label: _unflatten({k: rows[0].get(k) for k in sorted(sample_keys)}) for label, rows in by_sample.items()} + slim = [_unflatten({k: v for k, v in row.items() if k not in common_keys and k not in sample_keys}) for row in flat] + return common, per_sample, slim + +def registration(recipe): + """The instrument's frame as UTK stated it: one anchor in words (from recipe.context) and where r0c00 sits on the stage. + Recorded, not interpreted — a second anchor is needed for a rigid map (PROPOSAL §3.5).""" + ctx = recipe.get("context", "") + m = re.search(r"starting point is (.+?)(?:\.|$)", ctx) + r0 = next((s for s in recipe["sites"] if s["label"] == "r0c00"), recipe["sites"][0]) + return {"frame": "asylum-spm stage", "units": "m", "anchor": m.group(1).strip() if m else ctx, + f"{r0['label']}_stage_m": [r0["x_stage_m"], r0["y_stage_m"]]} + + +def sample_files(label, records, run_dir, slim_by_index): + """Files of one sample's measurement: one JSON per record (the fields unique to it) and, when the run folder has them, + the loop arrays and annotated plots. Returned as (relative name, payload) where payload is text or a Path.""" + out = [] + for r, slim in zip(records, slim_by_index): + step, point = r["labels"].get("step_index", 0), r["labels"].get("point_index", 0) + out.append((f"records/step{step}_pt{point:02d}.json", json.dumps(slim, indent=1))) + d = run_dir / "loops" / r.get("out_stem", "") + if r.get("out_stem") and d.is_dir(): + for f in sorted(d.iterdir()): + if f.suffix in (".npy", ".png"): + out.append((f"loops/{d.name}/{f.name}", f)) + return out + + +def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm"): + """The whole run folder as platform documents: sample set, samples, measurement set, one measurement per sample, files, one loop property per fully measured sample.""" + run_dir = Path(run_dir) + recipe, session, all_records = load_run(run_dir) + records = all_records[:limit_records] if limit_records else all_records + wid = wafer_id(recipe) + run_name = session.get("name") or run_dir.name + # the specimen's physical ID is the case-sticker text; it is how NLR's and UTK's data find the same set + sample_set = {"name": wid, "entitySetType": "ordered", + "metadata": {"label": wid, "recipe": recipe["name"], "context": recipe.get("context", "")}} + # NLR's HTEM deposition record(s) for this wafer, verbatim — the specimen's synthesis data. UTK drops the file into + # the run folder as deposition*.json; --deposition overrides that. + deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) + if deposition_files: + deposition_records = [] + for f in deposition_files: + d = json.loads(f.read_text()) + deposition_records.extend(d if isinstance(d, list) else [d]) + sample_set["metadata"]["deposition"] = deposition_records + # the specimen photograph: any image at the run-folder root + images = [f for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] + # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) + samples = {s["label"]: {"name": f"{wid} {s['label']}", "label": s["label"], + "position": {"coordinates": [s["x_stage_m"], s["y_stage_m"]], "units": "m"}, "metadata": {}} + for s in recipe["sites"]} + if limit_records: + # a trial run must be a prefix of a full one: only samples whose records ALL made the cut, so no sample is + # ever published with a partial loop count that a full run would then skip as "already there" + full_counts, kept_counts = {}, {} + for r in all_records: + full_counts[r["labels"]["site_label"]] = full_counts.get(r["labels"]["site_label"], 0) + 1 + for r in records: + kept_counts[r["labels"]["site_label"]] = kept_counts.get(r["labels"]["site_label"], 0) + 1 + complete = {label for label, n in kept_counts.items() if n == full_counts[label]} + records = [r for r in records if r["labels"]["site_label"] in complete] + samples = {label: sample for label, sample in samples.items() if label in complete} + common, per_sample, slim_records = factor_records(records) + for label, const in per_sample.items(): + if label in samples: + samples[label]["metadata"] = const + reg = registration(recipe) + workflow = build_workflow(recipe, list(samples)) + unit_id = workflow["subworkflows"][0]["units"][0]["flowchartId"] + measurement_set = {"name": run_name, "entitySetType": "ordered", + "metadata": {"session": session, "recipe": recipe["name"], "context": recipe.get("context", ""), + "common": common, "registration": reg}} + # the setup block, Measurement : setup :: Job : compute — the machine and the sitting; the technique + # (asylum-spm, SS-PFM) is the workflow's application. The run folder does not name the machine: --instrument does. + started = session.get("started_ts") + setup_block = {"name": instrument, + "session": {k: v for k, v in {"name": session.get("name"), + "started": datetime.fromtimestamp(started, timezone.utc).isoformat().replace("+00:00", "Z") if started else None, + "directory": session.get("instrument_directory")}.items() if v}, + **({"settings": common["instrument_params"]} if common.get("instrument_params") else {}), + "registration": reg} + by_sample, slim_by_sample = {}, {} + for r, slim in zip(records, slim_records): + lab = r["labels"]["site_label"] + by_sample.setdefault(lab, []).append(r); slim_by_sample.setdefault(lab, []).append(slim) + # one measurement per sample: Measurement : Sample :: Job : Material + measurements, files, properties, skipped = {}, {}, [], [] + for label in samples: + recs = by_sample.get(label, []) + measurements[label] = {"name": f"{run_name} {label}", "_sample": None, "workflow": workflow, + "setup": setup_block, "status": "finished", + "metadata": {"run_dir": session.get("run_dir", str(run_dir)), "recordsCount": len(recs)}} + slim_by_label = slim_by_sample.get(label, []) + measurements[label]["_records"] = slim_by_label # not sent; upload() decides files vs metadata + files[label] = sample_files(label, recs, run_dir, slim_by_sample.get(label, [])) + prop = combine_pad(label, recs, run_dir) if recs else None + (properties.append((label, unit_id, prop, 0)) if prop else skipped.append(label)) + return {"wafer": wid, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, + "measurement_set": measurement_set, "measurements": measurements, "files": files, + "records": records, "properties": properties, "skipped": skipped} + + +def holder(prop, measurement_id, sample_id, unit_id, repetition): + """The property holder the platform stores: the data, where it came from (measurement, sample, workflow unit) and a + repetition index — 0, since a measurement holds one sample and one loop property.""" + return {"data": prop, + "source": {"type": "external", + "info": {"origin": {"_id": measurement_id, "cls": "Measurement"}, + "subject": {"_id": sample_id, "cls": "Sample"}, + "unitId": unit_id}}, + "exabyteId": [], "repetition": repetition} + + +def validate(parsed): + """Validate every document against the ESSE schemas; returns the number of invalid ones. Skipped (returns 0) when the + optional mat3ra-esse package is not installed.""" + if ESSE is None: + print("schema validation skipped: `pip install mat3ra-esse` to enable it", flush=True) + return 0 + esse = ESSE() + schemas = {x["$id"]: x for x in esse.schemas} + errors = 0 + for smp in parsed["samples"].values(): + try: + SampleSchema(**smp); esse.validate(smp, schemas["sample"]) + except Exception as e: + errors += 1; print("SAMPLE INVALID", smp["label"], str(e)[:200]) + for label, m in parsed["measurements"].items(): + m = {k: v for k, v in m.items() if k != "_records"}; m["_sample"] = {"_id": "dryrun", "cls": "Sample", "slug": label} + try: + esse.validate(m, schemas["measurement"]) + except Exception as e: + errors += 1; print("MEASUREMENT INVALID", label, str(e)[:300]); break + for label, uid, prop, rep in parsed["properties"]: + try: + esse.validate(prop, schemas["properties-directory/non-scalar/hysteresis-loop"]) + esse.validate(holder(prop, "dryrun", "dryrun", uid, rep), schemas["property/holder"]) + except Exception as e: + errors += 1; print("PROPERTY INVALID", label, str(e)[:300]) + return errors + + +def base_url(host): + """`https://` unless told otherwise: a bare hostname becomes https, localhost/127.0.0.1 http, a URL is kept.""" + host = host.rstrip("/") + if host.startswith(("http://", "https://")): + return host + return ("http://" if host.split(":")[0] in ("localhost", "127.0.0.1") else "https://") + host + + +def find(endpoint, query, owner_id, limit=100): + """The account's entities matching `query` — a query bypasses the route's account scoping, so every lookup is + narrowed. A page as long as the limit is refused: a missed member means a duplicate on the next run.""" + found = endpoint.list(dict(query, **{"owner._id": owner_id}), {"limit": limit}) + if len(found) >= limit and limit > 1: + raise SystemExit(f"{endpoint.name}: more than {limit} match — this script cannot page yet; stop rather than duplicate") + return found + + +def account_id(client, slug_or_name): + """`--account ` → the id of that account: the slug the platform shows it under, else its display name.""" + accounts = client.list_accounts() + for field in ("slug", "name"): + account = next((a for a in accounts if a.get(field) == slug_or_name), None) + if account: + return account["_id"] + raise SystemExit(f"account '{slug_or_name}' is not one of yours: " + f"{', '.join(a.get('slug') or a['name'] for a in accounts)}") + + +THREAD = threading.local() + + +def thread_client(client): + """A client of the calling thread's own: the endpoints keep the last response on the connection they share, so + the file workers cannot use one between them. Building one costs nothing — no request is made until it is used.""" + if not hasattr(THREAD, "client"): + THREAD.client = APIClient(host=client.host, port=client.port, version=client.version, secure=client.secure, + auth=client.auth, timeout_seconds=client.timeout_seconds) + return THREAD.client + + +def put_file(client, name, payload, owner_id): + """One file into the account's file store: a file on disk through a signed PUT, text through the files route. + Eight connections at once is enough to make a name resolution fail now and then, so a refused connection is + tried again twice before it takes the run down with it.""" + for attempt in range(3): + try: + if isinstance(payload, Path): + return client.files.put(payload, name, account_id=owner_id) + return client.files.create(name, payload, account_id=owner_id) + except (requests.exceptions.ConnectionError, requests.exceptions.Timeout): + if attempt == 2: + raise + time.sleep(2 * (attempt + 1)) + + +def ensure_set(endpoint, doc, owner_id): + """The set with this name in the account, created when missing; returns (set, created).""" + found = find(endpoint, {"isEntitySet": True, "name": doc["name"]}, owner_id, 5) + return (found[0], False) if found else (endpoint.create_set(dict(doc, owner={"_id": owner_id})), True) + + +def find_set_by_label(endpoint, label, owner_id): + """The specimen is found by its physical ID (metadata.label); older sets by name.""" + found = find(endpoint, {"isEntitySet": True, "metadata.label": label}, owner_id, 5) or \ + find(endpoint, {"isEntitySet": True, "name": label}, owner_id, 5) + return found[0] if found else None + + +def upload(client, parsed, command="both", files="records"): + """Two separate imports joined by the Sample Set's physical label: + `synthesis` — NLR's record(s) + the photograph onto the set (created if missing; no samples, no measurements); + `measurement` — UTK's run: the set (created bare if NLR has not arrived), its samples, the measurement set tied to + it, one measurement per sample, files, properties; + `both` — synthesis if the folder has a deposition record, then measurement. + Idempotent: sets by label, members by name/label; files re-put; properties posted only when missing.""" + wafer_label, run_name = parsed["wafer"], parsed["run"] + owner = {"_id": client.my_account.id} + has_deposition = "deposition" in parsed["sample_set"]["metadata"] + sample_set = find_set_by_label(client.samples, wafer_label, owner["_id"]) + created_set = False + if sample_set is None: + doc = dict(parsed["sample_set"], owner=owner) + if command == "measurement": # UTK arrived first: a bare set, the synthesis record comes later + doc["metadata"] = {k: v for k, v in doc["metadata"].items() if k != "deposition"} + sample_set = client.samples.create_set(doc) + created_set = True + set_id = sample_set["_id"] + if command in ("synthesis", "both") and has_deposition and not created_set: + # the set was here first (UTK arrived before NLR): attach the record now + patch = {k: v for k, v in parsed["sample_set"]["metadata"].items() if k in ("label", "deposition")} + client.samples.update_set(set_id, {"metadata": patch}) + if command in ("synthesis", "both") and has_deposition: + print(f"sample set {set_id} ({wafer_label}{', created' if created_set else ', updated'}): synthesis record attached") + if command in ("synthesis", "both") and files != "none": + for image in parsed["images"]: + put_file(client, f"sets/{set_id}/{image.name}", image, owner["_id"]) + print(f" image {image.name} -> sets/{set_id}/") + if command == "synthesis": + return + if not created_set and not (sample_set.get("metadata") or {}).get("label"): + client.samples.update_set(set_id, {"metadata": {"label": wafer_label}}) + in_set = {s.get("label"): s for s in find(client.samples, {"inSet._id": set_id, "isEntitySet": {"$ne": True}}, owner["_id"], 500)} + sample_ids, created = {}, 0 + for label, sample_doc in parsed["samples"].items(): + if label in in_set: + sample_ids[label] = in_set[label]["_id"] + continue + doc = client.samples.create(dict(sample_doc, owner=owner)) + client.samples.move_to_set(doc["_id"], None, set_id) + sample_ids[label] = doc["_id"] + created += 1 + print(f"sample set {set_id} ({wafer_label}, ordered{', created' if created_set else ''}): {len(sample_ids)} samples, {created} created") + # the run's specimen is not stored on the set: each measurement names its sample, and the sample names the set + measurement_set, created_measurement_set = ensure_set(client.measurements, parsed["measurement_set"], owner["_id"]) + existing = {m["name"]: m for m in find(client.measurements, {"inSet._id": measurement_set["_id"], "isEntitySet": {"$ne": True}}, owner["_id"], 500)} + measurement_ids, measurements_created = {}, 0 + for label, measurement_doc in parsed["measurements"].items(): + if measurement_doc["name"] in existing: + measurement_ids[label] = existing[measurement_doc["name"]]["_id"] + continue + body = {k: v for k, v in measurement_doc.items() if k != "_records"} + body["_sample"] = {"_id": sample_ids[label], "cls": "Sample"} + if files == "none": # no file store: keep the raw records in the measurement's metadata + body["metadata"] = dict(body["metadata"], records=measurement_doc["_records"]) + doc = client.measurements.create(dict(body, owner=owner)) + client.measurements.move_to_set(doc["_id"], None, measurement_set["_id"]) + measurement_ids[label] = doc["_id"] + measurements_created += 1 + print(f"measurement set {measurement_set['_id']} ({run_name}, ordered{', created' if created_measurement_set else ''}): " + f"{len(measurement_ids)} measurements, {measurements_created} created") + if files != "none": + # One request per file (~1-2 s each), so: the record JSONs by default, the loop arrays and plots only with + # --files all, and eight uploads in flight at a time. + jobs = [(f"measurements/{measurement_ids[label]}/{name}", payload) + for label, file_list in parsed["files"].items() for name, payload in file_list + if files == "all" or not name.startswith("loops/")] + with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool: + for done, _ in enumerate(pool.map(lambda job: put_file(thread_client(client), *job, owner["_id"]), jobs), 1): + if done % 200 == 0: + print(f" files: {done}/{len(jobs)}", flush=True) + print(f"files: {len(jobs)} put under measurements// ({'records/*.json, loops/*' if files == 'all' else 'records/*.json; --files all adds loops/*'})") + posted = 0 + for label, unit_id, prop, repetition in parsed["properties"]: # properties/create is not idempotent: skip what is there + present = find(client.properties, {"source.info.origin._id": measurement_ids[label], "data.name": prop["name"], + "repetition": repetition}, owner["_id"], 1) + if present: + continue + client.properties.create(dict(holder(prop, measurement_ids[label], sample_ids[label], unit_id, repetition), owner=owner)) + posted += 1 + print(f"properties: {posted} hysteresis loops posted, {len(parsed['properties']) - posted} already present (one per measured sample)") + + +def main(): + """Command line: parse, validate, upload.""" + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("run_dir") + ap.add_argument("command", nargs="?", choices=["synthesis", "measurement", "both"], default="both", + help="synthesis: NLR record + photo onto the set · measurement: UTK run onto the set · both (default)") + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--emit-example", help="write the first combined loop property (most loops) to this path — the ESSE example") + ap.add_argument("--host", default=os.environ.get("MAT3RA_HOST", "localhost:3000"), + help="web app host or URL (or MAT3RA_HOST); https unless localhost, e.g. dev.mat3ra.com") + ap.add_argument("--account", help="slug of the account the data belongs to (reads scoped to it, writes owned by it)") + ap.add_argument("--files", choices=["records", "all", "none"], default="records", + help="which files to upload per measurement: the record JSONs (default), also the loop arrays and plots (all), or none") + ap.add_argument("--limit-records", type=int, help="trial: only the first N records and the samples they belong to") + ap.add_argument("--deposition", help="NLR HTEM record (json) attached to the wafer set's metadata") + ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") + a = ap.parse_args() + p = parse(a.run_dir, a.limit_records, a.deposition, a.instrument) + nfiles = sum(len(v) for v in p["files"].values()) + print(f"specimen {p['wafer']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " + f"(ordered set, one per sample) · {len(p['records'])} records -> {nfiles} files · {len(p['images'])} image(s) · " + f"{len(p['properties'])} samples with a combined loop" + (f" · no curves: {len(p['skipped'])} samples" if p["skipped"] else "")) + for label, _, prop, _rep in p["properties"]: + n = prop["parameters"]["off"].get("imprint", {}).get("count") + print(f" {label}: {n} loops combined, imprint off = {prop['parameters']['off'].get('imprint', {}).get('value')} V") + if a.emit_example and p["properties"]: + label, _, prop, _rep = max(p["properties"], key=lambda t: t[2]["parameters"]["off"].get("imprint", {}).get("count", 0)) + Path(a.emit_example).write_text(json.dumps(prop, indent=4) + "\n"); print(f"example written from sample {label} -> {a.emit_example}") + errors = validate(p) + print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") + if errors or a.dry_run: + sys.exit(1 if errors else 0) + url = urllib.parse.urlsplit(base_url(a.host)) + address = {"host": url.hostname, "port": url.port or (443 if url.scheme == "https" else 80), "secure": url.scheme == "https"} + client = APIClient.authenticate(**address) + if a.account: + client = APIClient.authenticate(account_id=account_id(client, a.account), **address) + upload(client, p, command=a.command, files=a.files) + + +if __name__ == "__main__": + main() diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 342e10e73..00782e70b 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -6,9 +6,8 @@ "source": [ "# Overview\n", "\n", - "A run folder is one scanning probe microscopy run as the instrument exports it: `summary.json` with the recipe, the session and one record per measured point, and `loops/` with the raw curves.\n", - "This example creates the specimen on the platform as a Sample Set holding one Sample per measured position, the run as a Measurement Set holding one Measurement per Sample with the Setup it was measured on, the run's records as files on each Measurement, and one hysteresis-loop Property per Sample.\n", - "Re-running it adds only what is missing." + "This example uploads one scanning probe microscopy run to the platform: the run folder becomes a Sample Set with one Sample per measured position, a Measurement Set with one Measurement per Sample and its Setup, the run's records as files, and one hysteresis-loop Property per Sample.\n", + "A run folder is what the instrument exports — `summary.json` with the recipe, the session and one record per measured point, and `loops/` with the raw curves — and re-running the notebook adds only what is missing." ] }, { @@ -17,7 +16,7 @@ "source": [ "## Install the API client\n", "\n", - "The samples, measurements and files endpoints are not released yet, so the client is installed from its branch until it merges." + "The samples, measurements and files endpoints are not released yet, so the client is installed from its branch until it merges. Restart the kernel after this cell." ] }, { @@ -33,13 +32,12 @@ "cell_type": "markdown", "metadata": {}, "source": [ - "## Authenticate and initialize API client\n", - "\n", - "### Authenticate\n", - "Authenticate in the browser (OIDC device flow) or via JupyterLite host injection. Credentials are stored in environment variables.\n", + "## Set Parameters\n", "\n", - "### Initialize API client\n", - "Create an authenticated API client and resolve the owner account ID." + "- **HOST**: platform the run is uploaded to\n", + "- **RUN_DIR**: the run folder beside this notebook — what the instrument exports, with `summary.json` and `loops/` inside it\n", + "- **ACCOUNT_SLUG**: account the data belongs to, empty for the default account\n", + "- **FILES**: which files to upload per measurement" ] }, { @@ -48,20 +46,32 @@ "metadata": {}, "outputs": [], "source": [ - "from mat3ra.notebooks_utils.packages import install_packages\n", + "import urllib.parse\n", "\n", - "await install_packages(\"api\")" + "HOST = \"https://platform.mat3ra.com\"\n", + "RUN_DIR = \"run\"\n", + "ACCOUNT_SLUG = \"\"\n", + "FILES = \"records\" # \"records\": the record JSONs, \"all\": also the loop arrays and plots, \"none\": no files\n", + "\n", + "url = urllib.parse.urlsplit(HOST)\n", + "address = {\n", + " \"host\": url.hostname,\n", + " \"port\": url.port or (443 if url.scheme == \"https\" else 80),\n", + " \"secure\": url.scheme == \"https\",\n", + "}" ] }, { - "cell_type": "code", - "execution_count": null, + "cell_type": "markdown", "metadata": {}, - "outputs": [], "source": [ - "from mat3ra.notebooks_utils.auth import authenticate\n", + "## Authenticate and initialize API client\n", "\n", - "await authenticate()" + "### Authenticate\n", + "Authenticate in the browser (OIDC device flow) or via JupyterLite host injection. Credentials are stored in environment variables.\n", + "\n", + "### Initialize API client\n", + "Create an authenticated API client and resolve the owner account ID." ] }, { @@ -70,25 +80,20 @@ "metadata": {}, "outputs": [], "source": [ - "import os\n", - "\n", - "from mat3ra.api_client import APIClient\n", + "from mat3ra.notebooks_utils.packages import install_packages\n", "\n", - "client = APIClient.authenticate()\n", - "selected_account = client.my_account\n", - "OWNER_ID = os.getenv(\"ORGANIZATION_ID\") or selected_account.id" + "await install_packages(\"api\")" ] }, { - "cell_type": "markdown", + "cell_type": "code", + "execution_count": null, "metadata": {}, + "outputs": [], "source": [ - "## Set Parameters\n", + "from mat3ra.notebooks_utils.auth import authenticate\n", "\n", - "- **HOST**: platform the run is uploaded to\n", - "- **RUN_DIR**: path to the run folder, relative to this notebook\n", - "- **ACCOUNT_SLUG**: account the data belongs to, empty for the default account\n", - "- **FILES**: which files to upload per measurement" + "await authenticate()" ] }, { @@ -97,19 +102,16 @@ "metadata": {}, "outputs": [], "source": [ - "HOST = os.environ.get(\"MAT3RA_HOST\", \"https://platform.mat3ra.com\")\n", - "RUN_DIR = \"20260901_171601_alscn_01448\"\n", - "ACCOUNT_SLUG = \"\"\n", - "FILES = \"records\" # \"records\": the record JSONs, \"all\": also the loop arrays and plots, \"none\": no files" + "from mat3ra.api_client import APIClient\n", + "\n", + "client = APIClient.authenticate(**address)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ - "## Fetch the uploader\n", - "\n", - "Download `upload_run.py`, the parser and uploader the platform serves, into the working directory." + "# Imports" ] }, { @@ -120,11 +122,6 @@ "source": [ "from pathlib import Path\n", "\n", - "from mat3ra.notebooks_utils.io import read_from_url\n", - "\n", - "if not Path(\"upload_run.py\").exists():\n", - " Path(\"upload_run.py\").write_text(await read_from_url(f\"{HOST}/upload_run.py\"))\n", - "\n", "from upload_run import account_id, parse, upload" ] }, @@ -169,7 +166,7 @@ "outputs": [], "source": [ "if ACCOUNT_SLUG:\n", - " client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG))" + " client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG), **address)" ] }, { @@ -187,14 +184,14 @@ "metadata": {}, "outputs": [], "source": [ - "upload(client, parsed, command=\"both\", files=FILES)" + "upload(client, parsed, files=FILES)" ] }, { "cell_type": "markdown", "metadata": {}, "source": [ - "## Print the link to the run\n", + "## Find the run in the web app\n", "\n", "The run is a folder in the account's Measurements tab, named after the run." ] @@ -205,8 +202,7 @@ "metadata": {}, "outputs": [], "source": [ - "account = next(item for item in client.list_accounts() if item[\"_id\"] == client.my_account.id)\n", - "print(f\"{HOST}/{account['slug']}/measurements\")" + "print(f\"Open {HOST}, your account's Measurements tab: {parsed['run']}\")" ] }, { @@ -215,8 +211,7 @@ "source": [ "## References\n", "\n", - "- [Mat3ra REST API](https://docs.mat3ra.com/rest-api/overview/)\n", - "- [Samples and measurements in the web app](https://github.com/mat3ra/web-app/pull/2976)" + "- [Mat3ra REST API](https://docs.mat3ra.com/rest-api/overview/)" ] } ], diff --git a/mkdocs.yml b/mkdocs.yml index f0f5fe884..da8e0acaa 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -58,6 +58,7 @@ nav: - Get File from Job: examples/job/get-file-from-job.ipynb - Run Simulations and Extract Properties: examples/job/run-simulations-and-extract-properties.ipynb - ML - Train Model Predict Properties: examples/job/ml-train-model-predict-properties.ipynb + - Upload an SPM Run: examples/measurement/upload_spm_run.ipynb plugins: - same-dir @@ -84,5 +85,6 @@ plugins: execute_ignore: - examples/system/get_authentication_params.ipynb - examples/job/run-simulations-and-extract-properties.ipynb + - examples/measurement/upload_spm_run.ipynb ignore: - "other/**/*.ipynb" From fc34ff1335f2ea41234c30c47f21619aa79f5eb3 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 16:14:04 -0700 Subject: [PATCH 03/36] =?UTF-8?q?fix(SOF-8051):=20uploader=20follows=20the?= =?UTF-8?q?=20placement=20model=20=E2=80=94=20a=20sample=20set=20per=20run?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5 --- examples/measurement/upload_run.py | 48 ++++++++---------------------- 1 file changed, 13 insertions(+), 35 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index be8fa5c5a..8b131015b 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -271,9 +271,10 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") records = all_records[:limit_records] if limit_records else all_records wid = wafer_id(recipe) run_name = session.get("name") or run_dir.name - # the specimen's physical ID is the case-sticker text; it is how NLR's and UTK's data find the same set - sample_set = {"name": wid, "entitySetType": "ordered", - "metadata": {"label": wid, "recipe": recipe["name"], "context": recipe.get("context", "")}} + reg = registration(recipe) + sample_set = {"name": run_name, "entitySetType": "ordered", "wafer": {"physicalId": wid}, + "origin": {"coordinates": [recipe["sites"][0]["x_stage_m"], recipe["sites"][0]["y_stage_m"]], "units": "m"}, + "metadata": {"recipe": recipe["name"], "context": recipe.get("context", ""), "registration": reg}} # NLR's HTEM deposition record(s) for this wafer, verbatim — the specimen's synthesis data. UTK drops the file into # the run folder as deposition*.json; --deposition overrides that. deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) @@ -304,7 +305,6 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") for label, const in per_sample.items(): if label in samples: samples[label]["metadata"] = const - reg = registration(recipe) workflow = build_workflow(recipe, list(samples)) unit_id = workflow["subworkflows"][0]["units"][0]["flowchartId"] measurement_set = {"name": run_name, "entitySetType": "ordered", @@ -315,10 +315,8 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") started = session.get("started_ts") setup_block = {"name": instrument, "session": {k: v for k, v in {"name": session.get("name"), - "started": datetime.fromtimestamp(started, timezone.utc).isoformat().replace("+00:00", "Z") if started else None, - "directory": session.get("instrument_directory")}.items() if v}, - **({"settings": common["instrument_params"]} if common.get("instrument_params") else {}), - "registration": reg} + "started": datetime.fromtimestamp(started, timezone.utc).isoformat().replace("+00:00", "Z") if started else None}.items() if v}, + **({"settings": common["instrument_params"]} if common.get("instrument_params") else {})} by_sample, slim_by_sample = {}, {} for r, slim in zip(records, slim_records): lab = r["labels"]["site_label"] @@ -329,7 +327,8 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") recs = by_sample.get(label, []) measurements[label] = {"name": f"{run_name} {label}", "_sample": None, "workflow": workflow, "setup": setup_block, "status": "finished", - "metadata": {"run_dir": session.get("run_dir", str(run_dir)), "recordsCount": len(recs)}} + "metadata": {"run_dir": session.get("run_dir", str(run_dir)), "recordsCount": len(recs), + "registration": reg, "instrumentDirectory": session.get("instrument_directory")}} slim_by_label = slim_by_sample.get(label, []) measurements[label]["_records"] = slim_by_label # not sent; upload() decides files vs metadata files[label] = sample_files(label, recs, run_dir, slim_by_sample.get(label, [])) @@ -441,46 +440,25 @@ def ensure_set(endpoint, doc, owner_id): return (found[0], False) if found else (endpoint.create_set(dict(doc, owner={"_id": owner_id})), True) -def find_set_by_label(endpoint, label, owner_id): - """The specimen is found by its physical ID (metadata.label); older sets by name.""" - found = find(endpoint, {"isEntitySet": True, "metadata.label": label}, owner_id, 5) or \ - find(endpoint, {"isEntitySet": True, "name": label}, owner_id, 5) - return found[0] if found else None - - def upload(client, parsed, command="both", files="records"): - """Two separate imports joined by the Sample Set's physical label: + """Two imports onto the run's own Sample Set — one placement of the wafer on one instrument: `synthesis` — NLR's record(s) + the photograph onto the set (created if missing; no samples, no measurements); - `measurement` — UTK's run: the set (created bare if NLR has not arrived), its samples, the measurement set tied to - it, one measurement per sample, files, properties; + `measurement` — UTK's run: the set, its samples, the measurement set tied to it, one measurement per sample, files, properties; `both` — synthesis if the folder has a deposition record, then measurement. - Idempotent: sets by label, members by name/label; files re-put; properties posted only when missing.""" + Idempotent: sets by run name, members by name/label; files re-put; properties posted only when missing.""" wafer_label, run_name = parsed["wafer"], parsed["run"] owner = {"_id": client.my_account.id} has_deposition = "deposition" in parsed["sample_set"]["metadata"] - sample_set = find_set_by_label(client.samples, wafer_label, owner["_id"]) - created_set = False - if sample_set is None: - doc = dict(parsed["sample_set"], owner=owner) - if command == "measurement": # UTK arrived first: a bare set, the synthesis record comes later - doc["metadata"] = {k: v for k, v in doc["metadata"].items() if k != "deposition"} - sample_set = client.samples.create_set(doc) - created_set = True + sample_set, created_set = ensure_set(client.samples, parsed["sample_set"], owner["_id"]) set_id = sample_set["_id"] - if command in ("synthesis", "both") and has_deposition and not created_set: - # the set was here first (UTK arrived before NLR): attach the record now - patch = {k: v for k, v in parsed["sample_set"]["metadata"].items() if k in ("label", "deposition")} - client.samples.update_set(set_id, {"metadata": patch}) if command in ("synthesis", "both") and has_deposition: - print(f"sample set {set_id} ({wafer_label}{', created' if created_set else ', updated'}): synthesis record attached") + print(f"sample set {set_id} ({wafer_label}{', created' if created_set else ''}): synthesis record attached") if command in ("synthesis", "both") and files != "none": for image in parsed["images"]: put_file(client, f"sets/{set_id}/{image.name}", image, owner["_id"]) print(f" image {image.name} -> sets/{set_id}/") if command == "synthesis": return - if not created_set and not (sample_set.get("metadata") or {}).get("label"): - client.samples.update_set(set_id, {"metadata": {"label": wafer_label}}) in_set = {s.get("label"): s for s in find(client.samples, {"inSet._id": set_id, "isEntitySet": {"$ne": True}}, owner["_id"], 500)} sample_ids, created = {}, 0 for label, sample_doc in parsed["samples"].items(): From c1e6e319dac40f19d3306515dd52976b3497c686 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 16:40:51 -0700 Subject: [PATCH 04/36] fix(SOF-8051): uploader review round; the notebook says wafer Co-Authored-By: Claude Opus 5 --- examples/measurement/upload_run.py | 33 ++++++++++++----------- examples/measurement/upload_spm_run.ipynb | 2 +- 2 files changed, 19 insertions(+), 16 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 8b131015b..a3ea6fff0 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -13,9 +13,9 @@ ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. Optional: `pip install mat3ra-esse` (tested with 2026.8.27-0) turns on schema validation before anything is uploaded. -Canonical copy: mat3ra/web-app `src/application/public/upload_run.py` (served by the web app at /upload_run.py, so the -instructions in the app always match). The planning repo keeps a working copy and the tests; scripts/publish.sh -copies web-app → plan by default and plan → web-app with --push. +Canonical copy: mat3ra/api-examples `examples/measurement/upload_run.py`, beside the notebook that imports it. The +planning repo keeps a working copy and the tests; scripts/publish.sh copies api-examples → plan by default and +plan → api-examples with --push. """ import argparse, ast, concurrent.futures, json, math, os, re, statistics, struct, sys, threading, time, urllib.parse, uuid from datetime import datetime, timezone @@ -84,7 +84,7 @@ def load_run(run_dir): def wafer_id(recipe): - """The specimen's physical ID as written on its case, taken from the recipe name (e.g. PDAC_COM5_01448).""" + """The wafer's physical ID as written on its case, taken from the recipe name (e.g. PDAC_COM5_01448).""" m = WAFER_ID.search(recipe.get("context", "")) return m.group(1) if m else recipe["name"] @@ -239,12 +239,17 @@ def factor_records(records): slim = [_unflatten({k: v for k, v in row.items() if k not in common_keys and k not in sample_keys}) for row in flat] return common, per_sample, slim +def starting_site(recipe): + """The site the run started from: r0c00 when the recipe has it, else the first one listed.""" + return next((s for s in recipe["sites"] if s["label"] == "r0c00"), recipe["sites"][0]) + + def registration(recipe): - """The instrument's frame as UTK stated it: one anchor in words (from recipe.context) and where r0c00 sits on the stage. - Recorded, not interpreted — a second anchor is needed for a rigid map (PROPOSAL §3.5).""" + """The instrument's frame as UTK stated it: one anchor in words (from recipe.context) and where the run started + on the stage. Recorded, not interpreted.""" ctx = recipe.get("context", "") m = re.search(r"starting point is (.+?)(?:\.|$)", ctx) - r0 = next((s for s in recipe["sites"] if s["label"] == "r0c00"), recipe["sites"][0]) + r0 = starting_site(recipe) return {"frame": "asylum-spm stage", "units": "m", "anchor": m.group(1).strip() if m else ctx, f"{r0['label']}_stage_m": [r0["x_stage_m"], r0["y_stage_m"]]} @@ -271,11 +276,11 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") records = all_records[:limit_records] if limit_records else all_records wid = wafer_id(recipe) run_name = session.get("name") or run_dir.name - reg = registration(recipe) + reg, start = registration(recipe), starting_site(recipe) sample_set = {"name": run_name, "entitySetType": "ordered", "wafer": {"physicalId": wid}, - "origin": {"coordinates": [recipe["sites"][0]["x_stage_m"], recipe["sites"][0]["y_stage_m"]], "units": "m"}, + "origin": {"coordinates": [start["x_stage_m"], start["y_stage_m"]], "units": "m"}, "metadata": {"recipe": recipe["name"], "context": recipe.get("context", ""), "registration": reg}} - # NLR's HTEM deposition record(s) for this wafer, verbatim — the specimen's synthesis data. UTK drops the file into + # NLR's HTEM deposition record(s) for this wafer, verbatim. UTK drops the file into # the run folder as deposition*.json; --deposition overrides that. deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) if deposition_files: @@ -284,7 +289,7 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") d = json.loads(f.read_text()) deposition_records.extend(d if isinstance(d, list) else [d]) sample_set["metadata"]["deposition"] = deposition_records - # the specimen photograph: any image at the run-folder root + # the wafer photograph: any image at the run-folder root images = [f for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) samples = {s["label"]: {"name": f"{wid} {s['label']}", "label": s["label"], @@ -327,8 +332,7 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") recs = by_sample.get(label, []) measurements[label] = {"name": f"{run_name} {label}", "_sample": None, "workflow": workflow, "setup": setup_block, "status": "finished", - "metadata": {"run_dir": session.get("run_dir", str(run_dir)), "recordsCount": len(recs), - "registration": reg, "instrumentDirectory": session.get("instrument_directory")}} + "metadata": {"run_dir": session.get("run_dir", str(run_dir)), "recordsCount": len(recs)}} slim_by_label = slim_by_sample.get(label, []) measurements[label]["_records"] = slim_by_label # not sent; upload() decides files vs metadata files[label] = sample_files(label, recs, run_dir, slim_by_sample.get(label, [])) @@ -470,7 +474,6 @@ def upload(client, parsed, command="both", files="records"): sample_ids[label] = doc["_id"] created += 1 print(f"sample set {set_id} ({wafer_label}, ordered{', created' if created_set else ''}): {len(sample_ids)} samples, {created} created") - # the run's specimen is not stored on the set: each measurement names its sample, and the sample names the set measurement_set, created_measurement_set = ensure_set(client.measurements, parsed["measurement_set"], owner["_id"]) existing = {m["name"]: m for m in find(client.measurements, {"inSet._id": measurement_set["_id"], "isEntitySet": {"$ne": True}}, owner["_id"], 500)} measurement_ids, measurements_created = {}, 0 @@ -529,7 +532,7 @@ def main(): a = ap.parse_args() p = parse(a.run_dir, a.limit_records, a.deposition, a.instrument) nfiles = sum(len(v) for v in p["files"].values()) - print(f"specimen {p['wafer']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " + print(f"wafer {p['wafer']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " f"(ordered set, one per sample) · {len(p['records'])} records -> {nfiles} files · {len(p['images'])} image(s) · " f"{len(p['properties'])} samples with a combined loop" + (f" · no curves: {len(p['skipped'])} samples" if p["skipped"] else "")) for label, _, prop, _rep in p["properties"]: diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 00782e70b..9d86df20c 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -143,7 +143,7 @@ "parsed = parse(Path(RUN_DIR))\n", "file_count = sum(len(files) for files in parsed[\"files\"].values())\n", "print(\n", - " f\"specimen {parsed['wafer']}: {len(parsed['samples'])} samples (ordered set) · run {parsed['run']}: \"\n", + " f\"wafer {parsed['wafer']}: {len(parsed['samples'])} samples (ordered set) · run {parsed['run']}: \"\n", " f\"{len(parsed['measurements'])} measurements (ordered set, one per sample) · {len(parsed['records'])} records \"\n", " f\"-> {file_count} files · {len(parsed['images'])} image(s) · {len(parsed['properties'])} samples with a combined \"\n", " f\"loop · no curves: {len(parsed['skipped'])} samples\"\n", From bc4b50dd1e4a725ba50eb77a7478e8981069c59b Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 18:24:28 -0700 Subject: [PATCH 05/36] fix(SOF-8051): --emit-example thins the curves Co-Authored-By: Claude Opus 5 --- examples/measurement/upload_run.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index a3ea6fff0..7903c820b 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -244,6 +244,16 @@ def starting_site(recipe): return next((s for s in recipe["sites"] if s["label"] == "r0c00"), recipe["sites"][0]) +def thinned_curves(prop, points=12): + """The curves at `points` evenly spaced samples: an ESSE example shows the shape, not the data.""" + x = prop["xDataArray"] + if len(x) <= points: + return {} + keep = [round(i * (len(x) - 1) / (points - 1)) for i in range(points)] + return {"xDataArray": [x[i] for i in keep], + "yDataSeries": [[s[i] for i in keep] for s in prop["yDataSeries"]]} + + def registration(recipe): """The instrument's frame as UTK stated it: one anchor in words (from recipe.context) and where the run started on the stage. Recorded, not interpreted.""" @@ -540,6 +550,7 @@ def main(): print(f" {label}: {n} loops combined, imprint off = {prop['parameters']['off'].get('imprint', {}).get('value')} V") if a.emit_example and p["properties"]: label, _, prop, _rep = max(p["properties"], key=lambda t: t[2]["parameters"]["off"].get("imprint", {}).get("count", 0)) + prop = dict(prop, **thinned_curves(prop)) Path(a.emit_example).write_text(json.dumps(prop, indent=4) + "\n"); print(f"example written from sample {label} -> {a.emit_example}") errors = validate(p) print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") From 0477bdbc68a9404e109dc00021a83b4584afef76 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 22:42:01 -0700 Subject: [PATCH 06/36] fix(SOF-8051): the uploader takes a physical id and leaves sets bare MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A Sample Set is a plain folder now — no wafer, no origin. Every Sample carries the --physical-id / PHYSICAL_ID the human gives and the frame its position was read in, in its metadata. Co-Authored-By: Claude Fable 5.1 --- examples/measurement/upload_run.py | 22 +++++++++++----------- examples/measurement/upload_spm_run.ipynb | 4 +++- 2 files changed, 14 insertions(+), 12 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 7903c820b..d03b1a563 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -6,8 +6,8 @@ the property is the pad's hysteresis loop — the eight loops combined — with the loop parameters (mean, population standard deviation, count over the loops) inside it. Individual loops stay in the measurement's metadata. - upload_run.py --account # upload everything - upload_run.py --dry-run [--emit-example out.json] # parse + validate only; write one property as the ESSE example + upload_run.py --physical-id --account # upload everything + upload_run.py --physical-id --dry-run [--emit-example out.json] # parse + validate only; write one property as the ESSE example Requires Python 3.9+ and `pip install mat3ra-api-client`, which talks to the platform and takes OIDC_ACCESS_TOKEN, or ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. Optional: @@ -279,17 +279,15 @@ def sample_files(label, records, run_dir, slim_by_index): return out -def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm"): +def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument="asylum-afm"): """The whole run folder as platform documents: sample set, samples, measurement set, one measurement per sample, files, one loop property per fully measured sample.""" run_dir = Path(run_dir) recipe, session, all_records = load_run(run_dir) records = all_records[:limit_records] if limit_records else all_records wid = wafer_id(recipe) run_name = session.get("name") or run_dir.name - reg, start = registration(recipe), starting_site(recipe) - sample_set = {"name": run_name, "entitySetType": "ordered", "wafer": {"physicalId": wid}, - "origin": {"coordinates": [start["x_stage_m"], start["y_stage_m"]], "units": "m"}, - "metadata": {"recipe": recipe["name"], "context": recipe.get("context", ""), "registration": reg}} + reg = registration(recipe) + sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} # NLR's HTEM deposition record(s) for this wafer, verbatim. UTK drops the file into # the run folder as deposition*.json; --deposition overrides that. deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) @@ -302,8 +300,9 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") # the wafer photograph: any image at the run-folder root images = [f for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) - samples = {s["label"]: {"name": f"{wid} {s['label']}", "label": s["label"], - "position": {"coordinates": [s["x_stage_m"], s["y_stage_m"]], "units": "m"}, "metadata": {}} + samples = {s["label"]: {"name": f"{wid} {s['label']}", "label": s["label"], "physicalId": physical_id, + "position": {"coordinates": [s["x_stage_m"], s["y_stage_m"]], "units": "m"}, + "metadata": {"registration": reg}} for s in recipe["sites"]} if limit_records: # a trial run must be a prefix of a full one: only samples whose records ALL made the cut, so no sample is @@ -319,7 +318,7 @@ def parse(run_dir, limit_records=None, deposition=None, instrument="asylum-afm") common, per_sample, slim_records = factor_records(records) for label, const in per_sample.items(): if label in samples: - samples[label]["metadata"] = const + samples[label]["metadata"].update(const) workflow = build_workflow(recipe, list(samples)) unit_id = workflow["subworkflows"][0]["units"][0]["flowchartId"] measurement_set = {"name": run_name, "entitySetType": "ordered", @@ -527,6 +526,7 @@ def main(): """Command line: parse, validate, upload.""" ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("run_dir") + ap.add_argument("--physical-id", required=True, help="the identifier written on the physical piece the samples are part of, e.g. PDAC_COM5_01448") ap.add_argument("command", nargs="?", choices=["synthesis", "measurement", "both"], default="both", help="synthesis: NLR record + photo onto the set · measurement: UTK run onto the set · both (default)") ap.add_argument("--dry-run", action="store_true") @@ -540,7 +540,7 @@ def main(): ap.add_argument("--deposition", help="NLR HTEM record (json) attached to the wafer set's metadata") ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") a = ap.parse_args() - p = parse(a.run_dir, a.limit_records, a.deposition, a.instrument) + p = parse(a.run_dir, a.physical_id, a.limit_records, a.deposition, a.instrument) nfiles = sum(len(v) for v in p["files"].values()) print(f"wafer {p['wafer']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " f"(ordered set, one per sample) · {len(p['records'])} records -> {nfiles} files · {len(p['images'])} image(s) · " diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 9d86df20c..da46b1def 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -36,6 +36,7 @@ "\n", "- **HOST**: platform the run is uploaded to\n", "- **RUN_DIR**: the run folder beside this notebook — what the instrument exports, with `summary.json` and `loops/` inside it\n", + "- **PHYSICAL_ID**: the identifier written on the physical piece the measured positions are part of — every Sample carries it\n", "- **ACCOUNT_SLUG**: account the data belongs to, empty for the default account\n", "- **FILES**: which files to upload per measurement" ] @@ -50,6 +51,7 @@ "\n", "HOST = \"https://platform.mat3ra.com\"\n", "RUN_DIR = \"run\"\n", + "PHYSICAL_ID = \"PDAC_COM5_01448\"\n", "ACCOUNT_SLUG = \"\"\n", "FILES = \"records\" # \"records\": the record JSONs, \"all\": also the loop arrays and plots, \"none\": no files\n", "\n", @@ -140,7 +142,7 @@ "metadata": {}, "outputs": [], "source": [ - "parsed = parse(Path(RUN_DIR))\n", + "parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n", "file_count = sum(len(files) for files in parsed[\"files\"].values())\n", "print(\n", " f\"wafer {parsed['wafer']}: {len(parsed['samples'])} samples (ordered set) · run {parsed['run']}: \"\n", From 2fa6494707872c4e3d0f373fd43c7bd3065c465a Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 22:43:38 -0700 Subject: [PATCH 07/36] fix(SOF-8051): sample names from the given physical id Co-Authored-By: Claude Fable 5.1 --- examples/measurement/upload_run.py | 9 +-------- 1 file changed, 1 insertion(+), 8 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index d03b1a563..9b9afe999 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -30,7 +30,6 @@ except ImportError: ESSE = SampleSchema = None -WAFER_ID = re.compile(r"(PDAC_COM\d+_\d+)") FIELD = {"off_field": "off", "on_field": "on"} # loop_params key -> parameters path (units: voltages in xAxis.units, responses in yAxis.units) PARAMETERS = { @@ -83,12 +82,6 @@ def load_run(run_dir): return recipe, session, records -def wafer_id(recipe): - """The wafer's physical ID as written on its case, taken from the recipe name (e.g. PDAC_COM5_01448).""" - m = WAFER_ID.search(recipe.get("context", "")) - return m.group(1) if m else recipe["name"] - - def response_curve(loops_dir, field, phase_offset_deg, bias): """UTK's X': rotate the lock-in quadratures by the loop's PCA angle so the switching lands in x', sign pinned so x' rises with bias (extract_loop_params). None if the arrays are not all present.""" @@ -284,7 +277,7 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= run_dir = Path(run_dir) recipe, session, all_records = load_run(run_dir) records = all_records[:limit_records] if limit_records else all_records - wid = wafer_id(recipe) + wid = physical_id run_name = session.get("name") or run_dir.name reg = registration(recipe) sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} From 44b459bd0f0713d458645326ed94d890f0c276da Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 23:16:07 -0700 Subject: [PATCH 08/36] =?UTF-8?q?fix(SOF-8051):=20PHYSICAL=5FID=20has=20no?= =?UTF-8?q?=20default=20=E2=80=94=20the=20human=20names=20the=20piece?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Fable 5.1 --- examples/measurement/upload_run.py | 25 ++++++++++++----------- examples/measurement/upload_spm_run.ipynb | 4 ++-- 2 files changed, 15 insertions(+), 14 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 9b9afe999..98e77708a 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -277,11 +277,12 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= run_dir = Path(run_dir) recipe, session, all_records = load_run(run_dir) records = all_records[:limit_records] if limit_records else all_records - wid = physical_id + if not physical_id: + raise ValueError("--physical-id: the identifier written on the physical piece is required") run_name = session.get("name") or run_dir.name reg = registration(recipe) sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} - # NLR's HTEM deposition record(s) for this wafer, verbatim. UTK drops the file into + # NLR's HTEM deposition record(s) for the piece, verbatim. UTK drops the file into # the run folder as deposition*.json; --deposition overrides that. deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) if deposition_files: @@ -290,10 +291,10 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= d = json.loads(f.read_text()) deposition_records.extend(d if isinstance(d, list) else [d]) sample_set["metadata"]["deposition"] = deposition_records - # the wafer photograph: any image at the run-folder root + # the photograph of the piece: any image at the run-folder root images = [f for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) - samples = {s["label"]: {"name": f"{wid} {s['label']}", "label": s["label"], "physicalId": physical_id, + samples = {s["label"]: {"name": f"{physical_id} {s['label']}", "label": s["label"], "physicalId": physical_id, "position": {"coordinates": [s["x_stage_m"], s["y_stage_m"]], "units": "m"}, "metadata": {"registration": reg}} for s in recipe["sites"]} @@ -340,7 +341,7 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= files[label] = sample_files(label, recs, run_dir, slim_by_sample.get(label, [])) prop = combine_pad(label, recs, run_dir) if recs else None (properties.append((label, unit_id, prop, 0)) if prop else skipped.append(label)) - return {"wafer": wid, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, + return {"physicalId": physical_id, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, "measurement_set": measurement_set, "measurements": measurements, "files": files, "records": records, "properties": properties, "skipped": skipped} @@ -447,18 +448,18 @@ def ensure_set(endpoint, doc, owner_id): def upload(client, parsed, command="both", files="records"): - """Two imports onto the run's own Sample Set — one placement of the wafer on one instrument: + """Two imports onto the run's own Sample Set: `synthesis` — NLR's record(s) + the photograph onto the set (created if missing; no samples, no measurements); `measurement` — UTK's run: the set, its samples, the measurement set tied to it, one measurement per sample, files, properties; `both` — synthesis if the folder has a deposition record, then measurement. Idempotent: sets by run name, members by name/label; files re-put; properties posted only when missing.""" - wafer_label, run_name = parsed["wafer"], parsed["run"] + physical_id, run_name = parsed["physicalId"], parsed["run"] owner = {"_id": client.my_account.id} has_deposition = "deposition" in parsed["sample_set"]["metadata"] sample_set, created_set = ensure_set(client.samples, parsed["sample_set"], owner["_id"]) set_id = sample_set["_id"] if command in ("synthesis", "both") and has_deposition: - print(f"sample set {set_id} ({wafer_label}{', created' if created_set else ''}): synthesis record attached") + print(f"sample set {set_id} ({run_name}{', created' if created_set else ''}): synthesis record attached") if command in ("synthesis", "both") and files != "none": for image in parsed["images"]: put_file(client, f"sets/{set_id}/{image.name}", image, owner["_id"]) @@ -475,7 +476,7 @@ def upload(client, parsed, command="both", files="records"): client.samples.move_to_set(doc["_id"], None, set_id) sample_ids[label] = doc["_id"] created += 1 - print(f"sample set {set_id} ({wafer_label}, ordered{', created' if created_set else ''}): {len(sample_ids)} samples, {created} created") + print(f"sample set {set_id} ({run_name}, ordered{', created' if created_set else ''}): {len(sample_ids)} samples, {created} created") measurement_set, created_measurement_set = ensure_set(client.measurements, parsed["measurement_set"], owner["_id"]) existing = {m["name"]: m for m in find(client.measurements, {"inSet._id": measurement_set["_id"], "isEntitySet": {"$ne": True}}, owner["_id"], 500)} measurement_ids, measurements_created = {}, 0 @@ -530,12 +531,12 @@ def main(): ap.add_argument("--files", choices=["records", "all", "none"], default="records", help="which files to upload per measurement: the record JSONs (default), also the loop arrays and plots (all), or none") ap.add_argument("--limit-records", type=int, help="trial: only the first N records and the samples they belong to") - ap.add_argument("--deposition", help="NLR HTEM record (json) attached to the wafer set's metadata") + ap.add_argument("--deposition", help="NLR HTEM record (json) kept in the run's sample set metadata") ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") - a = ap.parse_args() + a = ap.parse_intermixed_args() p = parse(a.run_dir, a.physical_id, a.limit_records, a.deposition, a.instrument) nfiles = sum(len(v) for v in p["files"].values()) - print(f"wafer {p['wafer']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " + print(f"{p['physicalId']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " f"(ordered set, one per sample) · {len(p['records'])} records -> {nfiles} files · {len(p['images'])} image(s) · " f"{len(p['properties'])} samples with a combined loop" + (f" · no curves: {len(p['skipped'])} samples" if p["skipped"] else "")) for label, _, prop, _rep in p["properties"]: diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index da46b1def..3e6094f6d 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -51,7 +51,7 @@ "\n", "HOST = \"https://platform.mat3ra.com\"\n", "RUN_DIR = \"run\"\n", - "PHYSICAL_ID = \"PDAC_COM5_01448\"\n", + "PHYSICAL_ID = \"\"\n", "ACCOUNT_SLUG = \"\"\n", "FILES = \"records\" # \"records\": the record JSONs, \"all\": also the loop arrays and plots, \"none\": no files\n", "\n", @@ -145,7 +145,7 @@ "parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n", "file_count = sum(len(files) for files in parsed[\"files\"].values())\n", "print(\n", - " f\"wafer {parsed['wafer']}: {len(parsed['samples'])} samples (ordered set) · run {parsed['run']}: \"\n", + " f\"{parsed['physicalId']}: {len(parsed['samples'])} samples (ordered set) · run {parsed['run']}: \"\n", " f\"{len(parsed['measurements'])} measurements (ordered set, one per sample) · {len(parsed['records'])} records \"\n", " f\"-> {file_count} files · {len(parsed['images'])} image(s) · {len(parsed['properties'])} samples with a combined \"\n", " f\"loop · no curves: {len(parsed['skipped'])} samples\"\n", From 9600f4077dfcc9fc32adfbb4e9c2a6d15458ba02 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 23:40:03 -0700 Subject: [PATCH 09/36] feat(SOF-8051): notebook that uploads NLR's data for one piece MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit upload_nlr_data.ipynb mirrors upload_spm_run.ipynb: the same install and authenticate cells, the folder NLR delivers instead of a run folder, and the two machines it was measured on as parameters. It parses into one Sample Set of the measured pads and two runs over them — the XRF map and the DC I-V sweep — and uploads them one after the other. upload_run.py gains parse_nlr(), which returns those two runs in the shape parse() returns, so upload() is unchanged apart from one line for a file that belongs to a run rather than to one measurement. Co-Authored-By: Claude Fable 5.1 --- examples/measurement/upload_nlr_data.ipynb | 249 +++++++++++++++++++++ examples/measurement/upload_run.py | 142 ++++++++++-- mkdocs.yml | 2 + 3 files changed, 373 insertions(+), 20 deletions(-) create mode 100644 examples/measurement/upload_nlr_data.ipynb diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb new file mode 100644 index 000000000..781b6e665 --- /dev/null +++ b/examples/measurement/upload_nlr_data.ipynb @@ -0,0 +1,249 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Overview\n", + "\n", + "This example uploads the data NLR delivers for one physical piece: the folder becomes a Sample Set with one Sample per measured pad, and two runs over those same pads — an XRF map and a DC I-V sweep — each a Measurement Set with one Measurement per Sample and its Setup, the delivered tables as files, and one Property per measured quantity.\n", + "A delivered folder holds the XRF grid table, the DC I-V tables and photographs of the piece, and re-running the notebook adds only what is missing." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Install the API client\n", + "\n", + "The samples, measurements and files endpoints are not released yet, so the client is installed from its branch until it merges. Restart the kernel after this cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "%pip install -q \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Set Parameters\n", + "\n", + "- **HOST**: platform the data is uploaded to\n", + "- **DATA_DIR**: the delivered folder beside this notebook — the XRF grid table, the DC I-V tables and the photographs\n", + "- **PHYSICAL_ID**: the identifier written on the physical piece the measured pads are part of — every Sample carries it\n", + "- **XRF_INSTRUMENT**: the machine the XRF map was measured on\n", + "- **IV_INSTRUMENT**: the machine the DC I-V sweep was measured on\n", + "- **ACCOUNT_SLUG**: account the data belongs to, empty for the default account\n", + "- **FILES**: which files to upload" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import urllib.parse\n", + "\n", + "HOST = \"https://platform.mat3ra.com\"\n", + "DATA_DIR = \"nlr\"\n", + "PHYSICAL_ID = \"\"\n", + "XRF_INSTRUMENT = \"\"\n", + "IV_INSTRUMENT = \"\"\n", + "ACCOUNT_SLUG = \"\"\n", + "FILES = \"records\" # \"records\": the delivered tables and photographs, \"none\": no files\n", + "\n", + "url = urllib.parse.urlsplit(HOST)\n", + "address = {\n", + " \"host\": url.hostname,\n", + " \"port\": url.port or (443 if url.scheme == \"https\" else 80),\n", + " \"secure\": url.scheme == \"https\",\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Authenticate and initialize API client\n", + "\n", + "### Authenticate\n", + "Authenticate in the browser (OIDC device flow) or via JupyterLite host injection. Credentials are stored in environment variables.\n", + "\n", + "### Initialize API client\n", + "Create an authenticated API client and resolve the owner account ID." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from mat3ra.notebooks_utils.packages import install_packages\n", + "\n", + "await install_packages(\"api\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from mat3ra.notebooks_utils.auth import authenticate\n", + "\n", + "await authenticate()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from mat3ra.api_client import APIClient\n", + "\n", + "client = APIClient.authenticate(**address)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Imports" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from pathlib import Path\n", + "\n", + "from upload_run import account_id, parse_nlr, upload" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Parse the delivered folder\n", + "\n", + "Read the folder into the documents the platform stores: one Sample Set, and one run per technique over its Samples. Nothing is uploaded yet." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "runs = parse_nlr(Path(DATA_DIR), PHYSICAL_ID, XRF_INSTRUMENT, IV_INSTRUMENT)\n", + "for run in runs:\n", + " print(\n", + " f\"{run['physicalId']}: {len(run['samples'])} samples (ordered set) · run {run['run']}: \"\n", + " f\"{len(run['measurements'])} measurements (ordered set, one per sample) · {len(run['set_files'])} files · \"\n", + " f\"{len(run['images'])} image(s) · {len(run['properties'])} properties\"\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Select the account\n", + "\n", + "`ACCOUNT_SLUG` re-authenticates the client against that account, so the run is read and written there." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "if ACCOUNT_SLUG:\n", + " client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG), **address)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Upload the data\n", + "\n", + "Create the Sample Set and its Samples, then a Measurement Set per technique with one Measurement per Sample, the delivered files and the Properties." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "for run in runs:\n", + " upload(client, run, files=FILES)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Find the data in the web app\n", + "\n", + "Each technique is a folder in the account's Measurements tab." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print(f\"Open {HOST}, your account's Measurements tab: {', '.join(run['run'] for run in runs)}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## References\n", + "\n", + "- [Mat3ra REST API](https://docs.mat3ra.com/rest-api/overview/)" + ] + } + ], + "metadata": { + "colab": { + "name": "upload_spm_run.ipynb", + "provenance": [] + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.6" + } + }, + "nbformat": 4, + "nbformat_minor": 1 +} diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 98e77708a..bbec162da 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -8,6 +8,7 @@ upload_run.py --physical-id --account # upload everything upload_run.py --physical-id --dry-run [--emit-example out.json] # parse + validate only; write one property as the ESSE example + upload_run.py --physical-id --nlr # NLR's XRF grid and DC I-V sweep instead of a UTK run Requires Python 3.9+ and `pip install mat3ra-api-client`, which talks to the platform and takes OIDC_ACCESS_TOKEN, or ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. Optional: @@ -342,10 +343,99 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= prop = combine_pad(label, recs, run_dir) if recs else None (properties.append((label, unit_id, prop, 0)) if prop else skipped.append(label)) return {"physicalId": physical_id, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, - "measurement_set": measurement_set, "measurements": measurements, "files": files, + "measurement_set": measurement_set, "measurements": measurements, "files": files, "set_files": [], "records": records, "properties": properties, "skipped": skipped} +NLR_FRAME = {"frame": "wafer", "units": "mm", "note": "x_mm, y_mm as delivered by NLR; corner and axes to be confirmed"} +XRF_APPLICATION = {"name": "xrf-mapper", "shortName": "xrf", "summary": "X-ray fluorescence mapper (film thickness and composition over a grid of positions)", + "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} +IV_APPLICATION = {"name": "probe-station", "shortName": "iv", "summary": "DC probe station (current through a pad over a bias sweep)", + "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} + + +def read_columns(path): + """Every line of a tab-separated file after its header, split into its cells.""" + return [line.split("\t") for line in Path(path).read_text().splitlines()[1:] if line.strip()] + + +def build_nlr_workflow(application, executable_name, flavor_name, name, properties): + """The procedure one of NLR's instruments runs, in the shape build_workflow gives UTK's: ONE workflow, one + subworkflow, one execution unit declaring what it produces, ids stable across uploads (uuid5 of the names). + Application, executable and flavor name the standata registry entries these two instruments still need.""" + results = [{"name": property_name} for property_name in properties] + monitors = [{"name": "standard_output"}] + executable = {"name": executable_name, "applicationName": application["name"], "applicationVersion": "*", "isDefault": True, + "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} + flavor = {"name": flavor_name, "executableName": executable_name, "applicationName": application["name"], "applicationVersion": "*", + "isDefault": True, "input": [], "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} + unit = {"type": "execution", "name": executable_name, "head": True, "status": "finished", + "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, f"{application['name']}/{executable_name}").hex[:24], + "application": application, "executable": executable, "flavor": flavor, "input": [], "context": [], + "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} + model = {"type": "unknown", "subtype": "unknown", "method": {"type": "unknown", "subtype": "unknown"}} + subworkflow_id = uuid.uuid5(WORKFLOW_NAMESPACE, f"{application['name']}/{flavor_name}").hex[:17] + subworkflow = {"_id": subworkflow_id, "name": flavor_name, "application": application, "model": model, + "properties": properties, "units": [unit]} + subworkflow_unit = {"_id": subworkflow_id, "type": "subworkflow", "name": flavor_name, "head": True, "status": "finished", + "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, f"{application['name']}/{flavor_name}/unit").hex[:24], + "preProcessors": [], "postProcessors": [], "monitors": [], "results": []} + return {"name": name, "isDefault": False, "tags": ["experimental"], "properties": properties, "application": application, + "subworkflows": [subworkflow], "units": [subworkflow_unit], "workflows": []} + + +def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): + """NLR's delivery for one piece as platform documents: one Sample Set of the pads they measured, and one run per + technique over those same pads — the XRF map, then the DC I-V sweep. Two runs, each in the shape parse() returns, + so upload() takes them one after the other: the first creates the Sample Set, the second finds it by name.""" + folder = Path(folder) + grid_file = sorted(folder.rglob("*xrf_grid.txt"))[0] + volts_file, amps_file = sorted(folder.rglob("IV_Volts.txt"))[0], sorted(folder.rglob("IV_Amps.txt"))[0] + run_name = grid_file.stem + images = [f for f in sorted(folder.rglob("*")) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] + sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} + xrf_run_name = f"{run_name} XRF" + xrf_workflow = build_nlr_workflow(XRF_APPLICATION, "map", "xrf_grid", "XRF Grid Map", + ["thickness", "al_atomic_fraction", "sc_atomic_fraction"]) + xrf_unit_id = xrf_workflow["subworkflows"][0]["units"][0]["flowchartId"] + grid = read_columns(grid_file) + samples, xrf_measurements, xrf_properties = {}, {}, [] + for row, column, x_mm, y_mm, thickness_um, aluminium_at_pct, scandium_at_pct in grid: + label = f"r{int(row)}c{int(column)}" + samples[label] = {"name": f"{physical_id} {label}", "label": label, "physicalId": physical_id, + "position": {"coordinates": [float(x_mm), float(y_mm)], "units": "mm"}, + "metadata": {"frame": NLR_FRAME, "row": int(row), "column": int(column)}} + xrf_measurements[label] = {"name": f"{xrf_run_name} {label}", "_sample": None, "workflow": xrf_workflow, + "setup": {"name": xrf_instrument}, "status": "finished", "_records": [], + "metadata": {"row": int(row), "column": int(column), "thickness_um": float(thickness_um), + "al_at_pct": float(aluminium_at_pct), "sc_at_pct": float(scandium_at_pct)}} + xrf_properties += [(label, xrf_unit_id, {"name": "thickness", "value": float(thickness_um), "units": "um"}, 0), + (label, xrf_unit_id, {"name": "al_atomic_fraction", "value": float(aluminium_at_pct), "units": "at%"}, 0), + (label, xrf_unit_id, {"name": "sc_atomic_fraction", "value": float(scandium_at_pct), "units": "at%"}, 0)] + iv_run_name = f"{run_name} DC IV" + iv_workflow = build_nlr_workflow(IV_APPLICATION, "sweep", "dc_iv", "DC I-V Sweep", ["iv_curve"]) + iv_unit_id = iv_workflow["subworkflows"][0]["units"][0]["flowchartId"] + volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] + amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] + # the sweep NLR ran, read off the voltages themselves; every row of the file holds the same one + iv_setup = {"name": iv_instrument, "settings": {"v_min": min(volts[0]), "v_max": max(volts[0]), "points": len(volts[0])}} + iv_measurements, iv_properties = {}, [] + for index, (label, bias, current) in enumerate(zip(samples, volts, amps)): + iv_measurements[label] = {"name": f"{iv_run_name} {label}", "_sample": None, "workflow": iv_workflow, + "setup": iv_setup, "status": "finished", "_records": [], "metadata": {"row_index": index}} + iv_properties.append((label, iv_unit_id, {"name": "iv_curve", "xAxis": {"label": "bias", "units": "V"}, + "yAxis": {"label": "current", "units": "A"}, + "xDataArray": bias, "yDataSeries": [current]}, 0)) + return [{"physicalId": physical_id, "run": xrf_run_name, "sample_set": sample_set, "images": images, "samples": samples, + "measurement_set": {"name": xrf_run_name, "entitySetType": "ordered", "metadata": {}}, + "measurements": xrf_measurements, "files": {}, "set_files": [(grid_file.name, grid_file)], + "records": grid, "properties": xrf_properties}, + {"physicalId": physical_id, "run": iv_run_name, "sample_set": sample_set, "images": images, "samples": samples, + "measurement_set": {"name": iv_run_name, "entitySetType": "ordered", "metadata": {}}, + "measurements": iv_measurements, "files": {}, "set_files": [(volts_file.name, volts_file), (amps_file.name, amps_file)], + "records": volts, "properties": iv_properties}] + + def holder(prop, measurement_id, sample_id, unit_id, repetition): """The property holder the platform stores: the data, where it came from (measurement, sample, workflow unit) and a repetition index — 0, since a measurement holds one sample and one loop property.""" @@ -476,7 +566,7 @@ def upload(client, parsed, command="both", files="records"): client.samples.move_to_set(doc["_id"], None, set_id) sample_ids[label] = doc["_id"] created += 1 - print(f"sample set {set_id} ({run_name}, ordered{', created' if created_set else ''}): {len(sample_ids)} samples, {created} created") + print(f"sample set {set_id} ({sample_set['name']}, ordered{', created' if created_set else ''}): {len(sample_ids)} samples, {created} created") measurement_set, created_measurement_set = ensure_set(client.measurements, parsed["measurement_set"], owner["_id"]) existing = {m["name"]: m for m in find(client.measurements, {"inSet._id": measurement_set["_id"], "isEntitySet": {"$ne": True}}, owner["_id"], 500)} measurement_ids, measurements_created = {}, 0 @@ -497,9 +587,10 @@ def upload(client, parsed, command="both", files="records"): if files != "none": # One request per file (~1-2 s each), so: the record JSONs by default, the loop arrays and plots only with # --files all, and eight uploads in flight at a time. - jobs = [(f"measurements/{measurement_ids[label]}/{name}", payload) - for label, file_list in parsed["files"].items() for name, payload in file_list - if files == "all" or not name.startswith("loops/")] + jobs = [(f"measurements/{measurement_set['_id']}/{name}", payload) for name, payload in parsed["set_files"]] + jobs += [(f"measurements/{measurement_ids[label]}/{name}", payload) + for label, file_list in parsed["files"].items() for name, payload in file_list + if files == "all" or not name.startswith("loops/")] with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool: for done, _ in enumerate(pool.map(lambda job: put_file(thread_client(client), *job, owner["_id"]), jobs), 1): if done % 200 == 0: @@ -513,7 +604,7 @@ def upload(client, parsed, command="both", files="records"): continue client.properties.create(dict(holder(prop, measurement_ids[label], sample_ids[label], unit_id, repetition), owner=owner)) posted += 1 - print(f"properties: {posted} hysteresis loops posted, {len(parsed['properties']) - posted} already present (one per measured sample)") + print(f"properties: {posted} posted, {len(parsed['properties']) - posted} already present") def main(): @@ -533,20 +624,30 @@ def main(): ap.add_argument("--limit-records", type=int, help="trial: only the first N records and the samples they belong to") ap.add_argument("--deposition", help="NLR HTEM record (json) kept in the run's sample set metadata") ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") + ap.add_argument("--nlr", nargs=2, metavar=("XRF_INSTRUMENT", "IV_INSTRUMENT"), + help="the folder holds NLR's delivery — an XRF grid and a DC I-V sweep over the same pads, measured on these two machines — not a UTK run") a = ap.parse_intermixed_args() - p = parse(a.run_dir, a.physical_id, a.limit_records, a.deposition, a.instrument) - nfiles = sum(len(v) for v in p["files"].values()) - print(f"{p['physicalId']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " - f"(ordered set, one per sample) · {len(p['records'])} records -> {nfiles} files · {len(p['images'])} image(s) · " - f"{len(p['properties'])} samples with a combined loop" + (f" · no curves: {len(p['skipped'])} samples" if p["skipped"] else "")) - for label, _, prop, _rep in p["properties"]: - n = prop["parameters"]["off"].get("imprint", {}).get("count") - print(f" {label}: {n} loops combined, imprint off = {prop['parameters']['off'].get('imprint', {}).get('value')} V") - if a.emit_example and p["properties"]: - label, _, prop, _rep = max(p["properties"], key=lambda t: t[2]["parameters"]["off"].get("imprint", {}).get("count", 0)) - prop = dict(prop, **thinned_curves(prop)) - Path(a.emit_example).write_text(json.dumps(prop, indent=4) + "\n"); print(f"example written from sample {label} -> {a.emit_example}") - errors = validate(p) + if a.nlr: + runs = parse_nlr(a.run_dir, a.physical_id, *a.nlr) + for p in runs: + print(f"{p['physicalId']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} " + f"measurements (ordered set, one per sample) · {len(p['records'])} rows -> {len(p['set_files'])} files · " + f"{len(p['images'])} image(s) · {len(p['properties'])} properties") + else: + runs = [parse(a.run_dir, a.physical_id, a.limit_records, a.deposition, a.instrument)] + p = runs[0] + nfiles = sum(len(v) for v in p["files"].values()) + print(f"{p['physicalId']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " + f"(ordered set, one per sample) · {len(p['records'])} records -> {nfiles} files · {len(p['images'])} image(s) · " + f"{len(p['properties'])} samples with a combined loop" + (f" · no curves: {len(p['skipped'])} samples" if p["skipped"] else "")) + for label, _, prop, _rep in p["properties"]: + n = prop["parameters"]["off"].get("imprint", {}).get("count") + print(f" {label}: {n} loops combined, imprint off = {prop['parameters']['off'].get('imprint', {}).get('value')} V") + if a.emit_example and p["properties"]: + label, _, prop, _rep = max(p["properties"], key=lambda t: t[2]["parameters"]["off"].get("imprint", {}).get("count", 0)) + prop = dict(prop, **thinned_curves(prop)) + Path(a.emit_example).write_text(json.dumps(prop, indent=4) + "\n"); print(f"example written from sample {label} -> {a.emit_example}") + errors = sum(validate(p) for p in runs) print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") if errors or a.dry_run: sys.exit(1 if errors else 0) @@ -555,7 +656,8 @@ def main(): client = APIClient.authenticate(**address) if a.account: client = APIClient.authenticate(account_id=account_id(client, a.account), **address) - upload(client, p, command=a.command, files=a.files) + for p in runs: + upload(client, p, command=a.command, files=a.files) if __name__ == "__main__": diff --git a/mkdocs.yml b/mkdocs.yml index da8e0acaa..38158feb8 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -59,6 +59,7 @@ nav: - Run Simulations and Extract Properties: examples/job/run-simulations-and-extract-properties.ipynb - ML - Train Model Predict Properties: examples/job/ml-train-model-predict-properties.ipynb - Upload an SPM Run: examples/measurement/upload_spm_run.ipynb + - Upload NLR Data: examples/measurement/upload_nlr_data.ipynb plugins: - same-dir @@ -86,5 +87,6 @@ plugins: - examples/system/get_authentication_params.ipynb - examples/job/run-simulations-and-extract-properties.ipynb - examples/measurement/upload_spm_run.ipynb + - examples/measurement/upload_nlr_data.ipynb ignore: - "other/**/*.ipynb" From a0922823a39a28f9786ed81d5f00d10014056f6b Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 21 Sep 2026 23:58:33 -0700 Subject: [PATCH 10/36] fix(SOF-8051): the notebooks point at alphafilm.mat3ra.com, where the labs upload Co-Authored-By: Claude Fable 5.1 --- examples/measurement/upload_nlr_data.ipynb | 2 +- examples/measurement/upload_spm_run.ipynb | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index 781b6e665..4f94e2cee 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -51,7 +51,7 @@ "source": [ "import urllib.parse\n", "\n", - "HOST = \"https://platform.mat3ra.com\"\n", + "HOST = \"https://alphafilm.mat3ra.com\"\n", "DATA_DIR = \"nlr\"\n", "PHYSICAL_ID = \"\"\n", "XRF_INSTRUMENT = \"\"\n", diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 3e6094f6d..227a6e175 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -49,7 +49,7 @@ "source": [ "import urllib.parse\n", "\n", - "HOST = \"https://platform.mat3ra.com\"\n", + "HOST = \"https://alphafilm.mat3ra.com\"\n", "RUN_DIR = \"run\"\n", "PHYSICAL_ID = \"\"\n", "ACCOUNT_SLUG = \"\"\n", From b314ad1d5b2a28ad8a06560f2976b62bc7c604f6 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 00:00:06 -0700 Subject: [PATCH 11/36] fix(SOF-8051): the install cell brings mat3ra-notebooks-utils, which the authenticate cells import Co-Authored-By: Claude Fable 5.1 --- examples/measurement/upload_nlr_data.ipynb | 2 +- examples/measurement/upload_spm_run.ipynb | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index 4f94e2cee..5c311ee06 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -25,7 +25,7 @@ "metadata": {}, "outputs": [], "source": [ - "%pip install -q \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" + "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" ] }, { diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 227a6e175..b4e9c8a44 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -25,7 +25,7 @@ "metadata": {}, "outputs": [], "source": [ - "%pip install -q \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" + "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" ] }, { From e603a4851567d50292b817f9970206f75bc5e5d0 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 10:13:45 -0700 Subject: [PATCH 12/36] fix(SOF-8051): the NLR path validates, and its inputs are checked before use Review (coderabbit) on #374, four findings, all real: - Every property was validated against the hysteresis-loop schema, so the NLR run reported 176 invalid documents and main exited before uploading anything. Each property now picks the schema named after it; thickness, the atomic fractions and the I-V curve have none in ESSE yet, so they are reported as unvalidated rather than failed against the wrong one. - A loop was accepted onto the average when its bias array merely had the same length. Same length is not the same voltages, and the result assigns responses to the wrong bias. Compared point by point now, within tolerance. - zip(samples, volts, amps) silently dropped pads when a file had fewer rows, and allowed a bias row and a current row of different lengths. Both are checked before any document is built. - The NLR notebook's Colab metadata named the SPM notebook. Dry runs after: NLR validation OK (was 176 invalid), UTK unchanged at OK. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_nlr_data.ipynb | 2 +- examples/measurement/upload_run.py | 28 ++++++++++++++++++++-- 2 files changed, 27 insertions(+), 3 deletions(-) diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index 5c311ee06..f04098152 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -223,7 +223,7 @@ ], "metadata": { "colab": { - "name": "upload_spm_run.ipynb", + "name": "upload_nlr_data.ipynb", "provenance": [] }, "kernelspec": { diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index bbec162da..e307fd71e 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -126,7 +126,9 @@ def combine_pad(label, records, run_dir): b = load_npy(bias_p) bias = bias or b curve = response_curve(loops_dir, field, lp["phase_offset_deg"], b) - if curve is not None and len(curve) == len(bias): + # averaged point by point, so a loop measured on a different bias axis is dropped rather + # than folded in: same length is not the same voltages + if curve is not None and same_axis(b, bias): series[field].append(curve) if bias is None or not series["on"] or not series["off"]: return None @@ -417,6 +419,13 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): iv_unit_id = iv_workflow["subworkflows"][0]["units"][0]["flowchartId"] volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] + # zip would silently drop pads, so the shapes are checked before any document is built + if not (len(samples) == len(volts) == len(amps)): + raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)}/{len(amps)} rows for " + f"{len(samples)} pads — every pad needs one row in each file") + for row, (bias_row, current_row) in enumerate(zip(volts, amps)): + if len(bias_row) != len(current_row): + raise SystemExit(f"row {row}: {len(bias_row)} bias points but {len(current_row)} current points") # the sweep NLR ran, read off the voltages themselves; every row of the file holds the same one iv_setup = {"name": iv_instrument, "settings": {"v_min": min(volts[0]), "v_max": max(volts[0]), "points": len(volts[0])}} iv_measurements, iv_properties = {}, [] @@ -467,15 +476,30 @@ def validate(parsed): esse.validate(m, schemas["measurement"]) except Exception as e: errors += 1; print("MEASUREMENT INVALID", label, str(e)[:300]); break + unvalidated = set() for label, uid, prop, rep in parsed["properties"]: + # by the property's own name: NLR's thickness, atomic fractions and I-V curve are not the + # hysteresis loop, and ESSE has no schema for them yet, so they are reported, not failed + schema_id = f"properties-directory/non-scalar/{prop['name'].replace('_', '-')}" + schema = schemas.get(schema_id) or schemas.get(schema_id.replace("non-scalar", "scalar")) + if schema is None: + unvalidated.add(prop["name"]); continue try: - esse.validate(prop, schemas["properties-directory/non-scalar/hysteresis-loop"]) + esse.validate(prop, schema) esse.validate(holder(prop, "dryrun", "dryrun", uid, rep), schemas["property/holder"]) except Exception as e: errors += 1; print("PROPERTY INVALID", label, str(e)[:300]) + if unvalidated: + print("no ESSE schema yet, not validated:", ", ".join(sorted(unvalidated))) return errors +def same_axis(candidate, reference, tolerance=1e-9): + """Whether two bias axes are the same sweep: equal length and equal voltages within tolerance.""" + return len(candidate) == len(reference) and all( + abs(a - b) <= tolerance + 1e-6 * abs(b) for a, b in zip(candidate, reference)) + + def base_url(host): """`https://` unless told otherwise: a bare hostname becomes https, localhost/127.0.0.1 http, a URL is kept.""" host = host.rstrip("/") From 0b17199ef2413b85d885fc9327d224787023465c Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 10:15:15 -0700 Subject: [PATCH 13/36] fix(SOF-8051): an existing sample set takes metadata it does not have yet Review (coderabbit) on #374: when ensure_set found an existing set, the parsed metadata was never persisted, so a synthesis run after a measurement run printed "synthesis record attached" while the deposition stayed absent. The existing set now takes the keys it lacks, through the branch client's update_set, and keeps the ones it already has. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_run.py | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index e307fd71e..47c9ca746 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -556,9 +556,20 @@ def put_file(client, name, payload, owner_id): def ensure_set(endpoint, doc, owner_id): - """The set with this name in the account, created when missing; returns (set, created).""" + """The set with this name in the account, created when missing; returns (set, created). + An existing set takes any metadata it does not have yet - a synthesis run after a measurement + run has the deposition record to add, and the set was created without it.""" found = find(endpoint, {"isEntitySet": True, "name": doc["name"]}, owner_id, 5) - return (found[0], False) if found else (endpoint.create_set(dict(doc, owner={"_id": owner_id})), True) + if not found: + return endpoint.create_set(dict(doc, owner={"_id": owner_id})), True + + existing = found[0] + incoming = doc.get("metadata") or {} + missing = {k: v for k, v in incoming.items() if k not in (existing.get("metadata") or {})} + if missing: + endpoint.update_set(existing["_id"], {"metadata": missing}) + existing = dict(existing, metadata={**(existing.get("metadata") or {}), **missing}) + return existing, False def upload(client, parsed, command="both", files="records"): From 588d8b8939d6fa8a8dcae68fc1fc19ada584514a Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 10:51:18 -0700 Subject: [PATCH 14/36] fix(SOF-8051): a duplicate pad is an error, and a set takes records inside a key it has Review (coderabbit) on #374, both on the previous round's fixes: - Two grid rows for one pad overwrote each other in `samples`, and the row-count check then compared against the collapsed dictionary, so 44 IV rows passed for a 45-row grid. A repeated pad now stops the run, and the counts compare against the grid itself. - ensure_set only added metadata keys the set lacked, so a second synthesis run whose deposition list held new records was dropped: `deposition` was already there. merge_metadata now merges, growing a list by the entries it does not hold, recursing into nested objects, and the set is written only when the result differs from what it has. Dry runs after: NLR OK, UTK OK. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_run.py | 33 +++++++++++++++++++++++------- 1 file changed, 26 insertions(+), 7 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 47c9ca746..e6bf367ab 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -404,6 +404,10 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): samples, xrf_measurements, xrf_properties = {}, {}, [] for row, column, x_mm, y_mm, thickness_um, aluminium_at_pct, scandium_at_pct in grid: label = f"r{int(row)}c{int(column)}" + # two rows for one pad would overwrite each other here and leave the row counts below + # agreeing against a dictionary that has already lost an entry + if label in samples: + raise SystemExit(f"{grid_file.name}: pad {label} appears twice") samples[label] = {"name": f"{physical_id} {label}", "label": label, "physicalId": physical_id, "position": {"coordinates": [float(x_mm), float(y_mm)], "units": "mm"}, "metadata": {"frame": NLR_FRAME, "row": int(row), "column": int(column)}} @@ -420,9 +424,9 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] # zip would silently drop pads, so the shapes are checked before any document is built - if not (len(samples) == len(volts) == len(amps)): + if not (len(grid) == len(volts) == len(amps)): raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)}/{len(amps)} rows for " - f"{len(samples)} pads — every pad needs one row in each file") + f"{len(grid)} pads in {grid_file.name} — every pad needs one row in each file") for row, (bias_row, current_row) in enumerate(zip(volts, amps)): if len(bias_row) != len(current_row): raise SystemExit(f"row {row}: {len(bias_row)} bias points but {len(current_row)} current points") @@ -555,6 +559,22 @@ def put_file(client, name, payload, owner_id): time.sleep(2 * (attempt + 1)) +def merge_metadata(existing, incoming): + """`incoming` on top of `existing`, keeping what neither replaces. A list grows by the entries it + does not already hold - a second synthesis run brings deposition records the set has never seen, + and taking only absent keys would drop them because `deposition` is already there.""" + merged = dict(existing) + for key, value in incoming.items(): + held = merged.get(key) + if isinstance(held, list) and isinstance(value, list): + merged[key] = held + [v for v in value if v not in held] + elif isinstance(held, dict) and isinstance(value, dict): + merged[key] = merge_metadata(held, value) + else: + merged[key] = value + return merged + + def ensure_set(endpoint, doc, owner_id): """The set with this name in the account, created when missing; returns (set, created). An existing set takes any metadata it does not have yet - a synthesis run after a measurement @@ -564,11 +584,10 @@ def ensure_set(endpoint, doc, owner_id): return endpoint.create_set(dict(doc, owner={"_id": owner_id})), True existing = found[0] - incoming = doc.get("metadata") or {} - missing = {k: v for k, v in incoming.items() if k not in (existing.get("metadata") or {})} - if missing: - endpoint.update_set(existing["_id"], {"metadata": missing}) - existing = dict(existing, metadata={**(existing.get("metadata") or {}), **missing}) + merged = merge_metadata(existing.get("metadata") or {}, doc.get("metadata") or {}) + if merged != (existing.get("metadata") or {}): + endpoint.update_set(existing["_id"], {"metadata": merged}) + existing = dict(existing, metadata=merged) return existing, False From 31e7b6af8e772e1df52addb65aea87192843bb5e Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 11:40:47 -0700 Subject: [PATCH 15/36] refactor(SOF-8051): one upload path, and `files` is the list of what to upload `upload` had three modes - synthesis, measurement, both - for an import that is only ever done one way: the set, its samples, the measurement set, one measurement per sample, the files, the properties. The modes are gone, along with the deposition special-casing that went with them; a later upload still brings metadata the set lacks, through ensure_set. `files` was one of three words meaning a mode. It is now the list of file groups to upload, "records" and "loops", which is how the notebook already passes it. An empty list uploads nothing and keeps the raw records in each measurement's metadata. An unknown group is rejected by name. run_files builds the (name, payload) pairs and destination says where each one lands, so the two questions - what to send and where it goes - are separate and readable. Dry runs after: NLR OK, UTK OK. File counts with the UTK run, 16 records: records 16, loops 240, both 256. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_run.py | 82 +++++++++++++++++------------- 1 file changed, 48 insertions(+), 34 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index e6bf367ab..eb2ed9d7d 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -6,7 +6,7 @@ the property is the pad's hysteresis loop — the eight loops combined — with the loop parameters (mean, population standard deviation, count over the loops) inside it. Individual loops stay in the measurement's metadata. - upload_run.py --physical-id --account # upload everything + upload_run.py --physical-id --account # upload the run upload_run.py --physical-id --dry-run [--emit-example out.json] # parse + validate only; write one property as the ESSE example upload_run.py --physical-id --nlr # NLR's XRF grid and DC I-V sweep instead of a UTK run @@ -561,7 +561,7 @@ def put_file(client, name, payload, owner_id): def merge_metadata(existing, incoming): """`incoming` on top of `existing`, keeping what neither replaces. A list grows by the entries it - does not already hold - a second synthesis run brings deposition records the set has never seen, + does not already hold - a re-upload brings deposition records the set has never seen, and taking only absent keys would drop them because `deposition` is already there.""" merged = dict(existing) for key, value in incoming.items(): @@ -577,8 +577,8 @@ def merge_metadata(existing, incoming): def ensure_set(endpoint, doc, owner_id): """The set with this name in the account, created when missing; returns (set, created). - An existing set takes any metadata it does not have yet - a synthesis run after a measurement - run has the deposition record to add, and the set was created without it.""" + An existing set takes any metadata it does not have yet - a later upload may carry a + deposition record the set was created without.""" found = find(endpoint, {"isEntitySet": True, "name": doc["name"]}, owner_id, 5) if not found: return endpoint.create_set(dict(doc, owner={"_id": owner_id})), True @@ -591,25 +591,44 @@ def ensure_set(endpoint, doc, owner_id): return existing, False -def upload(client, parsed, command="both", files="records"): - """Two imports onto the run's own Sample Set: - `synthesis` — NLR's record(s) + the photograph onto the set (created if missing; no samples, no measurements); - `measurement` — UTK's run: the set, its samples, the measurement set tied to it, one measurement per sample, files, properties; - `both` — synthesis if the folder has a deposition record, then measurement. - Idempotent: sets by run name, members by name/label; files re-put; properties posted only when missing.""" - physical_id, run_name = parsed["physicalId"], parsed["run"] +FILE_GROUPS = ("records", "loops") + + +def run_files(parsed, groups=("records",)): + """Every file this run puts, as (name, payload) pairs. `groups` names what to include per + measurement: "records" for the record JSONs, "loops" for the loop arrays and plots, which are + large. The set's own files - NLR's grid, the photographs of the piece - are few and always go. + `name` carries the label it belongs to so `destination` can address it.""" + unknown = set(groups) - set(FILE_GROUPS) + if unknown: + raise SystemExit(f"unknown file group(s): {', '.join(sorted(unknown))}; choose from {', '.join(FILE_GROUPS)}") + files = [(f"set/{name}", payload) for name, payload in parsed["set_files"]] + files += [(f"set/{image.name}", image) for image in parsed["images"]] + for label, file_list in parsed["files"].items(): + files += [(f"{label}/{name}", payload) for name, payload in file_list + if ("loops" if name.startswith("loops/") else "records") in groups] + return files + + +def destination(name, set_id, measurement_ids): + """Where one of `run_files`' entries is put: the set's own, or the measurement it belongs to.""" + label, _, rest = name.partition("/") + if label == "set": + return f"sets/{set_id}/{rest}" + return f"measurements/{measurement_ids[label]}/{rest}" + + +def upload(client, parsed, files=("records",)): + """The run onto its Sample Set: the set, its samples, the measurement set, one measurement per + sample, the files and the properties. `files` names which groups to upload - see FILE_GROUPS; + an empty list uploads none and keeps the raw records in each measurement's metadata instead. + Idempotent: sets by run name, members by name/label; files re-put; properties posted only when + missing.""" + uploads = run_files(parsed, files) if files else [] + run_name = parsed["run"] owner = {"_id": client.my_account.id} - has_deposition = "deposition" in parsed["sample_set"]["metadata"] sample_set, created_set = ensure_set(client.samples, parsed["sample_set"], owner["_id"]) set_id = sample_set["_id"] - if command in ("synthesis", "both") and has_deposition: - print(f"sample set {set_id} ({run_name}{', created' if created_set else ''}): synthesis record attached") - if command in ("synthesis", "both") and files != "none": - for image in parsed["images"]: - put_file(client, f"sets/{set_id}/{image.name}", image, owner["_id"]) - print(f" image {image.name} -> sets/{set_id}/") - if command == "synthesis": - return in_set = {s.get("label"): s for s in find(client.samples, {"inSet._id": set_id, "isEntitySet": {"$ne": True}}, owner["_id"], 500)} sample_ids, created = {}, 0 for label, sample_doc in parsed["samples"].items(): @@ -630,7 +649,7 @@ def upload(client, parsed, command="both", files="records"): continue body = {k: v for k, v in measurement_doc.items() if k != "_records"} body["_sample"] = {"_id": sample_ids[label], "cls": "Sample"} - if files == "none": # no file store: keep the raw records in the measurement's metadata + if not uploads: # no file store: keep the raw records in the measurement's metadata body["metadata"] = dict(body["metadata"], records=measurement_doc["_records"]) doc = client.measurements.create(dict(body, owner=owner)) client.measurements.move_to_set(doc["_id"], None, measurement_set["_id"]) @@ -638,18 +657,14 @@ def upload(client, parsed, command="both", files="records"): measurements_created += 1 print(f"measurement set {measurement_set['_id']} ({run_name}, ordered{', created' if created_measurement_set else ''}): " f"{len(measurement_ids)} measurements, {measurements_created} created") - if files != "none": - # One request per file (~1-2 s each), so: the record JSONs by default, the loop arrays and plots only with - # --files all, and eight uploads in flight at a time. - jobs = [(f"measurements/{measurement_set['_id']}/{name}", payload) for name, payload in parsed["set_files"]] - jobs += [(f"measurements/{measurement_ids[label]}/{name}", payload) - for label, file_list in parsed["files"].items() for name, payload in file_list - if files == "all" or not name.startswith("loops/")] + if uploads: + # one request per file (~1-2 s each), eight in flight + jobs = [(destination(name, set_id, measurement_ids), payload) for name, payload in uploads] with concurrent.futures.ThreadPoolExecutor(max_workers=8) as pool: for done, _ in enumerate(pool.map(lambda job: put_file(thread_client(client), *job, owner["_id"]), jobs), 1): if done % 200 == 0: print(f" files: {done}/{len(jobs)}", flush=True) - print(f"files: {len(jobs)} put under measurements// ({'records/*.json, loops/*' if files == 'all' else 'records/*.json; --files all adds loops/*'})") + print(f"files: {len(jobs)} put") posted = 0 for label, unit_id, prop, repetition in parsed["properties"]: # properties/create is not idempotent: skip what is there present = find(client.properties, {"source.info.origin._id": measurement_ids[label], "data.name": prop["name"], @@ -666,15 +681,14 @@ def main(): ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("run_dir") ap.add_argument("--physical-id", required=True, help="the identifier written on the physical piece the samples are part of, e.g. PDAC_COM5_01448") - ap.add_argument("command", nargs="?", choices=["synthesis", "measurement", "both"], default="both", - help="synthesis: NLR record + photo onto the set · measurement: UTK run onto the set · both (default)") ap.add_argument("--dry-run", action="store_true") ap.add_argument("--emit-example", help="write the first combined loop property (most loops) to this path — the ESSE example") ap.add_argument("--host", default=os.environ.get("MAT3RA_HOST", "localhost:3000"), help="web app host or URL (or MAT3RA_HOST); https unless localhost, e.g. dev.mat3ra.com") ap.add_argument("--account", help="slug of the account the data belongs to (reads scoped to it, writes owned by it)") - ap.add_argument("--files", choices=["records", "all", "none"], default="records", - help="which files to upload per measurement: the record JSONs (default), also the loop arrays and plots (all), or none") + ap.add_argument("--files", nargs="*", default=["records"], metavar="GROUP", + help=f"which file groups to upload ({', '.join(FILE_GROUPS)}); pass --files with no value " + "to upload none and keep the raw records in each measurement's metadata") ap.add_argument("--limit-records", type=int, help="trial: only the first N records and the samples they belong to") ap.add_argument("--deposition", help="NLR HTEM record (json) kept in the run's sample set metadata") ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") @@ -711,7 +725,7 @@ def main(): if a.account: client = APIClient.authenticate(account_id=account_id(client, a.account), **address) for p in runs: - upload(client, p, command=a.command, files=a.files) + upload(client, p, files=a.files) if __name__ == "__main__": From 7dedd1aa7176a868d9a5fd3b8c8de7e9dc2d7749 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 11:52:17 -0700 Subject: [PATCH 16/36] fix(SOF-8051): an image keeps the path that makes its name unique NLR photographs are collected with rglob, so two subdirectories can hold the same basename; mapping both to set/ uploaded one over the other. Both parsers now name each image - UTK by basename, NLR by its path relative to the run folder - and run_files uploads the name it is given. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_run.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index eb2ed9d7d..e868c46cd 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -295,7 +295,7 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= deposition_records.extend(d if isinstance(d, list) else [d]) sample_set["metadata"]["deposition"] = deposition_records # the photograph of the piece: any image at the run-folder root - images = [f for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] + images = [(f.name, f) for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) samples = {s["label"]: {"name": f"{physical_id} {s['label']}", "label": s["label"], "physicalId": physical_id, "position": {"coordinates": [s["x_stage_m"], s["y_stage_m"]], "units": "m"}, @@ -394,7 +394,9 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): grid_file = sorted(folder.rglob("*xrf_grid.txt"))[0] volts_file, amps_file = sorted(folder.rglob("IV_Volts.txt"))[0], sorted(folder.rglob("IV_Amps.txt"))[0] run_name = grid_file.stem - images = [f for f in sorted(folder.rglob("*")) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] + # searched recursively, so the name keeps the subdirectory: two photographs may share a basename + images = [(f.relative_to(folder).as_posix(), f) for f in sorted(folder.rglob("*")) + if f.suffix.lower() in (".jpg", ".jpeg", ".png")] sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} xrf_run_name = f"{run_name} XRF" xrf_workflow = build_nlr_workflow(XRF_APPLICATION, "map", "xrf_grid", "XRF Grid Map", @@ -603,7 +605,7 @@ def run_files(parsed, groups=("records",)): if unknown: raise SystemExit(f"unknown file group(s): {', '.join(sorted(unknown))}; choose from {', '.join(FILE_GROUPS)}") files = [(f"set/{name}", payload) for name, payload in parsed["set_files"]] - files += [(f"set/{image.name}", image) for image in parsed["images"]] + files += [(f"set/{name}", path) for name, path in parsed["images"]] for label, file_list in parsed["files"].items(): files += [(f"{label}/{name}", payload) for name, payload in file_list if ("loops" if name.startswith("loops/") else "records") in groups] From 8be3766878ce4b2eb508d1ab3f4810f112de83c3 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 11:58:13 -0700 Subject: [PATCH 17/36] refactor(SOF-8051): the uploader is the uploader; each instrument reads its own files MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit One 734-line file held two ad hoc parsers and the upload path, so the generic half could not be read without the specific half. Split along that seam: upload_run.py 306 client, sets, members, files, properties, CLI — instrument-agnostic parse_utk.py 305 a UTK SS-PFM run folder parse_nlr.py 91 NLR's XRF grid and DC I-V sweep workflow.py 43 the one workflow builder both parsers call The two near-identical workflow builders became one; `id_prefix` keeps each instrument's stable unit ids exactly as they were, so `source.info.unitId` still means the same thing across uploads. A third instrument is a third parser now, not an edit here. No behaviour change: the parsed documents and the file lists are byte-identical to the previous output for both runs, and `from upload_run import parse, parse_nlr` still resolves for the two notebooks. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/parse_nlr.py | 91 ++++++ examples/measurement/parse_utk.py | 305 ++++++++++++++++++++ examples/measurement/upload_run.py | 446 +---------------------------- examples/measurement/workflow.py | 43 +++ 4 files changed, 448 insertions(+), 437 deletions(-) create mode 100644 examples/measurement/parse_nlr.py create mode 100644 examples/measurement/parse_utk.py create mode 100644 examples/measurement/workflow.py diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py new file mode 100644 index 000000000..b2c84f48f --- /dev/null +++ b/examples/measurement/parse_nlr.py @@ -0,0 +1,91 @@ +"""NLR's delivery for one piece as platform documents: one sample set of the pads they measured, and one run per +technique over those same pads — the XRF map, then the DC I-V sweep. + +Ad hoc parser for SOF-8050: it reads the tab-separated files NLR ships and nothing else. +""" +from pathlib import Path + +from workflow import build_workflow, unit_id + +NLR_FRAME = {"frame": "wafer", "units": "mm", "note": "x_mm, y_mm as delivered by NLR; corner and axes to be confirmed"} +XRF_APPLICATION = {"name": "xrf-mapper", "shortName": "xrf", "summary": "X-ray fluorescence mapper (film thickness and composition over a grid of positions)", + "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} +IV_APPLICATION = {"name": "probe-station", "shortName": "iv", "summary": "DC probe station (current through a pad over a bias sweep)", + "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} + + +def read_columns(path): + """Every line of a tab-separated file after its header, split into its cells.""" + return [line.split("\t") for line in Path(path).read_text().splitlines()[1:] if line.strip()] + + +def nlr_workflow(application, executable_name, flavor_name, name, properties): + """NLR's two instruments are not in the standata registry yet, so their ids are scoped by application name.""" + return build_workflow(application, executable_name, flavor_name, name, properties, + id_prefix=f"{application['name']}/") + + +def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): + """NLR's delivery for one piece as platform documents: one Sample Set of the pads they measured, and one run per + technique over those same pads — the XRF map, then the DC I-V sweep. Two runs, each in the shape parse() returns, + so upload() takes them one after the other: the first creates the Sample Set, the second finds it by name.""" + folder = Path(folder) + grid_file = sorted(folder.rglob("*xrf_grid.txt"))[0] + volts_file, amps_file = sorted(folder.rglob("IV_Volts.txt"))[0], sorted(folder.rglob("IV_Amps.txt"))[0] + run_name = grid_file.stem + # searched recursively, so the name keeps the subdirectory: two photographs may share a basename + images = [(f.relative_to(folder).as_posix(), f) for f in sorted(folder.rglob("*")) + if f.suffix.lower() in (".jpg", ".jpeg", ".png")] + sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} + xrf_run_name = f"{run_name} XRF" + xrf_workflow = nlr_workflow(XRF_APPLICATION, "map", "xrf_grid", "XRF Grid Map", + ["thickness", "al_atomic_fraction", "sc_atomic_fraction"]) + xrf_unit_id = unit_id(xrf_workflow) + grid = read_columns(grid_file) + samples, xrf_measurements, xrf_properties = {}, {}, [] + for row, column, x_mm, y_mm, thickness_um, aluminium_at_pct, scandium_at_pct in grid: + label = f"r{int(row)}c{int(column)}" + # two rows for one pad would overwrite each other here and leave the row counts below + # agreeing against a dictionary that has already lost an entry + if label in samples: + raise SystemExit(f"{grid_file.name}: pad {label} appears twice") + samples[label] = {"name": f"{physical_id} {label}", "label": label, "physicalId": physical_id, + "position": {"coordinates": [float(x_mm), float(y_mm)], "units": "mm"}, + "metadata": {"frame": NLR_FRAME, "row": int(row), "column": int(column)}} + xrf_measurements[label] = {"name": f"{xrf_run_name} {label}", "_sample": None, "workflow": xrf_workflow, + "setup": {"name": xrf_instrument}, "status": "finished", "_records": [], + "metadata": {"row": int(row), "column": int(column), "thickness_um": float(thickness_um), + "al_at_pct": float(aluminium_at_pct), "sc_at_pct": float(scandium_at_pct)}} + xrf_properties += [(label, xrf_unit_id, {"name": "thickness", "value": float(thickness_um), "units": "um"}, 0), + (label, xrf_unit_id, {"name": "al_atomic_fraction", "value": float(aluminium_at_pct), "units": "at%"}, 0), + (label, xrf_unit_id, {"name": "sc_atomic_fraction", "value": float(scandium_at_pct), "units": "at%"}, 0)] + iv_run_name = f"{run_name} DC IV" + iv_workflow = nlr_workflow(IV_APPLICATION, "sweep", "dc_iv", "DC I-V Sweep", ["iv_curve"]) + iv_unit_id = unit_id(iv_workflow) + volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] + amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] + # zip would silently drop pads, so the shapes are checked before any document is built + if not (len(grid) == len(volts) == len(amps)): + raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)}/{len(amps)} rows for " + f"{len(grid)} pads in {grid_file.name} — every pad needs one row in each file") + for row, (bias_row, current_row) in enumerate(zip(volts, amps)): + if len(bias_row) != len(current_row): + raise SystemExit(f"row {row}: {len(bias_row)} bias points but {len(current_row)} current points") + # the sweep NLR ran, read off the voltages themselves; every row of the file holds the same one + iv_setup = {"name": iv_instrument, "settings": {"v_min": min(volts[0]), "v_max": max(volts[0]), "points": len(volts[0])}} + iv_measurements, iv_properties = {}, [] + for index, (label, bias, current) in enumerate(zip(samples, volts, amps)): + iv_measurements[label] = {"name": f"{iv_run_name} {label}", "_sample": None, "workflow": iv_workflow, + "setup": iv_setup, "status": "finished", "_records": [], "metadata": {"row_index": index}} + iv_properties.append((label, iv_unit_id, {"name": "iv_curve", "xAxis": {"label": "bias", "units": "V"}, + "yAxis": {"label": "current", "units": "A"}, + "xDataArray": bias, "yDataSeries": [current]}, 0)) + return [{"physicalId": physical_id, "run": xrf_run_name, "sample_set": sample_set, "images": images, "samples": samples, + "measurement_set": {"name": xrf_run_name, "entitySetType": "ordered", "metadata": {}}, + "measurements": xrf_measurements, "files": {}, "set_files": [(grid_file.name, grid_file)], + "records": grid, "properties": xrf_properties}, + {"physicalId": physical_id, "run": iv_run_name, "sample_set": sample_set, "images": images, "samples": samples, + "measurement_set": {"name": iv_run_name, "entitySetType": "ordered", "metadata": {}}, + "measurements": iv_measurements, "files": {}, "set_files": [(volts_file.name, volts_file), (amps_file.name, amps_file)], + "records": volts, "properties": iv_properties}] + diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py new file mode 100644 index 000000000..4b1a8534c --- /dev/null +++ b/examples/measurement/parse_utk.py @@ -0,0 +1,305 @@ +"""A UTK SS-PFM run folder as platform documents: a sample set of the pads, one measurement per pad, and one +hysteresis-loop property per pad — the eight loops combined, with the loop parameters (mean, population standard +deviation, count) inside it. Individual loops stay in the measurement's files. + +Ad hoc parser for SOF-8050: it reads the shape UTK's afm-lib writes and nothing else. +""" +import ast, json, math, re, statistics, struct +from datetime import datetime, timezone +from pathlib import Path + +from workflow import build_workflow, unit_id + + +FIELD = {"off_field": "off", "on_field": "on"} +# loop_params key -> parameters path (units: voltages in xAxis.units, responses in yAxis.units) +PARAMETERS = { + "imprint_v": ("imprint",), + "v_c_rising": ("coerciveVoltage", "rising"), + "v_c_falling": ("coerciveVoltage", "falling"), + "loop_width_v": ("loopWidth",), + "loop_height_m": ("loopHeight",), + "remnant_rising_m": ("remanentResponse", "rising"), + "remnant_falling_m": ("remanentResponse", "falling"), +} +INSTRUMENT = {"name": "asylum-spm", "shortName": "spm", "summary": "Asylum Research SPM driven by afm-lib (switching-spectroscopy PFM)", + "version": "1.0", "build": "afm-lib", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} +g = lambda v: float(f"{v:.6g}") + + +def load_npy(path): + """A one-dimensional float32/float64 .npy file as a list of floats (no numpy: the format is a header + raw values).""" + data = Path(path).read_bytes() + if data[:6] != b"\x93NUMPY": + raise ValueError(f"{path}: not a .npy file") + header_length = struct.unpack(" 1 else 0.0, "count": len(values)} + + +def combine_pad(label, records, run_dir): + """One hysteresis_loop property for a pad: point-wise mean of the response over its loops (on and off), and + each loop parameter as mean / spread / count over the loops. None when no loop has curves.""" + bias, series, params = None, {"on": [], "off": []}, {"on": {}, "off": {}} + for r in records: + loops_dir = run_dir / "loops" / (r.get("out_stem") or Path(r["file_path"]).stem) + for branch, field in FIELD.items(): + lp = r["loop_params"].get(branch) or {} + for key, path in PARAMETERS.items(): + if lp.get(key) is not None: + params[field].setdefault(path, []).append(lp[key]) + bias_p = loops_dir / f"bias_{field}.npy" + if not bias_p.exists() or lp.get("phase_offset_deg") is None: + continue + b = load_npy(bias_p) + bias = bias or b + curve = response_curve(loops_dir, field, lp["phase_offset_deg"], b) + # averaged point by point, so a loop measured on a different bias axis is dropped rather + # than folded in: same length is not the same voltages + if curve is not None and same_axis(b, bias): + series[field].append(curve) + if bias is None or not series["on"] or not series["off"]: + return None + parameters = {} + for field in ("on", "off"): + block = {} + for path, vals in params[field].items(): + node = block + for k in path[:-1]: + node = node.setdefault(k, {}) + node[path[-1]] = parameter_statistics(vals) + parameters[field] = block + return {"name": "hysteresis_loop", "legend": ["on", "off"], + "xAxis": {"label": "bias", "units": "V"}, "yAxis": {"label": "response", "units": "m"}, + "xDataArray": [g(v) for v in bias], "yDataSeries": [pointwise_mean(series["on"]), pointwise_mean(series["off"])], + "parameters": parameters} + +RECORD_GROUPS = ("labels", "file_path", "requested_params", "instrument_params", "loop_params", "channel_stats") +SAMPLE_KEY = "labels.site_label" # a record's reference to its sample — stays on every record + + +def _flatten(d, prefix=""): + """Nested dict → one level, keys joined with dots.""" + out = {} + for k, v in (d or {}).items(): + key = f"{prefix}.{k}" if prefix else k + if isinstance(v, dict): + out.update(_flatten(v, key)) + else: + out[key] = v + return out + + +def _unflatten(flat): + """The inverse of _flatten.""" + out = {} + for key, v in flat.items(): + node = out + parts = key.split(".") + for part in parts[:-1]: + node = node.setdefault(part, {}) + node[parts[-1]] = v + return out + + +def _constant_keys(flat_rows): + """Keys present in every row with the same value everywhere. A key one row lacks is not constant, even if the + rows that have it agree — otherwise a group that is null on one record and absent on another looks shared.""" + if not flat_rows: + return set() + keys = set.intersection(*(set(row) for row in flat_rows)) + return {k for k in keys if all(row[k] == flat_rows[0][k] for row in flat_rows)} + + +def factor_records(records): + """Each record keeps only what is unique to it. Fields identical across the whole run move to the measurement + (`common`); fields identical across one sample's records move to that sample's metadata; the record keeps its + sample reference and the values that actually vary per loop (measured 128 → 43 / 4 / 81 on the 704-record run).""" + if not records: + return {}, {}, [] + flat = [_flatten({k: r.get(k) for k in RECORD_GROUPS}) for r in records] + common_keys = _constant_keys(flat) - {SAMPLE_KEY} + by_sample = {} + for row in flat: + by_sample.setdefault(row[SAMPLE_KEY], []).append(row) + sample_keys = set.intersection(*(_constant_keys(rows) for rows in by_sample.values())) - common_keys - {SAMPLE_KEY} + # a group can be a dict on one record and null on another: read with .get, never index + common = _unflatten({k: flat[0].get(k) for k in sorted(common_keys)}) + per_sample = {label: _unflatten({k: rows[0].get(k) for k in sorted(sample_keys)}) for label, rows in by_sample.items()} + slim = [_unflatten({k: v for k, v in row.items() if k not in common_keys and k not in sample_keys}) for row in flat] + return common, per_sample, slim + +def starting_site(recipe): + """The site the run started from: r0c00 when the recipe has it, else the first one listed.""" + return next((s for s in recipe["sites"] if s["label"] == "r0c00"), recipe["sites"][0]) + + +def thinned_curves(prop, points=12): + """The curves at `points` evenly spaced samples: an ESSE example shows the shape, not the data.""" + x = prop["xDataArray"] + if len(x) <= points: + return {} + keep = [round(i * (len(x) - 1) / (points - 1)) for i in range(points)] + return {"xDataArray": [x[i] for i in keep], + "yDataSeries": [[s[i] for i in keep] for s in prop["yDataSeries"]]} + + +def registration(recipe): + """The instrument's frame as UTK stated it: one anchor in words (from recipe.context) and where the run started + on the stage. Recorded, not interpreted.""" + ctx = recipe.get("context", "") + m = re.search(r"starting point is (.+?)(?:\.|$)", ctx) + r0 = starting_site(recipe) + return {"frame": "asylum-spm stage", "units": "m", "anchor": m.group(1).strip() if m else ctx, + f"{r0['label']}_stage_m": [r0["x_stage_m"], r0["y_stage_m"]]} + + +def sample_files(label, records, run_dir, slim_by_index): + """Files of one sample's measurement: one JSON per record (the fields unique to it) and, when the run folder has them, + the loop arrays and annotated plots. Returned as (relative name, payload) where payload is text or a Path.""" + out = [] + for r, slim in zip(records, slim_by_index): + step, point = r["labels"].get("step_index", 0), r["labels"].get("point_index", 0) + out.append((f"records/step{step}_pt{point:02d}.json", json.dumps(slim, indent=1))) + d = run_dir / "loops" / r.get("out_stem", "") + if r.get("out_stem") and d.is_dir(): + for f in sorted(d.iterdir()): + if f.suffix in (".npy", ".png"): + out.append((f"loops/{d.name}/{f.name}", f)) + return out + + +def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument="asylum-afm"): + """The whole run folder as platform documents: sample set, samples, measurement set, one measurement per sample, files, one loop property per fully measured sample.""" + run_dir = Path(run_dir) + recipe, session, all_records = load_run(run_dir) + records = all_records[:limit_records] if limit_records else all_records + if not physical_id: + raise ValueError("--physical-id: the identifier written on the physical piece is required") + run_name = session.get("name") or run_dir.name + reg = registration(recipe) + sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} + # NLR's HTEM deposition record(s) for the piece, verbatim. UTK drops the file into + # the run folder as deposition*.json; --deposition overrides that. + deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) + if deposition_files: + deposition_records = [] + for f in deposition_files: + d = json.loads(f.read_text()) + deposition_records.extend(d if isinstance(d, list) else [d]) + sample_set["metadata"]["deposition"] = deposition_records + # the photograph of the piece: any image at the run-folder root + images = [(f.name, f) for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] + # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) + samples = {s["label"]: {"name": f"{physical_id} {s['label']}", "label": s["label"], "physicalId": physical_id, + "position": {"coordinates": [s["x_stage_m"], s["y_stage_m"]], "units": "m"}, + "metadata": {"registration": reg}} + for s in recipe["sites"]} + if limit_records: + # a trial run must be a prefix of a full one: only samples whose records ALL made the cut, so no sample is + # ever published with a partial loop count that a full run would then skip as "already there" + full_counts, kept_counts = {}, {} + for r in all_records: + full_counts[r["labels"]["site_label"]] = full_counts.get(r["labels"]["site_label"], 0) + 1 + for r in records: + kept_counts[r["labels"]["site_label"]] = kept_counts.get(r["labels"]["site_label"], 0) + 1 + complete = {label for label, n in kept_counts.items() if n == full_counts[label]} + records = [r for r in records if r["labels"]["site_label"] in complete] + samples = {label: sample for label, sample in samples.items() if label in complete} + common, per_sample, slim_records = factor_records(records) + for label, const in per_sample.items(): + if label in samples: + samples[label]["metadata"].update(const) + workflow = build_workflow(INSTRUMENT, "loop", "ss_pfm", "SS-PFM Hysteresis Loop", ["hysteresis_loop"], + unit_name="run_loop", tags=["experimental", "afm"], + metadata={"recipe": recipe, "loop_settings": recipe["per_site"][0]["loop_settings"], + "sites": list(samples)}) + unit = unit_id(workflow) + measurement_set = {"name": run_name, "entitySetType": "ordered", + "metadata": {"session": session, "recipe": recipe["name"], "context": recipe.get("context", ""), + "common": common, "registration": reg}} + # the setup block, Measurement : setup :: Job : compute — the machine and the sitting; the technique + # (asylum-spm, SS-PFM) is the workflow's application. The run folder does not name the machine: --instrument does. + started = session.get("started_ts") + setup_block = {"name": instrument, + "session": {k: v for k, v in {"name": session.get("name"), + "started": datetime.fromtimestamp(started, timezone.utc).isoformat().replace("+00:00", "Z") if started else None}.items() if v}, + **({"settings": common["instrument_params"]} if common.get("instrument_params") else {})} + by_sample, slim_by_sample = {}, {} + for r, slim in zip(records, slim_records): + lab = r["labels"]["site_label"] + by_sample.setdefault(lab, []).append(r); slim_by_sample.setdefault(lab, []).append(slim) + # one measurement per sample: Measurement : Sample :: Job : Material + measurements, files, properties, skipped = {}, {}, [], [] + for label in samples: + recs = by_sample.get(label, []) + measurements[label] = {"name": f"{run_name} {label}", "_sample": None, "workflow": workflow, + "setup": setup_block, "status": "finished", + "metadata": {"run_dir": session.get("run_dir", str(run_dir)), "recordsCount": len(recs)}} + slim_by_label = slim_by_sample.get(label, []) + measurements[label]["_records"] = slim_by_label # not sent; upload() decides files vs metadata + files[label] = sample_files(label, recs, run_dir, slim_by_sample.get(label, [])) + prop = combine_pad(label, recs, run_dir) if recs else None + (properties.append((label, unit, prop, 0)) if prop else skipped.append(label)) + return {"physicalId": physical_id, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, + "measurement_set": measurement_set, "measurements": measurements, "files": files, "set_files": [], + "records": records, "properties": properties, "skipped": skipped} + + + +def same_axis(candidate, reference, tolerance=1e-9): + """Whether two bias axes are the same sweep: equal length and equal voltages within tolerance.""" + return len(candidate) == len(reference) and all( + abs(a - b) <= tolerance + 1e-6 * abs(b) for a, b in zip(candidate, reference)) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index e868c46cd..a8b1f6708 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -1,456 +1,33 @@ #!/usr/bin/env python3 -"""UTK run dir -> platform documents (sample set, samples, measurement, one hysteresis-loop property per sample), -validated against the ESSE schemas, then uploaded through the REST API. - -Ad hoc parser for SOF-8050. Field kinds follow ONTOLOGY.md. The property model follows PLAN-S3-loop-property.md: -the property is the pad's hysteresis loop — the eight loops combined — with the loop parameters (mean, population -standard deviation, count over the loops) inside it. Individual loops stay in the measurement's metadata. +"""Parsed runs -> platform documents, validated against the ESSE schemas, then uploaded through the REST API. upload_run.py --physical-id --account # upload the run upload_run.py --physical-id --dry-run [--emit-example out.json] # parse + validate only; write one property as the ESSE example upload_run.py --physical-id --nlr # NLR's XRF grid and DC I-V sweep instead of a UTK run +The instrument-specific reading lives beside this file — `parse_utk.py` for a UTK SS-PFM run, `parse_nlr.py` for +NLR's delivery — and each returns the same dict, so everything below is the same for both. A third instrument is +a third parser, not a change here. + Requires Python 3.9+ and `pip install mat3ra-api-client`, which talks to the platform and takes OIDC_ACCESS_TOKEN, or ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. Optional: `pip install mat3ra-esse` (tested with 2026.8.27-0) turns on schema validation before anything is uploaded. - -Canonical copy: mat3ra/api-examples `examples/measurement/upload_run.py`, beside the notebook that imports it. The -planning repo keeps a working copy and the tests; scripts/publish.sh copies api-examples → plan by default and -plan → api-examples with --push. """ -import argparse, ast, concurrent.futures, json, math, os, re, statistics, struct, sys, threading, time, urllib.parse, uuid -from datetime import datetime, timezone +import argparse, concurrent.futures, json, os, sys, threading, time, urllib.parse from pathlib import Path import requests from mat3ra.api_client import APIClient +from parse_nlr import parse_nlr +from parse_utk import parse, thinned_curves + try: # optional: schema validation before anything is sent from mat3ra.esse import ESSE from mat3ra.esse.models.sample import SampleSchema except ImportError: ESSE = SampleSchema = None -FIELD = {"off_field": "off", "on_field": "on"} -# loop_params key -> parameters path (units: voltages in xAxis.units, responses in yAxis.units) -PARAMETERS = { - "imprint_v": ("imprint",), - "v_c_rising": ("coerciveVoltage", "rising"), - "v_c_falling": ("coerciveVoltage", "falling"), - "loop_width_v": ("loopWidth",), - "loop_height_m": ("loopHeight",), - "remnant_rising_m": ("remanentResponse", "rising"), - "remnant_falling_m": ("remanentResponse", "falling"), -} -INSTRUMENT = {"name": "asylum-spm", "shortName": "spm", "summary": "Asylum Research SPM driven by afm-lib (switching-spectroscopy PFM)", - "version": "1.0", "build": "afm-lib", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} -g = lambda v: float(f"{v:.6g}") - - -def load_npy(path): - """A one-dimensional float32/float64 .npy file as a list of floats (no numpy: the format is a header + raw values).""" - data = Path(path).read_bytes() - if data[:6] != b"\x93NUMPY": - raise ValueError(f"{path}: not a .npy file") - header_length = struct.unpack(" 1 else 0.0, "count": len(values)} - - -def combine_pad(label, records, run_dir): - """One hysteresis_loop property for a pad: point-wise mean of the response over its loops (on and off), and - each loop parameter as mean / spread / count over the loops. None when no loop has curves.""" - bias, series, params = None, {"on": [], "off": []}, {"on": {}, "off": {}} - for r in records: - loops_dir = run_dir / "loops" / (r.get("out_stem") or Path(r["file_path"]).stem) - for branch, field in FIELD.items(): - lp = r["loop_params"].get(branch) or {} - for key, path in PARAMETERS.items(): - if lp.get(key) is not None: - params[field].setdefault(path, []).append(lp[key]) - bias_p = loops_dir / f"bias_{field}.npy" - if not bias_p.exists() or lp.get("phase_offset_deg") is None: - continue - b = load_npy(bias_p) - bias = bias or b - curve = response_curve(loops_dir, field, lp["phase_offset_deg"], b) - # averaged point by point, so a loop measured on a different bias axis is dropped rather - # than folded in: same length is not the same voltages - if curve is not None and same_axis(b, bias): - series[field].append(curve) - if bias is None or not series["on"] or not series["off"]: - return None - parameters = {} - for field in ("on", "off"): - block = {} - for path, vals in params[field].items(): - node = block - for k in path[:-1]: - node = node.setdefault(k, {}) - node[path[-1]] = parameter_statistics(vals) - parameters[field] = block - return {"name": "hysteresis_loop", "legend": ["on", "off"], - "xAxis": {"label": "bias", "units": "V"}, "yAxis": {"label": "response", "units": "m"}, - "xDataArray": [g(v) for v in bias], "yDataSeries": [pointwise_mean(series["on"]), pointwise_mean(series["off"])], - "parameters": parameters} - - -WORKFLOW_NAMESPACE = uuid.UUID("6f3b0b0e-8c1e-4b7a-9f21-3a5f0e2d1c44") # stable ids: the same workflow every upload - - -def build_workflow(recipe, labels): - """The procedure that runs on UTK's Asylum SPM, the same shape as standata's `asylum-spm/ss_pfm` workflow: ONE - workflow "SS-PFM Hysteresis Loop", one subworkflow `ss_pfm`, one execution unit `run_loop` declaring - `hysteresis_loop`. It is the same workflow for every sample in the set — the measurement is not re-planned per - sample — and its ids are stable (uuid5 of the unit names), so a property's `source.info.unitId` means the same - thing across uploads. Application / executable / flavor are the standata registry entries (asylum-spm / loop / - ss_pfm) so the platform resolves the unit exactly as it does a job's. The recipe travels in metadata.""" - result = [{"name": "hysteresis_loop"}] - monitors = [{"name": "standard_output"}] - executable = {"name": "loop", "applicationName": INSTRUMENT["name"], "applicationVersion": "*", "isDefault": True, - "monitors": monitors, "results": result, "preProcessors": [], "postProcessors": []} - flavor = {"name": "ss_pfm", "executableName": "loop", "applicationName": INSTRUMENT["name"], "applicationVersion": "*", - "isDefault": True, "input": [], "monitors": monitors, "results": result, "preProcessors": [], "postProcessors": []} - unit = {"type": "execution", "name": "run_loop", "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, "run_loop").hex[:24], "head": True, "status": "finished", - "application": INSTRUMENT, "executable": executable, "flavor": flavor, "input": [], "context": [], - "monitors": monitors, "results": result, "preProcessors": [], "postProcessors": []} - # ESSE requires a model on every subworkflow; an experiment has none, so the legacy "unknown" model. - model = {"type": "unknown", "subtype": "unknown", "method": {"type": "unknown", "subtype": "unknown"}} - sw_id = uuid.uuid5(WORKFLOW_NAMESPACE, "ss_pfm").hex[:17] - subworkflow = {"_id": sw_id, "name": "ss_pfm", "application": INSTRUMENT, "model": model, - "properties": ["hysteresis_loop"], "units": [unit]} - # A subworkflow unit carries the subworkflow's own `_id` — that is how the platform pairs them. - sw_unit = {"_id": sw_id, "type": "subworkflow", "name": subworkflow["name"], "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, "ss_pfm/unit").hex[:24], - "head": True, "status": "finished", "preProcessors": [], "postProcessors": [], "monitors": [], "results": []} - return {"name": "SS-PFM Hysteresis Loop", "isDefault": False, "tags": ["experimental", "afm"], "properties": ["hysteresis_loop"], - "application": INSTRUMENT, "subworkflows": [subworkflow], "units": [sw_unit], "workflows": [], - "metadata": {"recipe": recipe, "loop_settings": recipe["per_site"][0]["loop_settings"], "sites": list(labels)}} - - - -RECORD_GROUPS = ("labels", "file_path", "requested_params", "instrument_params", "loop_params", "channel_stats") -SAMPLE_KEY = "labels.site_label" # a record's reference to its sample — stays on every record - - -def _flatten(d, prefix=""): - """Nested dict → one level, keys joined with dots.""" - out = {} - for k, v in (d or {}).items(): - key = f"{prefix}.{k}" if prefix else k - if isinstance(v, dict): - out.update(_flatten(v, key)) - else: - out[key] = v - return out - - -def _unflatten(flat): - """The inverse of _flatten.""" - out = {} - for key, v in flat.items(): - node = out - parts = key.split(".") - for part in parts[:-1]: - node = node.setdefault(part, {}) - node[parts[-1]] = v - return out - - -def _constant_keys(flat_rows): - """Keys present in every row with the same value everywhere. A key one row lacks is not constant, even if the - rows that have it agree — otherwise a group that is null on one record and absent on another looks shared.""" - if not flat_rows: - return set() - keys = set.intersection(*(set(row) for row in flat_rows)) - return {k for k in keys if all(row[k] == flat_rows[0][k] for row in flat_rows)} - - -def factor_records(records): - """Each record keeps only what is unique to it. Fields identical across the whole run move to the measurement - (`common`); fields identical across one sample's records move to that sample's metadata; the record keeps its - sample reference and the values that actually vary per loop (measured 128 → 43 / 4 / 81 on the 704-record run).""" - if not records: - return {}, {}, [] - flat = [_flatten({k: r.get(k) for k in RECORD_GROUPS}) for r in records] - common_keys = _constant_keys(flat) - {SAMPLE_KEY} - by_sample = {} - for row in flat: - by_sample.setdefault(row[SAMPLE_KEY], []).append(row) - sample_keys = set.intersection(*(_constant_keys(rows) for rows in by_sample.values())) - common_keys - {SAMPLE_KEY} - # a group can be a dict on one record and null on another: read with .get, never index - common = _unflatten({k: flat[0].get(k) for k in sorted(common_keys)}) - per_sample = {label: _unflatten({k: rows[0].get(k) for k in sorted(sample_keys)}) for label, rows in by_sample.items()} - slim = [_unflatten({k: v for k, v in row.items() if k not in common_keys and k not in sample_keys}) for row in flat] - return common, per_sample, slim - -def starting_site(recipe): - """The site the run started from: r0c00 when the recipe has it, else the first one listed.""" - return next((s for s in recipe["sites"] if s["label"] == "r0c00"), recipe["sites"][0]) - - -def thinned_curves(prop, points=12): - """The curves at `points` evenly spaced samples: an ESSE example shows the shape, not the data.""" - x = prop["xDataArray"] - if len(x) <= points: - return {} - keep = [round(i * (len(x) - 1) / (points - 1)) for i in range(points)] - return {"xDataArray": [x[i] for i in keep], - "yDataSeries": [[s[i] for i in keep] for s in prop["yDataSeries"]]} - - -def registration(recipe): - """The instrument's frame as UTK stated it: one anchor in words (from recipe.context) and where the run started - on the stage. Recorded, not interpreted.""" - ctx = recipe.get("context", "") - m = re.search(r"starting point is (.+?)(?:\.|$)", ctx) - r0 = starting_site(recipe) - return {"frame": "asylum-spm stage", "units": "m", "anchor": m.group(1).strip() if m else ctx, - f"{r0['label']}_stage_m": [r0["x_stage_m"], r0["y_stage_m"]]} - - -def sample_files(label, records, run_dir, slim_by_index): - """Files of one sample's measurement: one JSON per record (the fields unique to it) and, when the run folder has them, - the loop arrays and annotated plots. Returned as (relative name, payload) where payload is text or a Path.""" - out = [] - for r, slim in zip(records, slim_by_index): - step, point = r["labels"].get("step_index", 0), r["labels"].get("point_index", 0) - out.append((f"records/step{step}_pt{point:02d}.json", json.dumps(slim, indent=1))) - d = run_dir / "loops" / r.get("out_stem", "") - if r.get("out_stem") and d.is_dir(): - for f in sorted(d.iterdir()): - if f.suffix in (".npy", ".png"): - out.append((f"loops/{d.name}/{f.name}", f)) - return out - - -def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument="asylum-afm"): - """The whole run folder as platform documents: sample set, samples, measurement set, one measurement per sample, files, one loop property per fully measured sample.""" - run_dir = Path(run_dir) - recipe, session, all_records = load_run(run_dir) - records = all_records[:limit_records] if limit_records else all_records - if not physical_id: - raise ValueError("--physical-id: the identifier written on the physical piece is required") - run_name = session.get("name") or run_dir.name - reg = registration(recipe) - sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} - # NLR's HTEM deposition record(s) for the piece, verbatim. UTK drops the file into - # the run folder as deposition*.json; --deposition overrides that. - deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) - if deposition_files: - deposition_records = [] - for f in deposition_files: - d = json.loads(f.read_text()) - deposition_records.extend(d if isinstance(d, list) else [d]) - sample_set["metadata"]["deposition"] = deposition_records - # the photograph of the piece: any image at the run-folder root - images = [(f.name, f) for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] - # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) - samples = {s["label"]: {"name": f"{physical_id} {s['label']}", "label": s["label"], "physicalId": physical_id, - "position": {"coordinates": [s["x_stage_m"], s["y_stage_m"]], "units": "m"}, - "metadata": {"registration": reg}} - for s in recipe["sites"]} - if limit_records: - # a trial run must be a prefix of a full one: only samples whose records ALL made the cut, so no sample is - # ever published with a partial loop count that a full run would then skip as "already there" - full_counts, kept_counts = {}, {} - for r in all_records: - full_counts[r["labels"]["site_label"]] = full_counts.get(r["labels"]["site_label"], 0) + 1 - for r in records: - kept_counts[r["labels"]["site_label"]] = kept_counts.get(r["labels"]["site_label"], 0) + 1 - complete = {label for label, n in kept_counts.items() if n == full_counts[label]} - records = [r for r in records if r["labels"]["site_label"] in complete] - samples = {label: sample for label, sample in samples.items() if label in complete} - common, per_sample, slim_records = factor_records(records) - for label, const in per_sample.items(): - if label in samples: - samples[label]["metadata"].update(const) - workflow = build_workflow(recipe, list(samples)) - unit_id = workflow["subworkflows"][0]["units"][0]["flowchartId"] - measurement_set = {"name": run_name, "entitySetType": "ordered", - "metadata": {"session": session, "recipe": recipe["name"], "context": recipe.get("context", ""), - "common": common, "registration": reg}} - # the setup block, Measurement : setup :: Job : compute — the machine and the sitting; the technique - # (asylum-spm, SS-PFM) is the workflow's application. The run folder does not name the machine: --instrument does. - started = session.get("started_ts") - setup_block = {"name": instrument, - "session": {k: v for k, v in {"name": session.get("name"), - "started": datetime.fromtimestamp(started, timezone.utc).isoformat().replace("+00:00", "Z") if started else None}.items() if v}, - **({"settings": common["instrument_params"]} if common.get("instrument_params") else {})} - by_sample, slim_by_sample = {}, {} - for r, slim in zip(records, slim_records): - lab = r["labels"]["site_label"] - by_sample.setdefault(lab, []).append(r); slim_by_sample.setdefault(lab, []).append(slim) - # one measurement per sample: Measurement : Sample :: Job : Material - measurements, files, properties, skipped = {}, {}, [], [] - for label in samples: - recs = by_sample.get(label, []) - measurements[label] = {"name": f"{run_name} {label}", "_sample": None, "workflow": workflow, - "setup": setup_block, "status": "finished", - "metadata": {"run_dir": session.get("run_dir", str(run_dir)), "recordsCount": len(recs)}} - slim_by_label = slim_by_sample.get(label, []) - measurements[label]["_records"] = slim_by_label # not sent; upload() decides files vs metadata - files[label] = sample_files(label, recs, run_dir, slim_by_sample.get(label, [])) - prop = combine_pad(label, recs, run_dir) if recs else None - (properties.append((label, unit_id, prop, 0)) if prop else skipped.append(label)) - return {"physicalId": physical_id, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, - "measurement_set": measurement_set, "measurements": measurements, "files": files, "set_files": [], - "records": records, "properties": properties, "skipped": skipped} - - -NLR_FRAME = {"frame": "wafer", "units": "mm", "note": "x_mm, y_mm as delivered by NLR; corner and axes to be confirmed"} -XRF_APPLICATION = {"name": "xrf-mapper", "shortName": "xrf", "summary": "X-ray fluorescence mapper (film thickness and composition over a grid of positions)", - "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} -IV_APPLICATION = {"name": "probe-station", "shortName": "iv", "summary": "DC probe station (current through a pad over a bias sweep)", - "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} - - -def read_columns(path): - """Every line of a tab-separated file after its header, split into its cells.""" - return [line.split("\t") for line in Path(path).read_text().splitlines()[1:] if line.strip()] - - -def build_nlr_workflow(application, executable_name, flavor_name, name, properties): - """The procedure one of NLR's instruments runs, in the shape build_workflow gives UTK's: ONE workflow, one - subworkflow, one execution unit declaring what it produces, ids stable across uploads (uuid5 of the names). - Application, executable and flavor name the standata registry entries these two instruments still need.""" - results = [{"name": property_name} for property_name in properties] - monitors = [{"name": "standard_output"}] - executable = {"name": executable_name, "applicationName": application["name"], "applicationVersion": "*", "isDefault": True, - "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} - flavor = {"name": flavor_name, "executableName": executable_name, "applicationName": application["name"], "applicationVersion": "*", - "isDefault": True, "input": [], "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} - unit = {"type": "execution", "name": executable_name, "head": True, "status": "finished", - "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, f"{application['name']}/{executable_name}").hex[:24], - "application": application, "executable": executable, "flavor": flavor, "input": [], "context": [], - "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} - model = {"type": "unknown", "subtype": "unknown", "method": {"type": "unknown", "subtype": "unknown"}} - subworkflow_id = uuid.uuid5(WORKFLOW_NAMESPACE, f"{application['name']}/{flavor_name}").hex[:17] - subworkflow = {"_id": subworkflow_id, "name": flavor_name, "application": application, "model": model, - "properties": properties, "units": [unit]} - subworkflow_unit = {"_id": subworkflow_id, "type": "subworkflow", "name": flavor_name, "head": True, "status": "finished", - "flowchartId": uuid.uuid5(WORKFLOW_NAMESPACE, f"{application['name']}/{flavor_name}/unit").hex[:24], - "preProcessors": [], "postProcessors": [], "monitors": [], "results": []} - return {"name": name, "isDefault": False, "tags": ["experimental"], "properties": properties, "application": application, - "subworkflows": [subworkflow], "units": [subworkflow_unit], "workflows": []} - - -def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): - """NLR's delivery for one piece as platform documents: one Sample Set of the pads they measured, and one run per - technique over those same pads — the XRF map, then the DC I-V sweep. Two runs, each in the shape parse() returns, - so upload() takes them one after the other: the first creates the Sample Set, the second finds it by name.""" - folder = Path(folder) - grid_file = sorted(folder.rglob("*xrf_grid.txt"))[0] - volts_file, amps_file = sorted(folder.rglob("IV_Volts.txt"))[0], sorted(folder.rglob("IV_Amps.txt"))[0] - run_name = grid_file.stem - # searched recursively, so the name keeps the subdirectory: two photographs may share a basename - images = [(f.relative_to(folder).as_posix(), f) for f in sorted(folder.rglob("*")) - if f.suffix.lower() in (".jpg", ".jpeg", ".png")] - sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} - xrf_run_name = f"{run_name} XRF" - xrf_workflow = build_nlr_workflow(XRF_APPLICATION, "map", "xrf_grid", "XRF Grid Map", - ["thickness", "al_atomic_fraction", "sc_atomic_fraction"]) - xrf_unit_id = xrf_workflow["subworkflows"][0]["units"][0]["flowchartId"] - grid = read_columns(grid_file) - samples, xrf_measurements, xrf_properties = {}, {}, [] - for row, column, x_mm, y_mm, thickness_um, aluminium_at_pct, scandium_at_pct in grid: - label = f"r{int(row)}c{int(column)}" - # two rows for one pad would overwrite each other here and leave the row counts below - # agreeing against a dictionary that has already lost an entry - if label in samples: - raise SystemExit(f"{grid_file.name}: pad {label} appears twice") - samples[label] = {"name": f"{physical_id} {label}", "label": label, "physicalId": physical_id, - "position": {"coordinates": [float(x_mm), float(y_mm)], "units": "mm"}, - "metadata": {"frame": NLR_FRAME, "row": int(row), "column": int(column)}} - xrf_measurements[label] = {"name": f"{xrf_run_name} {label}", "_sample": None, "workflow": xrf_workflow, - "setup": {"name": xrf_instrument}, "status": "finished", "_records": [], - "metadata": {"row": int(row), "column": int(column), "thickness_um": float(thickness_um), - "al_at_pct": float(aluminium_at_pct), "sc_at_pct": float(scandium_at_pct)}} - xrf_properties += [(label, xrf_unit_id, {"name": "thickness", "value": float(thickness_um), "units": "um"}, 0), - (label, xrf_unit_id, {"name": "al_atomic_fraction", "value": float(aluminium_at_pct), "units": "at%"}, 0), - (label, xrf_unit_id, {"name": "sc_atomic_fraction", "value": float(scandium_at_pct), "units": "at%"}, 0)] - iv_run_name = f"{run_name} DC IV" - iv_workflow = build_nlr_workflow(IV_APPLICATION, "sweep", "dc_iv", "DC I-V Sweep", ["iv_curve"]) - iv_unit_id = iv_workflow["subworkflows"][0]["units"][0]["flowchartId"] - volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] - amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] - # zip would silently drop pads, so the shapes are checked before any document is built - if not (len(grid) == len(volts) == len(amps)): - raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)}/{len(amps)} rows for " - f"{len(grid)} pads in {grid_file.name} — every pad needs one row in each file") - for row, (bias_row, current_row) in enumerate(zip(volts, amps)): - if len(bias_row) != len(current_row): - raise SystemExit(f"row {row}: {len(bias_row)} bias points but {len(current_row)} current points") - # the sweep NLR ran, read off the voltages themselves; every row of the file holds the same one - iv_setup = {"name": iv_instrument, "settings": {"v_min": min(volts[0]), "v_max": max(volts[0]), "points": len(volts[0])}} - iv_measurements, iv_properties = {}, [] - for index, (label, bias, current) in enumerate(zip(samples, volts, amps)): - iv_measurements[label] = {"name": f"{iv_run_name} {label}", "_sample": None, "workflow": iv_workflow, - "setup": iv_setup, "status": "finished", "_records": [], "metadata": {"row_index": index}} - iv_properties.append((label, iv_unit_id, {"name": "iv_curve", "xAxis": {"label": "bias", "units": "V"}, - "yAxis": {"label": "current", "units": "A"}, - "xDataArray": bias, "yDataSeries": [current]}, 0)) - return [{"physicalId": physical_id, "run": xrf_run_name, "sample_set": sample_set, "images": images, "samples": samples, - "measurement_set": {"name": xrf_run_name, "entitySetType": "ordered", "metadata": {}}, - "measurements": xrf_measurements, "files": {}, "set_files": [(grid_file.name, grid_file)], - "records": grid, "properties": xrf_properties}, - {"physicalId": physical_id, "run": iv_run_name, "sample_set": sample_set, "images": images, "samples": samples, - "measurement_set": {"name": iv_run_name, "entitySetType": "ordered", "metadata": {}}, - "measurements": iv_measurements, "files": {}, "set_files": [(volts_file.name, volts_file), (amps_file.name, amps_file)], - "records": volts, "properties": iv_properties}] - - def holder(prop, measurement_id, sample_id, unit_id, repetition): """The property holder the platform stores: the data, where it came from (measurement, sample, workflow unit) and a repetition index — 0, since a measurement holds one sample and one loop property.""" @@ -500,11 +77,6 @@ def validate(parsed): return errors -def same_axis(candidate, reference, tolerance=1e-9): - """Whether two bias axes are the same sweep: equal length and equal voltages within tolerance.""" - return len(candidate) == len(reference) and all( - abs(a - b) <= tolerance + 1e-6 * abs(b) for a, b in zip(candidate, reference)) - def base_url(host): """`https://` unless told otherwise: a bare hostname becomes https, localhost/127.0.0.1 http, a URL is kept.""" diff --git a/examples/measurement/workflow.py b/examples/measurement/workflow.py new file mode 100644 index 000000000..f9def668c --- /dev/null +++ b/examples/measurement/workflow.py @@ -0,0 +1,43 @@ +"""The measurement workflow a parser attaches to its measurements. + +An experiment's workflow is a description, not a plan the platform executes: ONE workflow, one subworkflow, one +execution unit that declares what the instrument produces. Application, executable and flavor name the standata +registry entries, so the platform resolves the unit exactly as it does a job's, and the ids are stable (uuid5 of +the names) so a property's `source.info.unitId` means the same thing across uploads. +""" +import uuid + +WORKFLOW_NAMESPACE = uuid.UUID("6f3b0b0e-8c1e-4b7a-9f21-3a5f0e2d1c44") +# ESSE requires a model on every subworkflow; an experiment has none, so the legacy "unknown" model. +UNKNOWN_MODEL = {"type": "unknown", "subtype": "unknown", "method": {"type": "unknown", "subtype": "unknown"}} + + +def build_workflow(application, executable_name, flavor_name, name, properties, + unit_name=None, tags=("experimental",), metadata=None, id_prefix=""): + """`properties` are what the unit declares it produces. `unit_name` defaults to the executable's; + `id_prefix` scopes the stable ids, so two instruments may share an executable name.""" + unit_name = unit_name or executable_name + results = [{"name": property_name} for property_name in properties] + monitors = [{"name": "standard_output"}] + stable = lambda key, length: uuid.uuid5(WORKFLOW_NAMESPACE, f"{id_prefix}{key}").hex[:length] + executable = {"name": executable_name, "applicationName": application["name"], "applicationVersion": "*", "isDefault": True, + "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} + flavor = {"name": flavor_name, "executableName": executable_name, "applicationName": application["name"], "applicationVersion": "*", + "isDefault": True, "input": [], "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} + unit = {"type": "execution", "name": unit_name, "flowchartId": stable(unit_name, 24), "head": True, "status": "finished", + "application": application, "executable": executable, "flavor": flavor, "input": [], "context": [], + "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} + subworkflow_id = stable(flavor_name, 17) + subworkflow = {"_id": subworkflow_id, "name": flavor_name, "application": application, "model": UNKNOWN_MODEL, + "properties": list(properties), "units": [unit]} + # A subworkflow unit carries the subworkflow's own `_id` — that is how the platform pairs them. + subworkflow_unit = {"_id": subworkflow_id, "type": "subworkflow", "name": flavor_name, "flowchartId": stable(f"{flavor_name}/unit", 24), + "head": True, "status": "finished", "preProcessors": [], "postProcessors": [], "monitors": [], "results": []} + workflow = {"name": name, "isDefault": False, "tags": list(tags), "properties": list(properties), "application": application, + "subworkflows": [subworkflow], "units": [subworkflow_unit], "workflows": []} + return dict(workflow, metadata=metadata) if metadata is not None else workflow + + +def unit_id(workflow): + """The execution unit the properties of this workflow come from.""" + return workflow["subworkflows"][0]["units"][0]["flowchartId"] From 3cb193b8bb8037dff1ff63c06e883c0b23a2cf10 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 12:08:35 -0700 Subject: [PATCH 18/36] refactor(SOF-8051): the uploader takes documents; parsing, serializing and the workflow are elsewhere MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three concerns were one script. Now: parse_utk.py / parse_nlr.py read one lab's delivery and write a run document run_document.py the shape they agree on, and reading it back upload_run.py takes run documents + a host and an account, and uploads `upload_run.py --account ` names no instrument, no file format and no lab. A new source is a new parser and nothing here changes. Each parser has its own command line, so parsing is a step you can run, inspect and keep: the document lands beside the files it names, paths relative to itself, and derived files (the records cut out of a run) are written next to it. The workflow is no longer built in Python — `standata_workflow()` fetches the registry entry the platform itself resolves, so there is one source of truth: asylum-spm/SS-PFM Hysteresis Loop, xrf-mapper/XRF Grid Map, probe-station/DC I-V Sweep (the latter two added in standata d2dd2ca2). UTK's recipe travelled inside the workflow it built; it is this run's, not the procedure's, so it now sits in the measurement set's metadata. Images were a special case in run_files; they are simply the set's files now. Verified on both deliveries: every document validates against ESSE, and every one of the 5,976 + 7 files a document names is on disk. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/parse_nlr.py | 58 +++-- examples/measurement/parse_utk.py | 64 +++++- examples/measurement/run_document.py | 73 ++++++ examples/measurement/upload_nlr_data.ipynb | 18 +- examples/measurement/upload_run.py | 78 +++---- examples/measurement/upload_spm_run.ipynb | 252 ++++++++++++++++----- examples/measurement/workflow.py | 43 ---- 7 files changed, 410 insertions(+), 176 deletions(-) create mode 100644 examples/measurement/run_document.py delete mode 100644 examples/measurement/workflow.py diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index b2c84f48f..309c86f0f 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -3,15 +3,31 @@ Ad hoc parser for SOF-8050: it reads the tab-separated files NLR ships and nothing else. """ +import argparse from pathlib import Path -from workflow import build_workflow, unit_id +from run_document import serialize + + +def standata_workflow(application_name, workflow_name): + """The procedure the instrument runs, from the standata registry — the same entry the platform resolves a + job's workflow through. Building one here would be a second source of truth for something that already + has one; a new instrument is a new registry entry, not code.""" + from mat3ra.standata.workflows import WorkflowStandata + workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) + if workflow is None: + raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}: " + "add it to mat3ra/standata, or pin a release that has it") + return workflow + + +def unit_id(workflow): + """The execution unit a property of this workflow comes from.""" + return workflow["subworkflows"][0]["units"][0]["flowchartId"] NLR_FRAME = {"frame": "wafer", "units": "mm", "note": "x_mm, y_mm as delivered by NLR; corner and axes to be confirmed"} -XRF_APPLICATION = {"name": "xrf-mapper", "shortName": "xrf", "summary": "X-ray fluorescence mapper (film thickness and composition over a grid of positions)", - "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} -IV_APPLICATION = {"name": "probe-station", "shortName": "iv", "summary": "DC probe station (current through a pad over a bias sweep)", - "version": "1.0", "build": "Default", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} +XRF_APPLICATION = "xrf-mapper" # the standata applications whose workflows these two runs record +IV_APPLICATION = "probe-station" def read_columns(path): @@ -19,12 +35,6 @@ def read_columns(path): return [line.split("\t") for line in Path(path).read_text().splitlines()[1:] if line.strip()] -def nlr_workflow(application, executable_name, flavor_name, name, properties): - """NLR's two instruments are not in the standata registry yet, so their ids are scoped by application name.""" - return build_workflow(application, executable_name, flavor_name, name, properties, - id_prefix=f"{application['name']}/") - - def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): """NLR's delivery for one piece as platform documents: one Sample Set of the pads they measured, and one run per technique over those same pads — the XRF map, then the DC I-V sweep. Two runs, each in the shape parse() returns, @@ -38,8 +48,7 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): if f.suffix.lower() in (".jpg", ".jpeg", ".png")] sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} xrf_run_name = f"{run_name} XRF" - xrf_workflow = nlr_workflow(XRF_APPLICATION, "map", "xrf_grid", "XRF Grid Map", - ["thickness", "al_atomic_fraction", "sc_atomic_fraction"]) + xrf_workflow = standata_workflow(XRF_APPLICATION, "XRF Grid Map") xrf_unit_id = unit_id(xrf_workflow) grid = read_columns(grid_file) samples, xrf_measurements, xrf_properties = {}, {}, [] @@ -60,7 +69,7 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): (label, xrf_unit_id, {"name": "al_atomic_fraction", "value": float(aluminium_at_pct), "units": "at%"}, 0), (label, xrf_unit_id, {"name": "sc_atomic_fraction", "value": float(scandium_at_pct), "units": "at%"}, 0)] iv_run_name = f"{run_name} DC IV" - iv_workflow = nlr_workflow(IV_APPLICATION, "sweep", "dc_iv", "DC I-V Sweep", ["iv_curve"]) + iv_workflow = standata_workflow(IV_APPLICATION, "DC I-V Sweep") iv_unit_id = unit_id(iv_workflow) volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] @@ -89,3 +98,24 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): "measurements": iv_measurements, "files": {}, "set_files": [(volts_file.name, volts_file), (amps_file.name, amps_file)], "records": volts, "properties": iv_properties}] + + +def main(): + """Read NLR's delivery and write a run document per technique.""" + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("folder") + ap.add_argument("--physical-id", required=True, help="the identifier written on the physical piece, e.g. PDAC_COM5_01448") + ap.add_argument("--xrf-instrument", required=True, help="identity of the machine the grid was mapped on") + ap.add_argument("--iv-instrument", required=True, help="identity of the machine the sweep was measured on") + ap.add_argument("--out", default="parsed", help="directory for the run documents (default: parsed/)") + a = ap.parse_args() + + for parsed in parse_nlr(a.folder, a.physical_id, a.xrf_instrument, a.iv_instrument): + print(f"{parsed['physicalId']}: {len(parsed['samples'])} samples · run {parsed['run']}: " + f"{len(parsed['measurements'])} measurements · {len(parsed['records'])} rows -> " + f"{len(parsed['set_files'])} files · {len(parsed['images'])} image(s) · {len(parsed['properties'])} properties") + print("run document:", serialize(parsed, a.out, name=parsed["run"].replace(" ", "_") + ".json")) + + +if __name__ == "__main__": + main() diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py index 4b1a8534c..4e229b495 100644 --- a/examples/measurement/parse_utk.py +++ b/examples/measurement/parse_utk.py @@ -4,11 +4,29 @@ Ad hoc parser for SOF-8050: it reads the shape UTK's afm-lib writes and nothing else. """ -import ast, json, math, re, statistics, struct +import argparse, ast, json, math, re, statistics, struct from datetime import datetime, timezone from pathlib import Path -from workflow import build_workflow, unit_id +from run_document import serialize + + +def standata_workflow(application_name, workflow_name): + """The procedure the instrument runs, from the standata registry — the same entry the platform resolves a + job's workflow through. Building one here would be a second source of truth for something that already + has one; a new instrument is a new registry entry, not code.""" + from mat3ra.standata.workflows import WorkflowStandata + workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) + if workflow is None: + raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}: " + "add it to mat3ra/standata, or pin a release that has it") + return workflow + + +def unit_id(workflow): + """The execution unit a property of this workflow comes from.""" + return workflow["subworkflows"][0]["units"][0]["flowchartId"] + FIELD = {"off_field": "off", "on_field": "on"} @@ -22,8 +40,7 @@ "remnant_rising_m": ("remanentResponse", "rising"), "remnant_falling_m": ("remanentResponse", "falling"), } -INSTRUMENT = {"name": "asylum-spm", "shortName": "spm", "summary": "Asylum Research SPM driven by afm-lib (switching-spectroscopy PFM)", - "version": "1.0", "build": "afm-lib", "isUsingMaterial": False, "hasAdvancedComputeOptions": False} +INSTRUMENT_NAME = "asylum-spm" # the standata application whose workflow this run records g = lambda v: float(f"{v:.6g}") @@ -262,14 +279,12 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= for label, const in per_sample.items(): if label in samples: samples[label]["metadata"].update(const) - workflow = build_workflow(INSTRUMENT, "loop", "ss_pfm", "SS-PFM Hysteresis Loop", ["hysteresis_loop"], - unit_name="run_loop", tags=["experimental", "afm"], - metadata={"recipe": recipe, "loop_settings": recipe["per_site"][0]["loop_settings"], - "sites": list(samples)}) + workflow = standata_workflow(INSTRUMENT_NAME, "SS-PFM Hysteresis Loop") unit = unit_id(workflow) measurement_set = {"name": run_name, "entitySetType": "ordered", - "metadata": {"session": session, "recipe": recipe["name"], "context": recipe.get("context", ""), - "common": common, "registration": reg}} + "metadata": {"session": session, "recipe": recipe, "context": recipe.get("context", ""), + "loop_settings": recipe["per_site"][0]["loop_settings"], + "sites": list(samples), "common": common, "registration": reg}} # the setup block, Measurement : setup :: Job : compute — the machine and the sitting; the technique # (asylum-spm, SS-PFM) is the workflow's application. The run folder does not name the machine: --instrument does. started = session.get("started_ts") @@ -303,3 +318,32 @@ def same_axis(candidate, reference, tolerance=1e-9): """Whether two bias axes are the same sweep: equal length and equal voltages within tolerance.""" return len(candidate) == len(reference) and all( abs(a - b) <= tolerance + 1e-6 * abs(b) for a, b in zip(candidate, reference)) + + +def main(): + """Read a UTK run folder and write its run document.""" + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("run_dir") + ap.add_argument("--physical-id", required=True, help="the identifier written on the physical piece, e.g. PDAC_COM5_01448") + ap.add_argument("--out", default="parsed", help="directory for the run document and the records cut from the run (default: parsed/)") + ap.add_argument("--limit-records", type=int, help="trial: only the first N records and the samples they belong to") + ap.add_argument("--deposition", help="NLR HTEM record (json) kept in the run's sample set metadata") + ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") + ap.add_argument("--emit-example", help="write the property with the most loops to this path — the ESSE example") + a = ap.parse_args() + + parsed = parse(a.run_dir, a.physical_id, a.limit_records, a.deposition, a.instrument) + files = sum(len(v) for v in parsed["files"].values()) + print(f"{parsed['physicalId']}: {len(parsed['samples'])} samples · run {parsed['run']}: " + f"{len(parsed['measurements'])} measurements · {len(parsed['records'])} records -> {files} files · " + f"{len(parsed['properties'])} samples with a combined loop" + + (f" · no curves: {len(parsed['skipped'])} samples" if parsed["skipped"] else "")) + if a.emit_example and parsed["properties"]: + label, _, prop, _rep = max(parsed["properties"], key=lambda t: t[2]["parameters"]["off"].get("imprint", {}).get("count", 0)) + Path(a.emit_example).write_text(json.dumps(dict(prop, **thinned_curves(prop)), indent=4) + "\n") + print(f"example written from sample {label} -> {a.emit_example}") + print("run document:", serialize(parsed, a.out)) + + +if __name__ == "__main__": + main() diff --git a/examples/measurement/run_document.py b/examples/measurement/run_document.py new file mode 100644 index 000000000..0e5cfd852 --- /dev/null +++ b/examples/measurement/run_document.py @@ -0,0 +1,73 @@ +"""The run document: what a parser writes and the uploader takes. + +One JSON file per run, beside the files it names. Everything instrument-specific has already happened by the +time it exists — reading the lab's delivery is `parse_utk.py` / `parse_nlr.py`, uploading it is `upload_run.py`, +and neither knows anything about the other. + + { + "physicalId": "PDAC_COM5_01448", the identifier written on the physical piece + "run": "com5_1448_xrf_grid XRF", names the measurement set + "sample_set": {name, entitySetType, metadata}, + "samples": {label: sample document}, + "measurement_set": {name, entitySetType, metadata}, + "measurements": {label: measurement document}, _sample is filled in at upload + "properties": [[label, unitId, property document, repetition]], + "set_files": [[name, path]], the set's own files: a grid, a photograph of the piece + "files": {label: [[name, path]]}, one measurement's files; the name's first segment is its group + "skipped": [label] informational: samples the run produced no property for + } + +Paths are relative to the document, so a run folder moves as a whole. A parser that derives a file (a record +JSON it cut from a larger one) writes it here too — `serialize` takes text in place of a path and stores it. +""" +import json, os +from pathlib import Path + + +def serialize(parsed, out_dir, name="run.json"): + """The parsed run as a document on disk. Text payloads are written out; Paths are recorded relative to it.""" + out_dir = Path(out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + + def store(prefix, entries): + out = [] + for file_name, payload in entries: + if isinstance(payload, Path): + out.append([file_name, relative(payload, out_dir)]) + else: # derived text: this document owns it + target = out_dir / "files" / prefix / file_name + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(payload) + out.append([file_name, relative(target, out_dir)]) + return out + + document = {k: v for k, v in parsed.items() if k not in ("files", "set_files", "images", "measurements")} + document["measurements"] = {label: {k: v for k, v in m.items() if k != "_records"} + for label, m in parsed["measurements"].items()} + document["records_by_sample"] = {label: m["_records"] for label, m in parsed["measurements"].items() if m.get("_records")} + document["set_files"] = store("set", list(parsed.get("set_files", [])) + list(parsed.get("images", []))) + document["files"] = {label: store(label, entries) for label, entries in parsed.get("files", {}).items()} + document["properties"] = [list(p) for p in parsed.get("properties", [])] + path = out_dir / name + path.write_text(json.dumps(document, indent=1) + "\n") + return path + + +def relative(path, out_dir): + """`path` as the document will name it: relative when it can be, absolute when it lives elsewhere.""" + path, out_dir = Path(path).resolve(), Path(out_dir).resolve() + try: + return path.relative_to(out_dir).as_posix() + except ValueError: + return os.path.relpath(path, out_dir) + + +def load(path): + """A run document with its file paths resolved against its own location.""" + path = Path(path) + document = json.loads(path.read_text()) + resolve = lambda entries: [(name, (path.parent / file_path).resolve()) for name, file_path in entries] + document["set_files"] = resolve(document.get("set_files", [])) + document["files"] = {label: resolve(entries) for label, entries in document.get("files", {}).items()} + document["properties"] = [tuple(p) for p in document.get("properties", [])] + return document diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index f04098152..fdf0b454f 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -57,7 +57,7 @@ "XRF_INSTRUMENT = \"\"\n", "IV_INSTRUMENT = \"\"\n", "ACCOUNT_SLUG = \"\"\n", - "FILES = \"records\" # \"records\": the delivered tables and photographs, \"none\": no files\n", + "FILES = [\"records\"] # the delivered tables and photographs; [] uploads none\n", "\n", "url = urllib.parse.urlsplit(HOST)\n", "address = {\n", @@ -128,7 +128,9 @@ "source": [ "from pathlib import Path\n", "\n", - "from upload_run import account_id, parse_nlr, upload" + "from parse_nlr import parse_nlr\n", + "from run_document import load, serialize\n", + "from upload_run import account_id, upload" ] }, { @@ -146,12 +148,16 @@ "metadata": {}, "outputs": [], "source": [ - "runs = parse_nlr(Path(DATA_DIR), PHYSICAL_ID, XRF_INSTRUMENT, IV_INSTRUMENT)\n", - "for run in runs:\n", + "# reading the delivery and writing one run document per technique: nothing here talks to the platform\n", + "runs = []\n", + "for parsed in parse_nlr(Path(DATA_DIR), PHYSICAL_ID, XRF_INSTRUMENT, IV_INSTRUMENT):\n", + " document_path = serialize(parsed, \"parsed\", name=parsed[\"run\"].replace(\" \", \"_\") + \".json\")\n", + " run = load(document_path)\n", + " runs.append(run)\n", " print(\n", " f\"{run['physicalId']}: {len(run['samples'])} samples (ordered set) · run {run['run']}: \"\n", - " f\"{len(run['measurements'])} measurements (ordered set, one per sample) · {len(run['set_files'])} files · \"\n", - " f\"{len(run['images'])} image(s) · {len(run['properties'])} properties\"\n", + " f\"{len(run['measurements'])} measurements (ordered set, one per sample) · \"\n", + " f\"{len(run['set_files'])} files · {len(run['properties'])} properties · {document_path}\"\n", " )" ] }, diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index a8b1f6708..8a650b07a 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -1,26 +1,26 @@ #!/usr/bin/env python3 -"""Parsed runs -> platform documents, validated against the ESSE schemas, then uploaded through the REST API. +"""Run documents -> the platform: the sample set, its samples, the measurement set, one measurement per sample, +the files and the properties, validated against the ESSE schemas before anything is sent. - upload_run.py --physical-id --account # upload the run - upload_run.py --physical-id --dry-run [--emit-example out.json] # parse + validate only; write one property as the ESSE example - upload_run.py --physical-id --nlr # NLR's XRF grid and DC I-V sweep instead of a UTK run + upload_run.py parsed/run.json --account # upload it + upload_run.py parsed/*.json --account --files records loops + upload_run.py parsed/run.json --dry-run # validate only -The instrument-specific reading lives beside this file — `parse_utk.py` for a UTK SS-PFM run, `parse_nlr.py` for -NLR's delivery — and each returns the same dict, so everything below is the same for both. A third instrument is -a third parser, not a change here. +It knows nothing about any instrument. Reading a lab's delivery into a run document is a parser's job — +`parse_utk.py` for a UTK SS-PFM run, `parse_nlr.py` for NLR's delivery — and `run_document.py` states the +shape they agree on. A new lab is a new parser; nothing here changes. -Requires Python 3.9+ and `pip install mat3ra-api-client`, which talks to the platform and takes OIDC_ACCESS_TOKEN, or -ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. Optional: -`pip install mat3ra-esse` (tested with 2026.8.27-0) turns on schema validation before anything is uploaded. +Requires Python 3.9+ and `pip install mat3ra-api-client`, which talks to the platform and takes OIDC_ACCESS_TOKEN, +or ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. +Optional: `pip install mat3ra-esse` turns on schema validation before anything is uploaded. """ -import argparse, concurrent.futures, json, os, sys, threading, time, urllib.parse +import argparse, concurrent.futures, os, sys, threading, time, urllib.parse from pathlib import Path import requests from mat3ra.api_client import APIClient -from parse_nlr import parse_nlr -from parse_utk import parse, thinned_curves +from run_document import load try: # optional: schema validation before anything is sent from mat3ra.esse import ESSE @@ -177,7 +177,6 @@ def run_files(parsed, groups=("records",)): if unknown: raise SystemExit(f"unknown file group(s): {', '.join(sorted(unknown))}; choose from {', '.join(FILE_GROUPS)}") files = [(f"set/{name}", payload) for name, payload in parsed["set_files"]] - files += [(f"set/{name}", path) for name, path in parsed["images"]] for label, file_list in parsed["files"].items(): files += [(f"{label}/{name}", payload) for name, payload in file_list if ("loops" if name.startswith("loops/") else "records") in groups] @@ -221,10 +220,11 @@ def upload(client, parsed, files=("records",)): if measurement_doc["name"] in existing: measurement_ids[label] = existing[measurement_doc["name"]]["_id"] continue - body = {k: v for k, v in measurement_doc.items() if k != "_records"} + body = dict(measurement_doc) body["_sample"] = {"_id": sample_ids[label], "cls": "Sample"} - if not uploads: # no file store: keep the raw records in the measurement's metadata - body["metadata"] = dict(body["metadata"], records=measurement_doc["_records"]) + records = parsed.get("records_by_sample", {}).get(label) + if not uploads and records: # no file store: keep the raw records in the measurement's metadata + body["metadata"] = dict(body["metadata"], records=records) doc = client.measurements.create(dict(body, owner=owner)) client.measurements.move_to_set(doc["_id"], None, measurement_set["_id"]) measurement_ids[label] = doc["_id"] @@ -251,55 +251,31 @@ def upload(client, parsed, files=("records",)): def main(): - """Command line: parse, validate, upload.""" + """Command line: validate the run documents, then upload them.""" ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) - ap.add_argument("run_dir") - ap.add_argument("--physical-id", required=True, help="the identifier written on the physical piece the samples are part of, e.g. PDAC_COM5_01448") - ap.add_argument("--dry-run", action="store_true") - ap.add_argument("--emit-example", help="write the first combined loop property (most loops) to this path — the ESSE example") + ap.add_argument("documents", nargs="+", metavar="RUN.JSON", help="run documents, as a parser writes them") ap.add_argument("--host", default=os.environ.get("MAT3RA_HOST", "localhost:3000"), help="web app host or URL (or MAT3RA_HOST); https unless localhost, e.g. dev.mat3ra.com") ap.add_argument("--account", help="slug of the account the data belongs to (reads scoped to it, writes owned by it)") ap.add_argument("--files", nargs="*", default=["records"], metavar="GROUP", help=f"which file groups to upload ({', '.join(FILE_GROUPS)}); pass --files with no value " "to upload none and keep the raw records in each measurement's metadata") - ap.add_argument("--limit-records", type=int, help="trial: only the first N records and the samples they belong to") - ap.add_argument("--deposition", help="NLR HTEM record (json) kept in the run's sample set metadata") - ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") - ap.add_argument("--nlr", nargs=2, metavar=("XRF_INSTRUMENT", "IV_INSTRUMENT"), - help="the folder holds NLR's delivery — an XRF grid and a DC I-V sweep over the same pads, measured on these two machines — not a UTK run") - a = ap.parse_intermixed_args() - if a.nlr: - runs = parse_nlr(a.run_dir, a.physical_id, *a.nlr) - for p in runs: - print(f"{p['physicalId']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} " - f"measurements (ordered set, one per sample) · {len(p['records'])} rows -> {len(p['set_files'])} files · " - f"{len(p['images'])} image(s) · {len(p['properties'])} properties") - else: - runs = [parse(a.run_dir, a.physical_id, a.limit_records, a.deposition, a.instrument)] - p = runs[0] - nfiles = sum(len(v) for v in p["files"].values()) - print(f"{p['physicalId']}: {len(p['samples'])} samples (ordered set) · run {p['run']}: {len(p['measurements'])} measurements " - f"(ordered set, one per sample) · {len(p['records'])} records -> {nfiles} files · {len(p['images'])} image(s) · " - f"{len(p['properties'])} samples with a combined loop" + (f" · no curves: {len(p['skipped'])} samples" if p["skipped"] else "")) - for label, _, prop, _rep in p["properties"]: - n = prop["parameters"]["off"].get("imprint", {}).get("count") - print(f" {label}: {n} loops combined, imprint off = {prop['parameters']['off'].get('imprint', {}).get('value')} V") - if a.emit_example and p["properties"]: - label, _, prop, _rep = max(p["properties"], key=lambda t: t[2]["parameters"]["off"].get("imprint", {}).get("count", 0)) - prop = dict(prop, **thinned_curves(prop)) - Path(a.emit_example).write_text(json.dumps(prop, indent=4) + "\n"); print(f"example written from sample {label} -> {a.emit_example}") - errors = sum(validate(p) for p in runs) + ap.add_argument("--dry-run", action="store_true", help="validate the documents and stop") + a = ap.parse_args() + + runs = [load(path) for path in a.documents] + errors = sum(validate(run) for run in runs) print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") if errors or a.dry_run: sys.exit(1 if errors else 0) + url = urllib.parse.urlsplit(base_url(a.host)) address = {"host": url.hostname, "port": url.port or (443 if url.scheme == "https" else 80), "secure": url.scheme == "https"} client = APIClient.authenticate(**address) if a.account: client = APIClient.authenticate(account_id=account_id(client, a.account), **address) - for p in runs: - upload(client, p, files=a.files) + for run in runs: + upload(client, run, files=a.files) if __name__ == "__main__": diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index b4e9c8a44..7e5fef211 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -21,12 +21,35 @@ }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:43.416714Z", + "start_time": "2026-09-22T18:26:42.503987Z" + } + }, "source": [ "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" - ] + ], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + " \u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\r\n", + " \r\n", + " \u001b[31m×\u001b[0m \u001b[32mgit version\u001b[0m did not run successfully.\r\n", + " \u001b[31m│\u001b[0m exit code: \u001b[1;36m1\u001b[0m\r\n", + " \u001b[31m╰─>\u001b[0m \u001b[31m[2 lines of output]\u001b[0m\r\n", + " \u001b[31m \u001b[0m xcrun: error: invalid active developer path (/Library/Developer/CommandLineTools), missing xcrun at: /Library/Developer/CommandLineTools/usr/bin/xcrun\r\n", + " \u001b[31m \u001b[0m \u001b[31m[end of output]\u001b[0m\r\n", + " \r\n", + " \u001b[1;35mnote\u001b[0m: This error originates from a subprocess, and is likely not a problem with pip.\r\n", + "\u001b[31mERROR: Failed to build 'git+https://github.com/mat3ra/api-client.git@feature/SOF-8051' when git version\u001b[0m\u001b[31m\r\n", + "\u001b[0mNote: you may need to restart the kernel to use updated packages.\n" + ] + } + ], + "execution_count": 14 }, { "cell_type": "markdown", @@ -43,17 +66,20 @@ }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:43.434671Z", + "start_time": "2026-09-22T18:26:43.426063Z" + } + }, "source": [ "import urllib.parse\n", "\n", "HOST = \"https://alphafilm.mat3ra.com\"\n", - "RUN_DIR = \"run\"\n", - "PHYSICAL_ID = \"\"\n", - "ACCOUNT_SLUG = \"\"\n", - "FILES = \"records\" # \"records\": the record JSONs, \"all\": also the loop arrays and plots, \"none\": no files\n", + "RUN_DIR = \"/Users/mat3ra/code/work/SOF-8050/data/From_UTK\"\n", + "PHYSICAL_ID = \"test-01448\"\n", + "ACCOUNT_SLUG = \"demo\"\n", + "FILES = [\"records\", \"loops\"] # \"records\": the record JSONs, \"loops\": the loop arrays and plots; [] uploads none\n", "\n", "url = urllib.parse.urlsplit(HOST)\n", "address = {\n", @@ -61,7 +87,9 @@ " \"port\": url.port or (443 if url.scheme == \"https\" else 80),\n", " \"secure\": url.scheme == \"https\",\n", "}" - ] + ], + "outputs": [], + "execution_count": 15 }, { "cell_type": "markdown", @@ -78,36 +106,107 @@ }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:43.449043Z", + "start_time": "2026-09-22T18:26:43.435732Z" + } + }, "source": [ "from mat3ra.notebooks_utils.packages import install_packages\n", "\n", "await install_packages(\"api\")" - ] + ], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "To install packages, run `pip install \".[all]\"` in the terminal\n" + ] + } + ], + "execution_count": 16 }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:48.827050Z", + "start_time": "2026-09-22T18:26:43.455461Z" + } + }, "source": [ "from mat3ra.notebooks_utils.auth import authenticate\n", "\n", + "import os\n", + "\n", + "os.environ[\"API_HOST\"] = address[\"host\"]\n", + "os.environ[\"API_PORT\"] = str(address[\"port\"])\n", + "os.environ[\"API_SECURE\"] = str(address[\"secure\"])\n", + "os.environ[\"ACCOUNT_ID\"] = \"pR5gsAQJrJYJuapM3\" # from Preferences\n", + "os.environ[\"AUTH_TOKEN\"] = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\" # the API token\n", + "\n", + "\n", "await authenticate()" - ] + ], + "outputs": [ + { + "data": { + "text/plain": [ + "" + ], + "text/html": [ + "
Authentication Required
Enter this code: HZDR-QXGM
Click here if the browser tab did not open automatically
" + ] + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "data": { + "text/plain": [ + "" + ], + "application/javascript": "window.open('https://alphafilm.mat3ra.com/oidc/device?user_code=HZDR-QXGM', '_blank');" + }, + "metadata": {}, + "output_type": "display_data" + }, + { + "ename": "CancelledError", + "evalue": "", + "output_type": "error", + "traceback": [ + "\u001b[31m---------------------------------------------------------------------------\u001b[39m", + "\u001b[31mCancelledError\u001b[39m Traceback (most recent call last)", + "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[17]\u001b[39m\u001b[32m, line 12\u001b[39m\n\u001b[32m 8\u001b[39m os.environ[\u001b[33m\"ACCOUNT_ID\"\u001b[39m] = \u001b[33m\"pR5gsAQJrJYJuapM3\"\u001b[39m \u001b[38;5;66;03m# from Preferences\u001b[39;00m\n\u001b[32m 9\u001b[39m os.environ[\u001b[33m\"AUTH_TOKEN\"\u001b[39m] = \u001b[33m\"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\"\u001b[39m \u001b[38;5;66;03m# the API token\u001b[39;00m\n\u001b[32m 10\u001b[39m \n\u001b[32m 11\u001b[39m \n\u001b[32m---> \u001b[39m\u001b[32m12\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m authenticate()\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/auth.py:65\u001b[39m, in \u001b[36mauthenticate\u001b[39m\u001b[34m(force, globals_dict)\u001b[39m\n\u001b[32m 63\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m authenticate_jupyterlite(data_from_host)\n\u001b[32m 64\u001b[39m \u001b[38;5;28;01melif\u001b[39;00m ACCESS_TOKEN_ENV_VAR \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m os.environ \u001b[38;5;129;01mor\u001b[39;00m force:\n\u001b[32m---> \u001b[39m\u001b[32m65\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m _authenticate_oidc_with_cache(force)\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/auth.py:36\u001b[39m, in \u001b[36m_authenticate_oidc_with_cache\u001b[39m\u001b[34m(force)\u001b[39m\n\u001b[32m 33\u001b[39m store_token_data_in_environment(cached)\n\u001b[32m 34\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m\n\u001b[32m---> \u001b[39m\u001b[32m36\u001b[39m token_data = \u001b[38;5;28;01mawait\u001b[39;00m authenticate_oidc(show_popup=show_device_flow_popup)\n\u001b[32m 37\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m save_token(oidc_url, token_data)\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/core/api/auth.py:92\u001b[39m, in \u001b[36mauthenticate_oidc\u001b[39m\u001b[34m(oidc_base_url, client_id, scope, show_popup)\u001b[39m\n\u001b[32m 90\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m show_popup \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m:\n\u001b[32m 91\u001b[39m show_popup(device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mverification_uri_complete\u001b[39m\u001b[33m\"\u001b[39m], device_flow_state[\u001b[33m\"\u001b[39m\u001b[33muser_code\u001b[39m\u001b[33m\"\u001b[39m])\n\u001b[32m---> \u001b[39m\u001b[32m92\u001b[39m token_data = \u001b[38;5;28;01mawait\u001b[39;00m _poll_for_token_data(\n\u001b[32m 93\u001b[39m oidc_base_url=oidc_base_url,\n\u001b[32m 94\u001b[39m client_id=client_id,\n\u001b[32m 95\u001b[39m device_code=device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mdevice_code\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 96\u001b[39m polling_interval_seconds=device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mpolling_interval_seconds\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 97\u001b[39m expires_in_seconds=device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mexpires_in_seconds\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 98\u001b[39m )\n\u001b[32m 99\u001b[39m store_token_data_in_environment(token_data)\n\u001b[32m 100\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m token_data\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/core/api/auth.py:77\u001b[39m, in \u001b[36m_poll_for_token_data\u001b[39m\u001b[34m(oidc_base_url, client_id, device_code, polling_interval_seconds, expires_in_seconds)\u001b[39m\n\u001b[32m 75\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m token_response.status_code == \u001b[32m200\u001b[39m:\n\u001b[32m 76\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m token_response.json()\n\u001b[32m---> \u001b[39m\u001b[32m77\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m asyncio.sleep(polling_interval_seconds)\n\u001b[32m 78\u001b[39m \u001b[38;5;28;01mraise\u001b[39;00m \u001b[38;5;167;01mException\u001b[39;00m(\u001b[33m\"\u001b[39m\u001b[33mTimeout waiting for authorization.\u001b[39m\u001b[33m\"\u001b[39m)\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/.pyenv/versions/3.11.2/lib/python3.11/asyncio/tasks.py:639\u001b[39m, in \u001b[36msleep\u001b[39m\u001b[34m(delay, result)\u001b[39m\n\u001b[32m 635\u001b[39m h = loop.call_later(delay,\n\u001b[32m 636\u001b[39m futures._set_result_unless_cancelled,\n\u001b[32m 637\u001b[39m future, result)\n\u001b[32m 638\u001b[39m \u001b[38;5;28;01mtry\u001b[39;00m:\n\u001b[32m--> \u001b[39m\u001b[32m639\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28;01mawait\u001b[39;00m future\n\u001b[32m 640\u001b[39m \u001b[38;5;28;01mfinally\u001b[39;00m:\n\u001b[32m 641\u001b[39m h.cancel()\n", + "\u001b[31mCancelledError\u001b[39m: " + ] + } + ], + "execution_count": 17 }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:52.466952Z", + "start_time": "2026-09-22T18:26:52.442997Z" + } + }, "source": [ "from mat3ra.api_client import APIClient\n", "\n", "client = APIClient.authenticate(**address)" - ] + ], + "outputs": [], + "execution_count": 18 }, { "cell_type": "markdown", @@ -118,14 +217,21 @@ }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:52.908678Z", + "start_time": "2026-09-22T18:26:52.888832Z" + } + }, "source": [ "from pathlib import Path\n", "\n", - "from upload_run import account_id, parse, upload" - ] + "from parse_utk import parse\n", + "from run_document import load, serialize\n", + "from upload_run import account_id, upload" + ], + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -138,19 +244,28 @@ }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:54.188891Z", + "start_time": "2026-09-22T18:26:53.243339Z" + } + }, "source": [ + "# reading the run folder and writing the run document: nothing here talks to the platform\n", "parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n", - "file_count = sum(len(files) for files in parsed[\"files\"].values())\n", + "document_path = serialize(parsed, \"parsed\")\n", + "run = load(document_path)\n", + "file_count = sum(len(files) for files in run[\"files\"].values())\n", "print(\n", - " f\"{parsed['physicalId']}: {len(parsed['samples'])} samples (ordered set) · run {parsed['run']}: \"\n", - " f\"{len(parsed['measurements'])} measurements (ordered set, one per sample) · {len(parsed['records'])} records \"\n", - " f\"-> {file_count} files · {len(parsed['images'])} image(s) · {len(parsed['properties'])} samples with a combined \"\n", - " f\"loop · no curves: {len(parsed['skipped'])} samples\"\n", - ")" - ] + " f\"{run['physicalId']}: {len(run['samples'])} samples (ordered set) · run {run['run']}: \"\n", + " f\"{len(run['measurements'])} measurements (ordered set, one per sample) · {len(run['set_files'])} set files \"\n", + " f\"-> {file_count} measurement files · {len(run['properties'])} samples with a combined loop \"\n", + " f\"· no curves: {len(run['skipped'])} samples\"\n", + ")\n", + "print(\"run document:\", document_path)" + ], + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -163,13 +278,34 @@ }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:54.254975Z", + "start_time": "2026-09-22T18:26:54.202446Z" + } + }, "source": [ "if ACCOUNT_SLUG:\n", " client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG), **address)" - ] + ], + "outputs": [ + { + "ename": "ValueError", + "evalue": "Access token is required to fetch user data", + "output_type": "error", + "traceback": [ + "\u001b[31m---------------------------------------------------------------------------\u001b[39m", + "\u001b[31mValueError\u001b[39m Traceback (most recent call last)", + "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[21]\u001b[39m\u001b[32m, line 2\u001b[39m\n\u001b[32m 1\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m ACCOUNT_SLUG:\n\u001b[32m----> \u001b[39m\u001b[32m2\u001b[39m client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG), **address)\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/upload_run.py:522\u001b[39m, in \u001b[36maccount_id\u001b[39m\u001b[34m(client, slug_or_name)\u001b[39m\n\u001b[32m 520\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34maccount_id\u001b[39m(client, slug_or_name):\n\u001b[32m 521\u001b[39m \u001b[38;5;250m \u001b[39m\u001b[33;03m\"\"\"`--account ` → the id of that account: the slug the platform shows it under, else its display name.\"\"\"\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m522\u001b[39m accounts = \u001b[30;43mclient\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mlist_accounts\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 523\u001b[39m \u001b[38;5;28;01mfor\u001b[39;00m field \u001b[38;5;129;01min\u001b[39;00m (\u001b[33m\"\u001b[39m\u001b[33mslug\u001b[39m\u001b[33m\"\u001b[39m, \u001b[33m\"\u001b[39m\u001b[33mname\u001b[39m\u001b[33m\"\u001b[39m):\n\u001b[32m 524\u001b[39m account = \u001b[38;5;28mnext\u001b[39m((a \u001b[38;5;28;01mfor\u001b[39;00m a \u001b[38;5;129;01min\u001b[39;00m accounts \u001b[38;5;28;01mif\u001b[39;00m a.get(field) == slug_or_name), \u001b[38;5;28;01mNone\u001b[39;00m)\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/mat3ra/api_client/client.py:150\u001b[39m, in \u001b[36mAPIClient.list_accounts\u001b[39m\u001b[34m(self)\u001b[39m\n\u001b[32m 149\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34mlist_accounts\u001b[39m(\u001b[38;5;28mself\u001b[39m) -> List[\u001b[38;5;28mdict\u001b[39m]:\n\u001b[32m--> \u001b[39m\u001b[32m150\u001b[39m accounts = \u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43m_fetch_user_accounts\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 151\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m [\n\u001b[32m 152\u001b[39m {\n\u001b[32m 153\u001b[39m \u001b[33m\"\u001b[39m\u001b[33m_id\u001b[39m\u001b[33m\"\u001b[39m: account[\u001b[33m\"\u001b[39m\u001b[33mentity\u001b[39m\u001b[33m\"\u001b[39m][\u001b[33m\"\u001b[39m\u001b[33m_id\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m (...)\u001b[39m\u001b[32m 159\u001b[39m \u001b[38;5;28;01mfor\u001b[39;00m account \u001b[38;5;129;01min\u001b[39;00m accounts\n\u001b[32m 160\u001b[39m ]\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/mat3ra/api_client/client.py:147\u001b[39m, in \u001b[36mAPIClient._fetch_user_accounts\u001b[39m\u001b[34m(self)\u001b[39m\n\u001b[32m 146\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34m_fetch_user_accounts\u001b[39m(\u001b[38;5;28mself\u001b[39m) -> List[\u001b[38;5;28mdict\u001b[39m]:\n\u001b[32m--> \u001b[39m\u001b[32m147\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43m_fetch_data\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m)\u001b[39;49m.get(\u001b[33m\"\u001b[39m\u001b[33maccounts\u001b[39m\u001b[33m\"\u001b[39m, [])\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/mat3ra/api_client/client.py:139\u001b[39m, in \u001b[36mAPIClient._fetch_data\u001b[39m\u001b[34m(self)\u001b[39m\n\u001b[32m 137\u001b[39m access_token = \u001b[38;5;28mself\u001b[39m.auth.access_token \u001b[38;5;129;01mor\u001b[39;00m os.environ.get(ACCESS_TOKEN_ENV_VAR)\n\u001b[32m 138\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m access_token:\n\u001b[32m--> \u001b[39m\u001b[32m139\u001b[39m \u001b[38;5;28;01mraise\u001b[39;00m \u001b[38;5;167;01mValueError\u001b[39;00m(\u001b[33m\"\u001b[39m\u001b[33mAccess token is required to fetch user data\u001b[39m\u001b[33m\"\u001b[39m)\n\u001b[32m 141\u001b[39m url = _build_base_url(\u001b[38;5;28mself\u001b[39m.host, \u001b[38;5;28mself\u001b[39m.port, \u001b[38;5;28mself\u001b[39m.secure, \u001b[33m\"\u001b[39m\u001b[33m/api/v1/users/me\u001b[39m\u001b[33m\"\u001b[39m)\n\u001b[32m 142\u001b[39m response = requests.get(url, headers={\u001b[33m\"\u001b[39m\u001b[33mAuthorization\u001b[39m\u001b[33m\"\u001b[39m: \u001b[33mf\u001b[39m\u001b[33m\"\u001b[39m\u001b[33mBearer \u001b[39m\u001b[38;5;132;01m{\u001b[39;00maccess_token\u001b[38;5;132;01m}\u001b[39;00m\u001b[33m\"\u001b[39m}, timeout=\u001b[32m30\u001b[39m)\n", + "\u001b[31mValueError\u001b[39m: Access token is required to fetch user data" + ] + } + ], + "execution_count": 21 }, { "cell_type": "markdown", @@ -182,12 +318,12 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ - "upload(client, parsed, files=FILES)" - ] + "upload(client, run, files=FILES)" + ], + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -200,12 +336,17 @@ }, { "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T18:26:54.343400Z", + "start_time": "2026-09-22T18:26:54.305317Z" + } + }, "source": [ - "print(f\"Open {HOST}, your account's Measurements tab: {parsed['run']}\")" - ] + "print(f\"Open {HOST}, your account's Measurements tab: {run['run']}\")" + ], + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -215,6 +356,13 @@ "\n", "- [Mat3ra REST API](https://docs.mat3ra.com/rest-api/overview/)" ] + }, + { + "metadata": {}, + "cell_type": "code", + "outputs": [], + "execution_count": null, + "source": "" } ], "metadata": { diff --git a/examples/measurement/workflow.py b/examples/measurement/workflow.py deleted file mode 100644 index f9def668c..000000000 --- a/examples/measurement/workflow.py +++ /dev/null @@ -1,43 +0,0 @@ -"""The measurement workflow a parser attaches to its measurements. - -An experiment's workflow is a description, not a plan the platform executes: ONE workflow, one subworkflow, one -execution unit that declares what the instrument produces. Application, executable and flavor name the standata -registry entries, so the platform resolves the unit exactly as it does a job's, and the ids are stable (uuid5 of -the names) so a property's `source.info.unitId` means the same thing across uploads. -""" -import uuid - -WORKFLOW_NAMESPACE = uuid.UUID("6f3b0b0e-8c1e-4b7a-9f21-3a5f0e2d1c44") -# ESSE requires a model on every subworkflow; an experiment has none, so the legacy "unknown" model. -UNKNOWN_MODEL = {"type": "unknown", "subtype": "unknown", "method": {"type": "unknown", "subtype": "unknown"}} - - -def build_workflow(application, executable_name, flavor_name, name, properties, - unit_name=None, tags=("experimental",), metadata=None, id_prefix=""): - """`properties` are what the unit declares it produces. `unit_name` defaults to the executable's; - `id_prefix` scopes the stable ids, so two instruments may share an executable name.""" - unit_name = unit_name or executable_name - results = [{"name": property_name} for property_name in properties] - monitors = [{"name": "standard_output"}] - stable = lambda key, length: uuid.uuid5(WORKFLOW_NAMESPACE, f"{id_prefix}{key}").hex[:length] - executable = {"name": executable_name, "applicationName": application["name"], "applicationVersion": "*", "isDefault": True, - "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} - flavor = {"name": flavor_name, "executableName": executable_name, "applicationName": application["name"], "applicationVersion": "*", - "isDefault": True, "input": [], "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} - unit = {"type": "execution", "name": unit_name, "flowchartId": stable(unit_name, 24), "head": True, "status": "finished", - "application": application, "executable": executable, "flavor": flavor, "input": [], "context": [], - "monitors": monitors, "results": results, "preProcessors": [], "postProcessors": []} - subworkflow_id = stable(flavor_name, 17) - subworkflow = {"_id": subworkflow_id, "name": flavor_name, "application": application, "model": UNKNOWN_MODEL, - "properties": list(properties), "units": [unit]} - # A subworkflow unit carries the subworkflow's own `_id` — that is how the platform pairs them. - subworkflow_unit = {"_id": subworkflow_id, "type": "subworkflow", "name": flavor_name, "flowchartId": stable(f"{flavor_name}/unit", 24), - "head": True, "status": "finished", "preProcessors": [], "postProcessors": [], "monitors": [], "results": []} - workflow = {"name": name, "isDefault": False, "tags": list(tags), "properties": list(properties), "application": application, - "subworkflows": [subworkflow], "units": [subworkflow_unit], "workflows": []} - return dict(workflow, metadata=metadata) if metadata is not None else workflow - - -def unit_id(workflow): - """The execution unit the properties of this workflow come from.""" - return workflow["subworkflows"][0]["units"][0]["flowchartId"] From 54d86cc576a804c23c2d2c4b11dd38242c3a5df2 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 13:11:55 -0700 Subject: [PATCH 19/36] fix(SOF-8051): pin the standata branch the instruments are registered on MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `mat3ra-notebooks-utils[all]` does install mat3ra-standata, but the PyPI release (2026.8.1) predates asylum-spm, xrf-mapper and probe-station — they exist only on feature/SOF-8051, so the parsers' workflow lookup found nothing. Both notebooks now pin that branch beside api-client's, and each parser says so in its header. Verified in a clean venv: a branch install resolves all three workflows. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/parse_nlr.py | 3 + examples/measurement/parse_utk.py | 3 + examples/measurement/upload_nlr_data.ipynb | 2 +- examples/measurement/upload_spm_run.ipynb | 141 +++++++-------------- 4 files changed, 52 insertions(+), 97 deletions(-) diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index 309c86f0f..d42620068 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -2,6 +2,9 @@ technique over those same pads — the XRF map, then the DC I-V sweep. Ad hoc parser for SOF-8050: it reads the tab-separated files NLR ships and nothing else. + +Requires `pip install "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"` until that branch is released: +the instruments' registry entries are on it, and the PyPI release predates them. """ import argparse from pathlib import Path diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py index 4e229b495..e15e19aed 100644 --- a/examples/measurement/parse_utk.py +++ b/examples/measurement/parse_utk.py @@ -3,6 +3,9 @@ deviation, count) inside it. Individual loops stay in the measurement's files. Ad hoc parser for SOF-8050: it reads the shape UTK's afm-lib writes and nothing else. + +Requires `pip install "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"` until that branch is released: +the instruments' registry entries are on it, and the PyPI release predates them. """ import argparse, ast, json, math, re, statistics, struct from datetime import datetime, timezone diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index fdf0b454f..f5db5e51e 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -25,7 +25,7 @@ "metadata": {}, "outputs": [], "source": [ - "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" + "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\" \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"" ] }, { diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 7e5fef211..e76aeaec4 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -23,12 +23,12 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T18:26:43.416714Z", - "start_time": "2026-09-22T18:26:42.503987Z" + "end_time": "2026-09-22T20:09:30.844346Z", + "start_time": "2026-09-22T20:09:30.085762Z" } }, "source": [ - "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\"" + "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\" \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"" ], "outputs": [ { @@ -49,7 +49,7 @@ ] } ], - "execution_count": 14 + "execution_count": 30 }, { "cell_type": "markdown", @@ -68,8 +68,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T18:26:43.434671Z", - "start_time": "2026-09-22T18:26:43.426063Z" + "end_time": "2026-09-22T20:09:30.856023Z", + "start_time": "2026-09-22T20:09:30.845688Z" } }, "source": [ @@ -89,7 +89,7 @@ "}" ], "outputs": [], - "execution_count": 15 + "execution_count": 31 }, { "cell_type": "markdown", @@ -108,8 +108,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T18:26:43.449043Z", - "start_time": "2026-09-22T18:26:43.435732Z" + "end_time": "2026-09-22T20:09:30.868383Z", + "start_time": "2026-09-22T20:09:30.857101Z" } }, "source": [ @@ -126,14 +126,14 @@ ] } ], - "execution_count": 16 + "execution_count": 32 }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T18:26:48.827050Z", - "start_time": "2026-09-22T18:26:43.455461Z" + "end_time": "2026-09-22T20:09:30.881225Z", + "start_time": "2026-09-22T20:09:30.869178Z" } }, "source": [ @@ -148,56 +148,17 @@ "os.environ[\"AUTH_TOKEN\"] = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\" # the API token\n", "\n", "\n", - "await authenticate()" + "# await authenticate()" ], - "outputs": [ - { - "data": { - "text/plain": [ - "" - ], - "text/html": [ - "
Authentication Required
Enter this code: HZDR-QXGM
Click here if the browser tab did not open automatically
" - ] - }, - "metadata": {}, - "output_type": "display_data" - }, - { - "data": { - "text/plain": [ - "" - ], - "application/javascript": "window.open('https://alphafilm.mat3ra.com/oidc/device?user_code=HZDR-QXGM', '_blank');" - }, - "metadata": {}, - "output_type": "display_data" - }, - { - "ename": "CancelledError", - "evalue": "", - "output_type": "error", - "traceback": [ - "\u001b[31m---------------------------------------------------------------------------\u001b[39m", - "\u001b[31mCancelledError\u001b[39m Traceback (most recent call last)", - "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[17]\u001b[39m\u001b[32m, line 12\u001b[39m\n\u001b[32m 8\u001b[39m os.environ[\u001b[33m\"ACCOUNT_ID\"\u001b[39m] = \u001b[33m\"pR5gsAQJrJYJuapM3\"\u001b[39m \u001b[38;5;66;03m# from Preferences\u001b[39;00m\n\u001b[32m 9\u001b[39m os.environ[\u001b[33m\"AUTH_TOKEN\"\u001b[39m] = \u001b[33m\"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\"\u001b[39m \u001b[38;5;66;03m# the API token\u001b[39;00m\n\u001b[32m 10\u001b[39m \n\u001b[32m 11\u001b[39m \n\u001b[32m---> \u001b[39m\u001b[32m12\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m authenticate()\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/auth.py:65\u001b[39m, in \u001b[36mauthenticate\u001b[39m\u001b[34m(force, globals_dict)\u001b[39m\n\u001b[32m 63\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m authenticate_jupyterlite(data_from_host)\n\u001b[32m 64\u001b[39m \u001b[38;5;28;01melif\u001b[39;00m ACCESS_TOKEN_ENV_VAR \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m os.environ \u001b[38;5;129;01mor\u001b[39;00m force:\n\u001b[32m---> \u001b[39m\u001b[32m65\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m _authenticate_oidc_with_cache(force)\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/auth.py:36\u001b[39m, in \u001b[36m_authenticate_oidc_with_cache\u001b[39m\u001b[34m(force)\u001b[39m\n\u001b[32m 33\u001b[39m store_token_data_in_environment(cached)\n\u001b[32m 34\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m\n\u001b[32m---> \u001b[39m\u001b[32m36\u001b[39m token_data = \u001b[38;5;28;01mawait\u001b[39;00m authenticate_oidc(show_popup=show_device_flow_popup)\n\u001b[32m 37\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m save_token(oidc_url, token_data)\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/core/api/auth.py:92\u001b[39m, in \u001b[36mauthenticate_oidc\u001b[39m\u001b[34m(oidc_base_url, client_id, scope, show_popup)\u001b[39m\n\u001b[32m 90\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m show_popup \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m:\n\u001b[32m 91\u001b[39m show_popup(device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mverification_uri_complete\u001b[39m\u001b[33m\"\u001b[39m], device_flow_state[\u001b[33m\"\u001b[39m\u001b[33muser_code\u001b[39m\u001b[33m\"\u001b[39m])\n\u001b[32m---> \u001b[39m\u001b[32m92\u001b[39m token_data = \u001b[38;5;28;01mawait\u001b[39;00m _poll_for_token_data(\n\u001b[32m 93\u001b[39m oidc_base_url=oidc_base_url,\n\u001b[32m 94\u001b[39m client_id=client_id,\n\u001b[32m 95\u001b[39m device_code=device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mdevice_code\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 96\u001b[39m polling_interval_seconds=device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mpolling_interval_seconds\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 97\u001b[39m expires_in_seconds=device_flow_state[\u001b[33m\"\u001b[39m\u001b[33mexpires_in_seconds\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 98\u001b[39m )\n\u001b[32m 99\u001b[39m store_token_data_in_environment(token_data)\n\u001b[32m 100\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m token_data\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/src/py/mat3ra/notebooks_utils/core/api/auth.py:77\u001b[39m, in \u001b[36m_poll_for_token_data\u001b[39m\u001b[34m(oidc_base_url, client_id, device_code, polling_interval_seconds, expires_in_seconds)\u001b[39m\n\u001b[32m 75\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m token_response.status_code == \u001b[32m200\u001b[39m:\n\u001b[32m 76\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m token_response.json()\n\u001b[32m---> \u001b[39m\u001b[32m77\u001b[39m \u001b[38;5;28;01mawait\u001b[39;00m asyncio.sleep(polling_interval_seconds)\n\u001b[32m 78\u001b[39m \u001b[38;5;28;01mraise\u001b[39;00m \u001b[38;5;167;01mException\u001b[39;00m(\u001b[33m\"\u001b[39m\u001b[33mTimeout waiting for authorization.\u001b[39m\u001b[33m\"\u001b[39m)\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/.pyenv/versions/3.11.2/lib/python3.11/asyncio/tasks.py:639\u001b[39m, in \u001b[36msleep\u001b[39m\u001b[34m(delay, result)\u001b[39m\n\u001b[32m 635\u001b[39m h = loop.call_later(delay,\n\u001b[32m 636\u001b[39m futures._set_result_unless_cancelled,\n\u001b[32m 637\u001b[39m future, result)\n\u001b[32m 638\u001b[39m \u001b[38;5;28;01mtry\u001b[39;00m:\n\u001b[32m--> \u001b[39m\u001b[32m639\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28;01mawait\u001b[39;00m future\n\u001b[32m 640\u001b[39m \u001b[38;5;28;01mfinally\u001b[39;00m:\n\u001b[32m 641\u001b[39m h.cancel()\n", - "\u001b[31mCancelledError\u001b[39m: " - ] - } - ], - "execution_count": 17 + "outputs": [], + "execution_count": 33 }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T18:26:52.466952Z", - "start_time": "2026-09-22T18:26:52.442997Z" + "end_time": "2026-09-22T20:09:30.892985Z", + "start_time": "2026-09-22T20:09:30.885791Z" } }, "source": [ @@ -206,7 +167,7 @@ "client = APIClient.authenticate(**address)" ], "outputs": [], - "execution_count": 18 + "execution_count": 34 }, { "cell_type": "markdown", @@ -219,8 +180,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T18:26:52.908678Z", - "start_time": "2026-09-22T18:26:52.888832Z" + "end_time": "2026-09-22T20:09:30.900779Z", + "start_time": "2026-09-22T20:09:30.894789Z" } }, "source": [ @@ -231,7 +192,7 @@ "from upload_run import account_id, upload" ], "outputs": [], - "execution_count": null + "execution_count": 35 }, { "cell_type": "markdown", @@ -246,8 +207,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T18:26:54.188891Z", - "start_time": "2026-09-22T18:26:53.243339Z" + "end_time": "2026-09-22T20:09:31.044814Z", + "start_time": "2026-09-22T20:09:30.902690Z" } }, "source": [ @@ -264,8 +225,22 @@ ")\n", "print(\"run document:\", document_path)" ], - "outputs": [], - "execution_count": null + "outputs": [ + { + "ename": "ModuleNotFoundError", + "evalue": "No module named 'mat3ra.standata'", + "output_type": "error", + "traceback": [ + "\u001b[31m---------------------------------------------------------------------------\u001b[39m", + "\u001b[31mModuleNotFoundError\u001b[39m Traceback (most recent call last)", + "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[36]\u001b[39m\u001b[32m, line 2\u001b[39m\n\u001b[32m 1\u001b[39m \u001b[38;5;66;03m# reading the run folder and writing the run document: nothing here talks to the platform\u001b[39;00m\n\u001b[32m----> \u001b[39m\u001b[32m2\u001b[39m parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n\u001b[32m 3\u001b[39m document_path = serialize(parsed, \u001b[33m\"parsed\"\u001b[39m)\n\u001b[32m 4\u001b[39m run = load(document_path)\n\u001b[32m 5\u001b[39m file_count = sum(len(files) \u001b[38;5;28;01mfor\u001b[39;00m files \u001b[38;5;28;01min\u001b[39;00m run[\u001b[33m\"files\"\u001b[39m].values())\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:282\u001b[39m, in \u001b[36mparse\u001b[39m\u001b[34m(run_dir, physical_id, limit_records, deposition, instrument)\u001b[39m\n\u001b[32m 280\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m label \u001b[38;5;129;01min\u001b[39;00m samples:\n\u001b[32m 281\u001b[39m samples[label][\u001b[33m\"\u001b[39m\u001b[33mmetadata\u001b[39m\u001b[33m\"\u001b[39m].update(const)\n\u001b[32m--> \u001b[39m\u001b[32m282\u001b[39m workflow = \u001b[30;43mstandata_workflow\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43mINSTRUMENT_NAME\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mSS-PFM Hysteresis Loop\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 283\u001b[39m unit = unit_id(workflow)\n\u001b[32m 284\u001b[39m measurement_set = {\u001b[33m\"\u001b[39m\u001b[33mname\u001b[39m\u001b[33m\"\u001b[39m: run_name, \u001b[33m\"\u001b[39m\u001b[33mentitySetType\u001b[39m\u001b[33m\"\u001b[39m: \u001b[33m\"\u001b[39m\u001b[33mordered\u001b[39m\u001b[33m\"\u001b[39m,\n\u001b[32m 285\u001b[39m \u001b[33m\"\u001b[39m\u001b[33mmetadata\u001b[39m\u001b[33m\"\u001b[39m: {\u001b[33m\"\u001b[39m\u001b[33msession\u001b[39m\u001b[33m\"\u001b[39m: session, \u001b[33m\"\u001b[39m\u001b[33mrecipe\u001b[39m\u001b[33m\"\u001b[39m: recipe, \u001b[33m\"\u001b[39m\u001b[33mcontext\u001b[39m\u001b[33m\"\u001b[39m: recipe.get(\u001b[33m\"\u001b[39m\u001b[33mcontext\u001b[39m\u001b[33m\"\u001b[39m, \u001b[33m\"\u001b[39m\u001b[33m\"\u001b[39m),\n\u001b[32m 286\u001b[39m \u001b[33m\"\u001b[39m\u001b[33mloop_settings\u001b[39m\u001b[33m\"\u001b[39m: recipe[\u001b[33m\"\u001b[39m\u001b[33mper_site\u001b[39m\u001b[33m\"\u001b[39m][\u001b[32m0\u001b[39m][\u001b[33m\"\u001b[39m\u001b[33mloop_settings\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 287\u001b[39m \u001b[33m\"\u001b[39m\u001b[33msites\u001b[39m\u001b[33m\"\u001b[39m: \u001b[38;5;28mlist\u001b[39m(samples), \u001b[33m\"\u001b[39m\u001b[33mcommon\u001b[39m\u001b[33m\"\u001b[39m: common, \u001b[33m\"\u001b[39m\u001b[33mregistration\u001b[39m\u001b[33m\"\u001b[39m: reg}}\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:18\u001b[39m, in \u001b[36mstandata_workflow\u001b[39m\u001b[34m(application_name, workflow_name)\u001b[39m\n\u001b[32m 14\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34mstandata_workflow\u001b[39m(application_name, workflow_name):\n\u001b[32m 15\u001b[39m \u001b[38;5;250m \u001b[39m\u001b[33;03m\"\"\"The procedure the instrument runs, from the standata registry — the same entry the platform resolves a\u001b[39;00m\n\u001b[32m 16\u001b[39m \u001b[33;03m job's workflow through. Building one here would be a second source of truth for something that already\u001b[39;00m\n\u001b[32m 17\u001b[39m \u001b[33;03m has one; a new instrument is a new registry entry, not code.\"\"\"\u001b[39;00m\n\u001b[32m---> \u001b[39m\u001b[32m18\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mmat3ra\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01mstandata\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01mworkflows\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m WorkflowStandata\n\u001b[32m 19\u001b[39m workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name)\n\u001b[32m 20\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m workflow \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m:\n", + "\u001b[31mModuleNotFoundError\u001b[39m: No module named 'mat3ra.standata'" + ] + } + ], + "execution_count": 36 }, { "cell_type": "markdown", @@ -278,34 +253,13 @@ }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T18:26:54.254975Z", - "start_time": "2026-09-22T18:26:54.202446Z" - } - }, + "metadata": {}, "source": [ "if ACCOUNT_SLUG:\n", " client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG), **address)" ], - "outputs": [ - { - "ename": "ValueError", - "evalue": "Access token is required to fetch user data", - "output_type": "error", - "traceback": [ - "\u001b[31m---------------------------------------------------------------------------\u001b[39m", - "\u001b[31mValueError\u001b[39m Traceback (most recent call last)", - "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[21]\u001b[39m\u001b[32m, line 2\u001b[39m\n\u001b[32m 1\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m ACCOUNT_SLUG:\n\u001b[32m----> \u001b[39m\u001b[32m2\u001b[39m client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG), **address)\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/upload_run.py:522\u001b[39m, in \u001b[36maccount_id\u001b[39m\u001b[34m(client, slug_or_name)\u001b[39m\n\u001b[32m 520\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34maccount_id\u001b[39m(client, slug_or_name):\n\u001b[32m 521\u001b[39m \u001b[38;5;250m \u001b[39m\u001b[33;03m\"\"\"`--account ` → the id of that account: the slug the platform shows it under, else its display name.\"\"\"\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m522\u001b[39m accounts = \u001b[30;43mclient\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mlist_accounts\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 523\u001b[39m \u001b[38;5;28;01mfor\u001b[39;00m field \u001b[38;5;129;01min\u001b[39;00m (\u001b[33m\"\u001b[39m\u001b[33mslug\u001b[39m\u001b[33m\"\u001b[39m, \u001b[33m\"\u001b[39m\u001b[33mname\u001b[39m\u001b[33m\"\u001b[39m):\n\u001b[32m 524\u001b[39m account = \u001b[38;5;28mnext\u001b[39m((a \u001b[38;5;28;01mfor\u001b[39;00m a \u001b[38;5;129;01min\u001b[39;00m accounts \u001b[38;5;28;01mif\u001b[39;00m a.get(field) == slug_or_name), \u001b[38;5;28;01mNone\u001b[39;00m)\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/mat3ra/api_client/client.py:150\u001b[39m, in \u001b[36mAPIClient.list_accounts\u001b[39m\u001b[34m(self)\u001b[39m\n\u001b[32m 149\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34mlist_accounts\u001b[39m(\u001b[38;5;28mself\u001b[39m) -> List[\u001b[38;5;28mdict\u001b[39m]:\n\u001b[32m--> \u001b[39m\u001b[32m150\u001b[39m accounts = \u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43m_fetch_user_accounts\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 151\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m [\n\u001b[32m 152\u001b[39m {\n\u001b[32m 153\u001b[39m \u001b[33m\"\u001b[39m\u001b[33m_id\u001b[39m\u001b[33m\"\u001b[39m: account[\u001b[33m\"\u001b[39m\u001b[33mentity\u001b[39m\u001b[33m\"\u001b[39m][\u001b[33m\"\u001b[39m\u001b[33m_id\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m (...)\u001b[39m\u001b[32m 159\u001b[39m \u001b[38;5;28;01mfor\u001b[39;00m account \u001b[38;5;129;01min\u001b[39;00m accounts\n\u001b[32m 160\u001b[39m ]\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/mat3ra/api_client/client.py:147\u001b[39m, in \u001b[36mAPIClient._fetch_user_accounts\u001b[39m\u001b[34m(self)\u001b[39m\n\u001b[32m 146\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34m_fetch_user_accounts\u001b[39m(\u001b[38;5;28mself\u001b[39m) -> List[\u001b[38;5;28mdict\u001b[39m]:\n\u001b[32m--> \u001b[39m\u001b[32m147\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43m_fetch_data\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m)\u001b[39;49m.get(\u001b[33m\"\u001b[39m\u001b[33maccounts\u001b[39m\u001b[33m\"\u001b[39m, [])\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/mat3ra/api_client/client.py:139\u001b[39m, in \u001b[36mAPIClient._fetch_data\u001b[39m\u001b[34m(self)\u001b[39m\n\u001b[32m 137\u001b[39m access_token = \u001b[38;5;28mself\u001b[39m.auth.access_token \u001b[38;5;129;01mor\u001b[39;00m os.environ.get(ACCESS_TOKEN_ENV_VAR)\n\u001b[32m 138\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m access_token:\n\u001b[32m--> \u001b[39m\u001b[32m139\u001b[39m \u001b[38;5;28;01mraise\u001b[39;00m \u001b[38;5;167;01mValueError\u001b[39;00m(\u001b[33m\"\u001b[39m\u001b[33mAccess token is required to fetch user data\u001b[39m\u001b[33m\"\u001b[39m)\n\u001b[32m 141\u001b[39m url = _build_base_url(\u001b[38;5;28mself\u001b[39m.host, \u001b[38;5;28mself\u001b[39m.port, \u001b[38;5;28mself\u001b[39m.secure, \u001b[33m\"\u001b[39m\u001b[33m/api/v1/users/me\u001b[39m\u001b[33m\"\u001b[39m)\n\u001b[32m 142\u001b[39m response = requests.get(url, headers={\u001b[33m\"\u001b[39m\u001b[33mAuthorization\u001b[39m\u001b[33m\"\u001b[39m: \u001b[33mf\u001b[39m\u001b[33m\"\u001b[39m\u001b[33mBearer \u001b[39m\u001b[38;5;132;01m{\u001b[39;00maccess_token\u001b[38;5;132;01m}\u001b[39;00m\u001b[33m\"\u001b[39m}, timeout=\u001b[32m30\u001b[39m)\n", - "\u001b[31mValueError\u001b[39m: Access token is required to fetch user data" - ] - } - ], - "execution_count": 21 + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -336,12 +290,7 @@ }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T18:26:54.343400Z", - "start_time": "2026-09-22T18:26:54.305317Z" - } - }, + "metadata": {}, "source": [ "print(f\"Open {HOST}, your account's Measurements tab: {run['run']}\")" ], @@ -360,9 +309,9 @@ { "metadata": {}, "cell_type": "code", + "source": "", "outputs": [], - "execution_count": null, - "source": "" + "execution_count": null } ], "metadata": { From 0a6eb9ccbbd15ddd802b4cb264efb3c8f800879f Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 13:19:44 -0700 Subject: [PATCH 20/36] feat(SOF-8051): one file states exactly what to install, and the example says how to run it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A lab could not get this running without being told things by hand. Now `examples/measurement/requirements.txt` names every package and the exact source of each, and README.md gives the three commands: install, parse, upload. Three pins are by branch, and the reason is the same each time — the PyPI release predates what this example uses: api-client the samples/measurements/files endpoints, standata the three instrument registry entries, esse the Sample and Measurement schemas. A direct reference wins over any PyPI version, so the pins hold even where PyPI carries a higher version number (standata does today). They become ordinary version pins when those branches release. Two things a clean-venv install turned up, both fixed here: - mat3ra-esse from PyPI has no mat3ra.esse.models.sample, so the optional import failed and every document went up unvalidated. - That skip printed "validation: OK". A skipped check is not a passed one; it now says SKIPPED and names what is missing. Verified end to end in a bare venv: install from requirements.txt, parse NLR's delivery, validate both run documents. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/README.md | 88 +++++++++++++++++ examples/measurement/parse_nlr.py | 19 +++- examples/measurement/parse_utk.py | 19 +++- examples/measurement/requirements.txt | 20 ++++ examples/measurement/upload_nlr_data.ipynb | 6 +- examples/measurement/upload_run.py | 12 ++- examples/measurement/upload_spm_run.ipynb | 106 ++++++++++----------- 7 files changed, 205 insertions(+), 65 deletions(-) create mode 100644 examples/measurement/README.md create mode 100644 examples/measurement/requirements.txt diff --git a/examples/measurement/README.md b/examples/measurement/README.md new file mode 100644 index 000000000..32be14650 --- /dev/null +++ b/examples/measurement/README.md @@ -0,0 +1,88 @@ +# Uploading a measurement run + +Data a lab delivers becomes Samples, Measurements and Properties on the platform in two steps that stay apart: +**parse** reads one lab's delivery and writes a *run document*; **upload** takes run documents and sends them. +Nothing in the uploader knows what an instrument is, and no parser talks to the platform. + +``` + delivery ──► parse_utk.py ──► run document ──► upload_run.py ──► platform + parse_nlr.py (parsed/*.json) +``` + +## Install + +```bash +python -m venv venv && . venv/bin/activate # Windows: venv\Scripts\activate +pip install -r requirements.txt +``` + +That file names every package and the exact source of each. Three of them are pinned to a branch: the PyPI +releases do not yet have the REST endpoints, the instrument registry entries, or the Sample and Measurement +schemas this example uses. The pins become ordinary version numbers when those branches release. + +Check it worked — this must print three workflow names, not `None`: + +```bash +python -c "from mat3ra.standata.workflows import WorkflowStandata as W; print([ + W.find_by_application_and_name(a, n)['name'] for a, n in + (('asylum-spm','SS-PFM Hysteresis Loop'),('xrf-mapper','XRF Grid Map'),('probe-station','DC I-V Sweep'))])" +``` + +## Credentials + +The uploader reads them from the environment. Either an API token from the platform's Preferences page: + +```bash +export ACCOUNT_ID=... AUTH_TOKEN=... +export MAT3RA_HOST=alphafilm.mat3ra.com # or pass --host +``` + +or, if you signed in through the browser, `OIDC_ACCESS_TOKEN`. + +## Parse + +Each parser reads one lab's delivery. Point it at the folder as delivered and give it the identifier written +on the physical piece — every Sample carries it, and it is how the piece is found again later. + +```bash +# UTK: an Asylum SPM run folder (summary.json or recipe.json + records/ + loops/) +python parse_utk.py ~/data/From_UTK --physical-id PDAC_COM5_01448 --out parsed + +# NLR: an XRF grid and a DC I-V sweep over the same pads +python parse_nlr.py ~/data/From_NLR --physical-id PDAC_COM5_01448 \ + --xrf-instrument bruker-m4 --iv-instrument keithley-4200 --out parsed +``` + +`parsed/` now holds a run document per run, plus any file a parser derived. Read it — it is the whole upload, +in JSON, before anything is sent. `run_document.py` states the shape. + +## Upload + +```bash +python upload_run.py parsed/*.json --account --files records +``` + +- `--files` picks which file groups go up: `records` (the per-measurement JSONs, the delivered tables, the + photographs) and `loops` (the raw arrays and plots — thousands of files, tens of minutes). `--files` with no + value uploads none and keeps the raw records in each measurement's metadata instead. +- `--dry-run` validates the documents against the ESSE schemas and stops. +- Re-running is safe: sets are found by name, samples and measurements by label, properties by what they belong + to. A second pass creates nothing. + +Then open the platform's Measurements tab: the run is there as a set, one measurement per sample. + +## Adding a lab + +Write a parser. It reads whatever that lab ships and returns the dict `run_document.py` describes — one sample +per measured position, one measurement per sample, the properties each measurement produced. Take the +measurement's workflow from standata (`standata_workflow(application, name)`); if the instrument is not in that +registry yet, add it there — `mat3ra/standata`, `assets/applications/` and `assets/workflows/` — rather than +building a workflow in Python, so the platform resolves it the same way it resolves a job's. + +Nothing in `upload_run.py` changes. + +## The notebooks + +`upload_spm_run.ipynb` and `upload_nlr_data.ipynb` run the same three steps with the same code, for people who +would rather not use a terminal. They install the pins themselves; restart the kernel after that cell if either +package was already imported. diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index d42620068..5f997b62b 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -19,11 +19,26 @@ def standata_workflow(application_name, workflow_name): from mat3ra.standata.workflows import WorkflowStandata workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) if workflow is None: - raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}: " - "add it to mat3ra/standata, or pin a release that has it") + raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}.\n{standata_origin()}\n" + 'Install the branch it is registered on and RESTART THE KERNEL (a %pip install does not\n' + 'replace a module this session already imported):\n' + ' pip install --force-reinstall --no-deps "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"') return workflow +def standata_origin(): + """Which mat3ra-standata is in this interpreter — the answer to 'but it is installed'.""" + import importlib.metadata as metadata + try: + distribution = metadata.distribution("mat3ra-standata") + except metadata.PackageNotFoundError: + return "mat3ra-standata is not installed." + direct_url = distribution.read_text("direct_url.json") + source = "the feature branch" if direct_url and "feature/SOF-8051" in direct_url else \ + "another branch" if direct_url else "PyPI, which predates these entries" + return f"You have mat3ra-standata {distribution.version} from {source}." + + def unit_id(workflow): """The execution unit a property of this workflow comes from.""" return workflow["subworkflows"][0]["units"][0]["flowchartId"] diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py index e15e19aed..304338e5e 100644 --- a/examples/measurement/parse_utk.py +++ b/examples/measurement/parse_utk.py @@ -21,11 +21,26 @@ def standata_workflow(application_name, workflow_name): from mat3ra.standata.workflows import WorkflowStandata workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) if workflow is None: - raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}: " - "add it to mat3ra/standata, or pin a release that has it") + raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}.\n{standata_origin()}\n" + 'Install the branch it is registered on and RESTART THE KERNEL (a %pip install does not\n' + 'replace a module this session already imported):\n' + ' pip install --force-reinstall --no-deps "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"') return workflow +def standata_origin(): + """Which mat3ra-standata is in this interpreter — the answer to 'but it is installed'.""" + import importlib.metadata as metadata + try: + distribution = metadata.distribution("mat3ra-standata") + except metadata.PackageNotFoundError: + return "mat3ra-standata is not installed." + direct_url = distribution.read_text("direct_url.json") + source = "the feature branch" if direct_url and "feature/SOF-8051" in direct_url else \ + "another branch" if direct_url else "PyPI, which predates these entries" + return f"You have mat3ra-standata {distribution.version} from {source}." + + def unit_id(workflow): """The execution unit a property of this workflow comes from.""" return workflow["subworkflows"][0]["units"][0]["flowchartId"] diff --git a/examples/measurement/requirements.txt b/examples/measurement/requirements.txt new file mode 100644 index 000000000..5c63fe123 --- /dev/null +++ b/examples/measurement/requirements.txt @@ -0,0 +1,20 @@ +# Everything the measurement upload needs, exactly. Install it and the example runs: +# +# python -m venv venv && . venv/bin/activate +# pip install -r requirements.txt +# +# api-client and standata are named by branch because the PyPI releases predate what this +# example uses — api-client the samples/measurements/files endpoints, standata the three +# instrument registry entries. A direct reference like this one wins over any PyPI version, +# so it holds even though PyPI's standata carries a higher version number than the branch. +# Both lines become ordinary version pins once those branches merge and release. + +mat3ra-api-client @ git+https://github.com/mat3ra/api-client.git@feature/SOF-8051 +mat3ra-standata @ git+https://github.com/mat3ra/standata.git@feature/SOF-8051 + +# schema validation before anything is uploaded — by branch for the same reason: PyPI's esse +# has no Sample or Measurement schema yet, and without them validation silently does nothing +mat3ra-esse @ git+https://github.com/mat3ra/esse.git@feature/SOF-8051 + +# the file uploads retry a refused connection +requests diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index f5db5e51e..18d6f4d05 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -25,7 +25,11 @@ "metadata": {}, "outputs": [], "source": [ - "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\" \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"" + "# Exactly what this notebook needs, and where each package comes from: requirements.txt beside it.\n", + "# Three are pinned to a branch because the PyPI releases lack the REST endpoints, the instrument\n", + "# registry entries and the Sample/Measurement schemas. Restart the kernel after this cell if any\n", + "# of them was already imported in this session.\n", + "%pip install -q -r requirements.txt" ] }, { diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 8a650b07a..1456cb007 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -43,8 +43,7 @@ def validate(parsed): """Validate every document against the ESSE schemas; returns the number of invalid ones. Skipped (returns 0) when the optional mat3ra-esse package is not installed.""" if ESSE is None: - print("schema validation skipped: `pip install mat3ra-esse` to enable it", flush=True) - return 0 + return None # not installed: the caller says so rather than reporting a pass esse = ESSE() schemas = {x["$id"]: x for x in esse.schemas} errors = 0 @@ -264,8 +263,13 @@ def main(): a = ap.parse_args() runs = [load(path) for path in a.documents] - errors = sum(validate(run) for run in runs) - print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") + counts = [validate(run) for run in runs] + if any(count is None for count in counts): + print("validation: SKIPPED — mat3ra-esse is not installed (see requirements.txt); nothing was checked") + errors = 0 + else: + errors = sum(counts) + print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") if errors or a.dry_run: sys.exit(1 if errors else 0) diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index e76aeaec4..5b53eb397 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -23,18 +23,23 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:09:30.844346Z", - "start_time": "2026-09-22T20:09:30.085762Z" + "end_time": "2026-09-22T20:15:27.733918Z", + "start_time": "2026-09-22T20:15:26.503559Z" } }, "source": [ - "%pip install -q \"mat3ra-notebooks-utils[all]\" \"git+https://github.com/mat3ra/api-client.git@feature/SOF-8051\" \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"" + "# Exactly what this notebook needs, and where each package comes from: requirements.txt beside it.\n", + "# Three are pinned to a branch because the PyPI releases lack the REST endpoints, the instrument\n", + "# registry entries and the Sample/Measurement schemas. Restart the kernel after this cell if any\n", + "# of them was already imported in this session.\n", + "%pip install -q -r requirements.txt" ], "outputs": [ { "name": "stdout", "output_type": "stream", "text": [ + "Note: you may need to restart the kernel to use updated packages.\n", " \u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\r\n", " \r\n", " \u001b[31m×\u001b[0m \u001b[32mgit version\u001b[0m did not run successfully.\r\n", @@ -49,7 +54,7 @@ ] } ], - "execution_count": 30 + "execution_count": 1 }, { "cell_type": "markdown", @@ -68,13 +73,17 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:09:30.856023Z", - "start_time": "2026-09-22T20:09:30.845688Z" + "end_time": "2026-09-22T20:15:27.747010Z", + "start_time": "2026-09-22T20:15:27.737309Z" } }, "source": [ "import urllib.parse\n", "\n", + "# NOTE: generate at https://alphafilm.mat3ra.com/demo/preferences API Tokens\n", + "ACCOUNT_ID = \"pR5gsAQJrJYJuapM3\" # from Preferences\n", + "AUTH_TOKEN = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\" # the API token\n", + "\n", "HOST = \"https://alphafilm.mat3ra.com\"\n", "RUN_DIR = \"/Users/mat3ra/code/work/SOF-8050/data/From_UTK\"\n", "PHYSICAL_ID = \"test-01448\"\n", @@ -89,7 +98,7 @@ "}" ], "outputs": [], - "execution_count": 31 + "execution_count": 2 }, { "cell_type": "markdown", @@ -108,8 +117,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:09:30.868383Z", - "start_time": "2026-09-22T20:09:30.857101Z" + "end_time": "2026-09-22T20:15:27.770556Z", + "start_time": "2026-09-22T20:15:27.751353Z" } }, "source": [ @@ -126,39 +135,39 @@ ] } ], - "execution_count": 32 + "execution_count": 3 }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:09:30.881225Z", - "start_time": "2026-09-22T20:09:30.869178Z" + "end_time": "2026-09-22T20:15:27.937543Z", + "start_time": "2026-09-22T20:15:27.771266Z" } }, "source": [ "from mat3ra.notebooks_utils.auth import authenticate\n", "\n", - "import os\n", - "\n", - "os.environ[\"API_HOST\"] = address[\"host\"]\n", - "os.environ[\"API_PORT\"] = str(address[\"port\"])\n", - "os.environ[\"API_SECURE\"] = str(address[\"secure\"])\n", - "os.environ[\"ACCOUNT_ID\"] = \"pR5gsAQJrJYJuapM3\" # from Preferences\n", - "os.environ[\"AUTH_TOKEN\"] = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\" # the API token\n", + "os.environ[\"ACCOUNT_ID\"] = ACCOUNT_ID\n", + "os.environ[\"AUTH_TOKEN\"] = AUTH_TOKEN\n", "\n", + "import os\n", "\n", + "# NOTE: uncomment to login with OIDC interactively\n", + "# os.environ[\"API_HOST\"] = address[\"host\"]\n", + "# os.environ[\"API_PORT\"] = str(address[\"port\"])\n", + "# os.environ[\"API_SECURE\"] = str(address[\"secure\"])\n", "# await authenticate()" ], "outputs": [], - "execution_count": 33 + "execution_count": 4 }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:09:30.892985Z", - "start_time": "2026-09-22T20:09:30.885791Z" + "end_time": "2026-09-22T20:15:27.945465Z", + "start_time": "2026-09-22T20:15:27.938607Z" } }, "source": [ @@ -167,7 +176,7 @@ "client = APIClient.authenticate(**address)" ], "outputs": [], - "execution_count": 34 + "execution_count": 5 }, { "cell_type": "markdown", @@ -180,8 +189,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:09:30.900779Z", - "start_time": "2026-09-22T20:09:30.894789Z" + "end_time": "2026-09-22T20:15:28.047309Z", + "start_time": "2026-09-22T20:15:27.948299Z" } }, "source": [ @@ -192,7 +201,7 @@ "from upload_run import account_id, upload" ], "outputs": [], - "execution_count": 35 + "execution_count": 6 }, { "cell_type": "markdown", @@ -207,8 +216,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:09:31.044814Z", - "start_time": "2026-09-22T20:09:30.902690Z" + "end_time": "2026-09-22T20:15:28.548348Z", + "start_time": "2026-09-22T20:15:28.048312Z" } }, "source": [ @@ -227,39 +236,24 @@ ], "outputs": [ { - "ename": "ModuleNotFoundError", - "evalue": "No module named 'mat3ra.standata'", + "ename": "SystemExit", + "evalue": "standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm.\nYou have mat3ra-standata 2026.9.11.post2 from PyPI, which predates these entries.\nInstall the branch it is registered on and RESTART THE KERNEL (a %pip install does not\nreplace a module this session already imported):\n pip install --force-reinstall --no-deps \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"", "output_type": "error", "traceback": [ - "\u001b[31m---------------------------------------------------------------------------\u001b[39m", - "\u001b[31mModuleNotFoundError\u001b[39m Traceback (most recent call last)", - "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[36]\u001b[39m\u001b[32m, line 2\u001b[39m\n\u001b[32m 1\u001b[39m \u001b[38;5;66;03m# reading the run folder and writing the run document: nothing here talks to the platform\u001b[39;00m\n\u001b[32m----> \u001b[39m\u001b[32m2\u001b[39m parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n\u001b[32m 3\u001b[39m document_path = serialize(parsed, \u001b[33m\"parsed\"\u001b[39m)\n\u001b[32m 4\u001b[39m run = load(document_path)\n\u001b[32m 5\u001b[39m file_count = sum(len(files) \u001b[38;5;28;01mfor\u001b[39;00m files \u001b[38;5;28;01min\u001b[39;00m run[\u001b[33m\"files\"\u001b[39m].values())\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:282\u001b[39m, in \u001b[36mparse\u001b[39m\u001b[34m(run_dir, physical_id, limit_records, deposition, instrument)\u001b[39m\n\u001b[32m 280\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m label \u001b[38;5;129;01min\u001b[39;00m samples:\n\u001b[32m 281\u001b[39m samples[label][\u001b[33m\"\u001b[39m\u001b[33mmetadata\u001b[39m\u001b[33m\"\u001b[39m].update(const)\n\u001b[32m--> \u001b[39m\u001b[32m282\u001b[39m workflow = \u001b[30;43mstandata_workflow\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43mINSTRUMENT_NAME\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mSS-PFM Hysteresis Loop\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 283\u001b[39m unit = unit_id(workflow)\n\u001b[32m 284\u001b[39m measurement_set = {\u001b[33m\"\u001b[39m\u001b[33mname\u001b[39m\u001b[33m\"\u001b[39m: run_name, \u001b[33m\"\u001b[39m\u001b[33mentitySetType\u001b[39m\u001b[33m\"\u001b[39m: \u001b[33m\"\u001b[39m\u001b[33mordered\u001b[39m\u001b[33m\"\u001b[39m,\n\u001b[32m 285\u001b[39m \u001b[33m\"\u001b[39m\u001b[33mmetadata\u001b[39m\u001b[33m\"\u001b[39m: {\u001b[33m\"\u001b[39m\u001b[33msession\u001b[39m\u001b[33m\"\u001b[39m: session, \u001b[33m\"\u001b[39m\u001b[33mrecipe\u001b[39m\u001b[33m\"\u001b[39m: recipe, \u001b[33m\"\u001b[39m\u001b[33mcontext\u001b[39m\u001b[33m\"\u001b[39m: recipe.get(\u001b[33m\"\u001b[39m\u001b[33mcontext\u001b[39m\u001b[33m\"\u001b[39m, \u001b[33m\"\u001b[39m\u001b[33m\"\u001b[39m),\n\u001b[32m 286\u001b[39m \u001b[33m\"\u001b[39m\u001b[33mloop_settings\u001b[39m\u001b[33m\"\u001b[39m: recipe[\u001b[33m\"\u001b[39m\u001b[33mper_site\u001b[39m\u001b[33m\"\u001b[39m][\u001b[32m0\u001b[39m][\u001b[33m\"\u001b[39m\u001b[33mloop_settings\u001b[39m\u001b[33m\"\u001b[39m],\n\u001b[32m 287\u001b[39m \u001b[33m\"\u001b[39m\u001b[33msites\u001b[39m\u001b[33m\"\u001b[39m: \u001b[38;5;28mlist\u001b[39m(samples), \u001b[33m\"\u001b[39m\u001b[33mcommon\u001b[39m\u001b[33m\"\u001b[39m: common, \u001b[33m\"\u001b[39m\u001b[33mregistration\u001b[39m\u001b[33m\"\u001b[39m: reg}}\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:18\u001b[39m, in \u001b[36mstandata_workflow\u001b[39m\u001b[34m(application_name, workflow_name)\u001b[39m\n\u001b[32m 14\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34mstandata_workflow\u001b[39m(application_name, workflow_name):\n\u001b[32m 15\u001b[39m \u001b[38;5;250m \u001b[39m\u001b[33;03m\"\"\"The procedure the instrument runs, from the standata registry — the same entry the platform resolves a\u001b[39;00m\n\u001b[32m 16\u001b[39m \u001b[33;03m job's workflow through. Building one here would be a second source of truth for something that already\u001b[39;00m\n\u001b[32m 17\u001b[39m \u001b[33;03m has one; a new instrument is a new registry entry, not code.\"\"\"\u001b[39;00m\n\u001b[32m---> \u001b[39m\u001b[32m18\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mmat3ra\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01mstandata\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01mworkflows\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m WorkflowStandata\n\u001b[32m 19\u001b[39m workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name)\n\u001b[32m 20\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m workflow \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m:\n", - "\u001b[31mModuleNotFoundError\u001b[39m: No module named 'mat3ra.standata'" + "An exception has occurred, use %tb to see the full traceback.\n", + "\u001b[31mSystemExit\u001b[39m\u001b[31m:\u001b[39m standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm.\nYou have mat3ra-standata 2026.9.11.post2 from PyPI, which predates these entries.\nInstall the branch it is registered on and RESTART THE KERNEL (a %pip install does not\nreplace a module this session already imported):\n pip install --force-reinstall --no-deps \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"\n" + ] + }, + { + "name": "stderr", + "output_type": "stream", + "text": [ + "/Users/mat3ra/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/IPython/core/interactiveshell.py:3831: UserWarning: To exit: use 'exit', 'quit', or Ctrl-D.\n", + " warn(\"To exit: use 'exit', 'quit', or Ctrl-D.\", stacklevel=1)\n" ] } ], - "execution_count": 36 - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Select the account\n", - "\n", - "`ACCOUNT_SLUG` re-authenticates the client against that account, so the run is read and written there." - ] - }, - { - "cell_type": "code", - "metadata": {}, - "source": [ - "if ACCOUNT_SLUG:\n", - " client = APIClient.authenticate(account_id=account_id(client, ACCOUNT_SLUG), **address)" - ], - "outputs": [], - "execution_count": null + "execution_count": 7 }, { "cell_type": "markdown", From 85a786761ca498425a5d43ee6355e0e51ec12828 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 13:23:32 -0700 Subject: [PATCH 21/36] update cleanup --- examples/measurement/upload_spm_run.ipynb | 26 +++++------------------ 1 file changed, 5 insertions(+), 21 deletions(-) diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 5b53eb397..7d815e0e0 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -80,15 +80,15 @@ "source": [ "import urllib.parse\n", "\n", + "HOST = \"https://alphafilm.mat3ra.com\"\n", + "\n", "# NOTE: generate at https://alphafilm.mat3ra.com/demo/preferences API Tokens\n", "ACCOUNT_ID = \"pR5gsAQJrJYJuapM3\" # from Preferences\n", "AUTH_TOKEN = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\" # the API token\n", "\n", - "HOST = \"https://alphafilm.mat3ra.com\"\n", - "RUN_DIR = \"/Users/mat3ra/code/work/SOF-8050/data/From_UTK\"\n", - "PHYSICAL_ID = \"test-01448\"\n", - "ACCOUNT_SLUG = \"demo\"\n", - "FILES = [\"records\", \"loops\"] # \"records\": the record JSONs, \"loops\": the loop arrays and plots; [] uploads none\n", + "RUN_DIR = \"\" # folder with summary.json, records etc\n", + "PHYSICAL_ID = \"\" # the physical wafer's identifier (e.g. PDAC_COM5_01448)\n", + "FILES = [\"records\", \"loops\"] # Which folders to upload as files per measurement.\n", "\n", "url = urllib.parse.urlsplit(HOST)\n", "address = {\n", @@ -290,22 +290,6 @@ ], "outputs": [], "execution_count": null - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## References\n", - "\n", - "- [Mat3ra REST API](https://docs.mat3ra.com/rest-api/overview/)" - ] - }, - { - "metadata": {}, - "cell_type": "code", - "source": "", - "outputs": [], - "execution_count": null } ], "metadata": { From ef1b1dff48ecfbca51229ad1e04352849857a1dd Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 13:32:02 -0700 Subject: [PATCH 22/36] fix(SOF-8051): pin the versions, drop the machinery that asked which version you had MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit requirements.txt names the correct package for each of api-client, standata and esse, so the code does not need to check. Gone: check_environment.py, the notebook cell that ran it, standata_origin() and the instructions it printed, and the optional-import dance around esse — it is a requirement, so it is imported. `install_packages("api")` went with it: outside JupyterLite it only prints advice that contradicts requirements.txt, and two install paths is one too many. Clean-venv run of both deliveries after: parse, then validate, OK. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/README.md | 13 +----- examples/measurement/parse_nlr.py | 32 +++----------- examples/measurement/parse_utk.py | 32 +++----------- examples/measurement/requirements.txt | 19 +++----- examples/measurement/upload_nlr_data.ipynb | 15 ------- examples/measurement/upload_run.py | 26 +++-------- examples/measurement/upload_spm_run.ipynb | 51 +--------------------- 7 files changed, 26 insertions(+), 162 deletions(-) diff --git a/examples/measurement/README.md b/examples/measurement/README.md index 32be14650..00e9fa588 100644 --- a/examples/measurement/README.md +++ b/examples/measurement/README.md @@ -16,17 +16,8 @@ python -m venv venv && . venv/bin/activate # Windows: venv\Scripts\activate pip install -r requirements.txt ``` -That file names every package and the exact source of each. Three of them are pinned to a branch: the PyPI -releases do not yet have the REST endpoints, the instrument registry entries, or the Sample and Measurement -schemas this example uses. The pins become ordinary version numbers when those branches release. - -Check it worked — this must print three workflow names, not `None`: - -```bash -python -c "from mat3ra.standata.workflows import WorkflowStandata as W; print([ - W.find_by_application_and_name(a, n)['name'] for a, n in - (('asylum-spm','SS-PFM Hysteresis Loop'),('xrf-mapper','XRF Grid Map'),('probe-station','DC I-V Sweep'))])" -``` +api-client, standata and esse are pinned to the branch carrying the REST endpoints, the instrument registry +entries and the Sample and Measurement schemas. They become version pins when it releases. ## Credentials diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index 5f997b62b..41230a285 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -2,41 +2,19 @@ technique over those same pads — the XRF map, then the DC I-V sweep. Ad hoc parser for SOF-8050: it reads the tab-separated files NLR ships and nothing else. - -Requires `pip install "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"` until that branch is released: -the instruments' registry entries are on it, and the PyPI release predates them. """ import argparse from pathlib import Path +from mat3ra.standata.workflows import WorkflowStandata + from run_document import serialize def standata_workflow(application_name, workflow_name): - """The procedure the instrument runs, from the standata registry — the same entry the platform resolves a - job's workflow through. Building one here would be a second source of truth for something that already - has one; a new instrument is a new registry entry, not code.""" - from mat3ra.standata.workflows import WorkflowStandata - workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) - if workflow is None: - raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}.\n{standata_origin()}\n" - 'Install the branch it is registered on and RESTART THE KERNEL (a %pip install does not\n' - 'replace a module this session already imported):\n' - ' pip install --force-reinstall --no-deps "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"') - return workflow - - -def standata_origin(): - """Which mat3ra-standata is in this interpreter — the answer to 'but it is installed'.""" - import importlib.metadata as metadata - try: - distribution = metadata.distribution("mat3ra-standata") - except metadata.PackageNotFoundError: - return "mat3ra-standata is not installed." - direct_url = distribution.read_text("direct_url.json") - source = "the feature branch" if direct_url and "feature/SOF-8051" in direct_url else \ - "another branch" if direct_url else "PyPI, which predates these entries" - return f"You have mat3ra-standata {distribution.version} from {source}." + """The procedure the instrument runs, from the standata registry — the entry the platform resolves + a job's workflow through.""" + return WorkflowStandata.find_by_application_and_name(application_name, workflow_name) def unit_id(workflow): diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py index 304338e5e..ec0489397 100644 --- a/examples/measurement/parse_utk.py +++ b/examples/measurement/parse_utk.py @@ -3,42 +3,20 @@ deviation, count) inside it. Individual loops stay in the measurement's files. Ad hoc parser for SOF-8050: it reads the shape UTK's afm-lib writes and nothing else. - -Requires `pip install "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"` until that branch is released: -the instruments' registry entries are on it, and the PyPI release predates them. """ import argparse, ast, json, math, re, statistics, struct from datetime import datetime, timezone from pathlib import Path +from mat3ra.standata.workflows import WorkflowStandata + from run_document import serialize def standata_workflow(application_name, workflow_name): - """The procedure the instrument runs, from the standata registry — the same entry the platform resolves a - job's workflow through. Building one here would be a second source of truth for something that already - has one; a new instrument is a new registry entry, not code.""" - from mat3ra.standata.workflows import WorkflowStandata - workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) - if workflow is None: - raise SystemExit(f"standata has no '{workflow_name}' workflow for {application_name}.\n{standata_origin()}\n" - 'Install the branch it is registered on and RESTART THE KERNEL (a %pip install does not\n' - 'replace a module this session already imported):\n' - ' pip install --force-reinstall --no-deps "git+https://github.com/mat3ra/standata.git@feature/SOF-8051"') - return workflow - - -def standata_origin(): - """Which mat3ra-standata is in this interpreter — the answer to 'but it is installed'.""" - import importlib.metadata as metadata - try: - distribution = metadata.distribution("mat3ra-standata") - except metadata.PackageNotFoundError: - return "mat3ra-standata is not installed." - direct_url = distribution.read_text("direct_url.json") - source = "the feature branch" if direct_url and "feature/SOF-8051" in direct_url else \ - "another branch" if direct_url else "PyPI, which predates these entries" - return f"You have mat3ra-standata {distribution.version} from {source}." + """The procedure the instrument runs, from the standata registry — the entry the platform resolves + a job's workflow through.""" + return WorkflowStandata.find_by_application_and_name(application_name, workflow_name) def unit_id(workflow): diff --git a/examples/measurement/requirements.txt b/examples/measurement/requirements.txt index 5c63fe123..8d3694456 100644 --- a/examples/measurement/requirements.txt +++ b/examples/measurement/requirements.txt @@ -1,20 +1,11 @@ -# Everything the measurement upload needs, exactly. Install it and the example runs: +# The measurement upload: pip install -r requirements.txt # -# python -m venv venv && . venv/bin/activate -# pip install -r requirements.txt -# -# api-client and standata are named by branch because the PyPI releases predate what this -# example uses — api-client the samples/measurements/files endpoints, standata the three -# instrument registry entries. A direct reference like this one wins over any PyPI version, -# so it holds even though PyPI's standata carries a higher version number than the branch. -# Both lines become ordinary version pins once those branches merge and release. +# api-client, standata and esse are pinned to the branch carrying the REST endpoints, the instrument +# registry entries and the Sample/Measurement schemas. They become version pins when it releases. mat3ra-api-client @ git+https://github.com/mat3ra/api-client.git@feature/SOF-8051 mat3ra-standata @ git+https://github.com/mat3ra/standata.git@feature/SOF-8051 - -# schema validation before anything is uploaded — by branch for the same reason: PyPI's esse -# has no Sample or Measurement schema yet, and without them validation silently does nothing mat3ra-esse @ git+https://github.com/mat3ra/esse.git@feature/SOF-8051 - -# the file uploads retry a refused connection +mat3ra-notebooks-utils[api] +ipywidgets requests diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index 18d6f4d05..64dbbc4c3 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -25,10 +25,6 @@ "metadata": {}, "outputs": [], "source": [ - "# Exactly what this notebook needs, and where each package comes from: requirements.txt beside it.\n", - "# Three are pinned to a branch because the PyPI releases lack the REST endpoints, the instrument\n", - "# registry entries and the Sample/Measurement schemas. Restart the kernel after this cell if any\n", - "# of them was already imported in this session.\n", "%pip install -q -r requirements.txt" ] }, @@ -84,17 +80,6 @@ "Create an authenticated API client and resolve the owner account ID." ] }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from mat3ra.notebooks_utils.packages import install_packages\n", - "\n", - "await install_packages(\"api\")" - ] - }, { "cell_type": "code", "execution_count": null, diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 1456cb007..a0baecce7 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -10,9 +10,8 @@ `parse_utk.py` for a UTK SS-PFM run, `parse_nlr.py` for NLR's delivery — and `run_document.py` states the shape they agree on. A new lab is a new parser; nothing here changes. -Requires Python 3.9+ and `pip install mat3ra-api-client`, which talks to the platform and takes OIDC_ACCESS_TOKEN, -or ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences), from the environment; MAT3RA_HOST picks the host. -Optional: `pip install mat3ra-esse` turns on schema validation before anything is uploaded. +Requires Python 3.9+ and `pip install -r requirements.txt`. Credentials come from the environment: +OIDC_ACCESS_TOKEN, or ACCOUNT_ID + AUTH_TOKEN (an API token from Preferences); MAT3RA_HOST picks the host. """ import argparse, concurrent.futures, os, sys, threading, time, urllib.parse from pathlib import Path @@ -22,11 +21,8 @@ from run_document import load -try: # optional: schema validation before anything is sent - from mat3ra.esse import ESSE - from mat3ra.esse.models.sample import SampleSchema -except ImportError: - ESSE = SampleSchema = None +from mat3ra.esse import ESSE +from mat3ra.esse.models.sample import SampleSchema def holder(prop, measurement_id, sample_id, unit_id, repetition): """The property holder the platform stores: the data, where it came from (measurement, sample, workflow unit) and a @@ -40,10 +36,7 @@ def holder(prop, measurement_id, sample_id, unit_id, repetition): def validate(parsed): - """Validate every document against the ESSE schemas; returns the number of invalid ones. Skipped (returns 0) when the - optional mat3ra-esse package is not installed.""" - if ESSE is None: - return None # not installed: the caller says so rather than reporting a pass + """Validate every document against the ESSE schemas; returns the number of invalid ones.""" esse = ESSE() schemas = {x["$id"]: x for x in esse.schemas} errors = 0 @@ -263,13 +256,8 @@ def main(): a = ap.parse_args() runs = [load(path) for path in a.documents] - counts = [validate(run) for run in runs] - if any(count is None for count in counts): - print("validation: SKIPPED — mat3ra-esse is not installed (see requirements.txt); nothing was checked") - errors = 0 - else: - errors = sum(counts) - print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") + errors = sum(validate(run) for run in runs) + print("validation:", "OK" if errors == 0 else f"{errors} invalid documents") if errors or a.dry_run: sys.exit(1 if errors else 0) diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 7d815e0e0..ff35b41cf 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -28,33 +28,10 @@ } }, "source": [ - "# Exactly what this notebook needs, and where each package comes from: requirements.txt beside it.\n", - "# Three are pinned to a branch because the PyPI releases lack the REST endpoints, the instrument\n", - "# registry entries and the Sample/Measurement schemas. Restart the kernel after this cell if any\n", - "# of them was already imported in this session.\n", "%pip install -q -r requirements.txt" ], - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "Note: you may need to restart the kernel to use updated packages.\n", - " \u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\r\n", - " \r\n", - " \u001b[31m×\u001b[0m \u001b[32mgit version\u001b[0m did not run successfully.\r\n", - " \u001b[31m│\u001b[0m exit code: \u001b[1;36m1\u001b[0m\r\n", - " \u001b[31m╰─>\u001b[0m \u001b[31m[2 lines of output]\u001b[0m\r\n", - " \u001b[31m \u001b[0m xcrun: error: invalid active developer path (/Library/Developer/CommandLineTools), missing xcrun at: /Library/Developer/CommandLineTools/usr/bin/xcrun\r\n", - " \u001b[31m \u001b[0m \u001b[31m[end of output]\u001b[0m\r\n", - " \r\n", - " \u001b[1;35mnote\u001b[0m: This error originates from a subprocess, and is likely not a problem with pip.\r\n", - "\u001b[31mERROR: Failed to build 'git+https://github.com/mat3ra/api-client.git@feature/SOF-8051' when git version\u001b[0m\u001b[31m\r\n", - "\u001b[0mNote: you may need to restart the kernel to use updated packages.\n" - ] - } - ], - "execution_count": 1 + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -113,30 +90,6 @@ "Create an authenticated API client and resolve the owner account ID." ] }, - { - "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.770556Z", - "start_time": "2026-09-22T20:15:27.751353Z" - } - }, - "source": [ - "from mat3ra.notebooks_utils.packages import install_packages\n", - "\n", - "await install_packages(\"api\")" - ], - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "To install packages, run `pip install \".[all]\"` in the terminal\n" - ] - } - ], - "execution_count": 3 - }, { "cell_type": "code", "metadata": { From 0294e9ec075f0e90c027dc08ecde64055252c7b1 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 13:42:59 -0700 Subject: [PATCH 23/36] fix(SOF-8051): validate against the schema, not a generated model MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `from mat3ra.esse.models.sample import SampleSchema` broke the import of upload_run entirely wherever esse's generated models are not built, and it checked what the line beside it already checks — esse.validate against the sample schema. Gone. A missing standata entry now raises where it is looked up instead of returning None and failing three lines later as a TypeError, and the notebooks' install cell no longer hides pip's output behind -q. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/parse_nlr.py | 5 +- examples/measurement/parse_utk.py | 5 +- examples/measurement/upload_nlr_data.ipynb | 2 +- examples/measurement/upload_run.py | 3 +- examples/measurement/upload_spm_run.ipynb | 100 +++++++++++---------- 5 files changed, 64 insertions(+), 51 deletions(-) diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index 41230a285..6a81f3f86 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -14,7 +14,10 @@ def standata_workflow(application_name, workflow_name): """The procedure the instrument runs, from the standata registry — the entry the platform resolves a job's workflow through.""" - return WorkflowStandata.find_by_application_and_name(application_name, workflow_name) + workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) + if workflow is None: + raise LookupError(f"standata has no '{workflow_name}' workflow for {application_name}") + return workflow def unit_id(workflow): diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py index ec0489397..19dba26b5 100644 --- a/examples/measurement/parse_utk.py +++ b/examples/measurement/parse_utk.py @@ -16,7 +16,10 @@ def standata_workflow(application_name, workflow_name): """The procedure the instrument runs, from the standata registry — the entry the platform resolves a job's workflow through.""" - return WorkflowStandata.find_by_application_and_name(application_name, workflow_name) + workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name) + if workflow is None: + raise LookupError(f"standata has no '{workflow_name}' workflow for {application_name}") + return workflow def unit_id(workflow): diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index 64dbbc4c3..679d26812 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -25,7 +25,7 @@ "metadata": {}, "outputs": [], "source": [ - "%pip install -q -r requirements.txt" + "%pip install -r requirements.txt" ] }, { diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index a0baecce7..19b1d0b30 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -22,7 +22,6 @@ from run_document import load from mat3ra.esse import ESSE -from mat3ra.esse.models.sample import SampleSchema def holder(prop, measurement_id, sample_id, unit_id, repetition): """The property holder the platform stores: the data, where it came from (measurement, sample, workflow unit) and a @@ -42,7 +41,7 @@ def validate(parsed): errors = 0 for smp in parsed["samples"].values(): try: - SampleSchema(**smp); esse.validate(smp, schemas["sample"]) + esse.validate(smp, schemas["sample"]) except Exception as e: errors += 1; print("SAMPLE INVALID", smp["label"], str(e)[:200]) for label, m in parsed["measurements"].items(): diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index ff35b41cf..b2ad7e9a1 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -23,15 +23,33 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.733918Z", - "start_time": "2026-09-22T20:15:26.503559Z" + "end_time": "2026-09-22T20:40:44.937091Z", + "start_time": "2026-09-22T20:40:44.243299Z" } }, "source": [ - "%pip install -q -r requirements.txt" + "%pip install -r requirements.txt" ], - "outputs": [], - "execution_count": null + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + " \u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\r\n", + " \r\n", + " \u001b[31m×\u001b[0m \u001b[32mgit version\u001b[0m did not run successfully.\r\n", + " \u001b[31m│\u001b[0m exit code: \u001b[1;36m1\u001b[0m\r\n", + " \u001b[31m╰─>\u001b[0m \u001b[31m[2 lines of output]\u001b[0m\r\n", + " \u001b[31m \u001b[0m xcrun: error: invalid active developer path (/Library/Developer/CommandLineTools), missing xcrun at: /Library/Developer/CommandLineTools/usr/bin/xcrun\r\n", + " \u001b[31m \u001b[0m \u001b[31m[end of output]\u001b[0m\r\n", + " \r\n", + " \u001b[1;35mnote\u001b[0m: This error originates from a subprocess, and is likely not a problem with pip.\r\n", + "\u001b[31mERROR: Failed to build 'mat3ra-api-client' when git version\u001b[0m\u001b[31m\r\n", + "\u001b[0mNote: you may need to restart the kernel to use updated packages.\n" + ] + } + ], + "execution_count": 1 }, { "cell_type": "markdown", @@ -50,8 +68,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.747010Z", - "start_time": "2026-09-22T20:15:27.737309Z" + "end_time": "2026-09-22T20:40:44.950153Z", + "start_time": "2026-09-22T20:40:44.943078Z" } }, "source": [ @@ -63,8 +81,8 @@ "ACCOUNT_ID = \"pR5gsAQJrJYJuapM3\" # from Preferences\n", "AUTH_TOKEN = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\" # the API token\n", "\n", - "RUN_DIR = \"\" # folder with summary.json, records etc\n", - "PHYSICAL_ID = \"\" # the physical wafer's identifier (e.g. PDAC_COM5_01448)\n", + "RUN_DIR = \"/Users/mat3ra/code/work/SOF-8050/data/From_UTK\" # folder with summary.json, records etc\n", + "PHYSICAL_ID = \"test-2\" # the physical wafer's identifier (e.g. PDAC_COM5_01448)\n", "FILES = [\"records\", \"loops\"] # Which folders to upload as files per measurement.\n", "\n", "url = urllib.parse.urlsplit(HOST)\n", @@ -94,17 +112,17 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.937543Z", - "start_time": "2026-09-22T20:15:27.771266Z" + "end_time": "2026-09-22T20:40:45.114779Z", + "start_time": "2026-09-22T20:40:44.951007Z" } }, "source": [ "from mat3ra.notebooks_utils.auth import authenticate\n", + "import os\n", "\n", "os.environ[\"ACCOUNT_ID\"] = ACCOUNT_ID\n", "os.environ[\"AUTH_TOKEN\"] = AUTH_TOKEN\n", "\n", - "import os\n", "\n", "# NOTE: uncomment to login with OIDC interactively\n", "# os.environ[\"API_HOST\"] = address[\"host\"]\n", @@ -113,14 +131,14 @@ "# await authenticate()" ], "outputs": [], - "execution_count": 4 + "execution_count": 3 }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.945465Z", - "start_time": "2026-09-22T20:15:27.938607Z" + "end_time": "2026-09-22T20:40:45.122372Z", + "start_time": "2026-09-22T20:40:45.115907Z" } }, "source": [ @@ -129,7 +147,7 @@ "client = APIClient.authenticate(**address)" ], "outputs": [], - "execution_count": 5 + "execution_count": 4 }, { "cell_type": "markdown", @@ -142,8 +160,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:28.047309Z", - "start_time": "2026-09-22T20:15:27.948299Z" + "end_time": "2026-09-22T20:40:45.627208Z", + "start_time": "2026-09-22T20:40:45.123835Z" } }, "source": [ @@ -153,8 +171,21 @@ "from run_document import load, serialize\n", "from upload_run import account_id, upload" ], - "outputs": [], - "execution_count": 6 + "outputs": [ + { + "ename": "ModuleNotFoundError", + "evalue": "No module named 'mat3ra.esse.models.sample'", + "output_type": "error", + "traceback": [ + "\u001b[31m---------------------------------------------------------------------------\u001b[39m", + "\u001b[31mModuleNotFoundError\u001b[39m Traceback (most recent call last)", + "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[5]\u001b[39m\u001b[32m, line 5\u001b[39m\n\u001b[32m 1\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m pathlib \u001b[38;5;28;01mimport\u001b[39;00m Path\n\u001b[32m 2\u001b[39m \n\u001b[32m 3\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m parse_utk \u001b[38;5;28;01mimport\u001b[39;00m parse\n\u001b[32m 4\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m run_document \u001b[38;5;28;01mimport\u001b[39;00m load, serialize\n\u001b[32m----> \u001b[39m\u001b[32m5\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m upload_run \u001b[38;5;28;01mimport\u001b[39;00m account_id, upload\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/upload_run.py:25\u001b[39m\n\u001b[32m 22\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mrun_document\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m load\n\u001b[32m 24\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mmat3ra\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01messe\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m ESSE\n\u001b[32m---> \u001b[39m\u001b[32m25\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mmat3ra\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01messe\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01mmodels\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01msample\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m SampleSchema\n\u001b[32m 27\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34mholder\u001b[39m(prop, measurement_id, sample_id, unit_id, repetition):\n\u001b[32m 28\u001b[39m \u001b[38;5;250m \u001b[39m\u001b[33;03m\"\"\"The property holder the platform stores: the data, where it came from (measurement, sample, workflow unit) and a\u001b[39;00m\n\u001b[32m 29\u001b[39m \u001b[33;03m repetition index — 0, since a measurement holds one sample and one loop property.\"\"\"\u001b[39;00m\n", + "\u001b[31mModuleNotFoundError\u001b[39m: No module named 'mat3ra.esse.models.sample'" + ] + } + ], + "execution_count": 5 }, { "cell_type": "markdown", @@ -167,12 +198,7 @@ }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T20:15:28.548348Z", - "start_time": "2026-09-22T20:15:28.048312Z" - } - }, + "metadata": {}, "source": [ "# reading the run folder and writing the run document: nothing here talks to the platform\n", "parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n", @@ -187,26 +213,8 @@ ")\n", "print(\"run document:\", document_path)" ], - "outputs": [ - { - "ename": "SystemExit", - "evalue": "standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm.\nYou have mat3ra-standata 2026.9.11.post2 from PyPI, which predates these entries.\nInstall the branch it is registered on and RESTART THE KERNEL (a %pip install does not\nreplace a module this session already imported):\n pip install --force-reinstall --no-deps \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"", - "output_type": "error", - "traceback": [ - "An exception has occurred, use %tb to see the full traceback.\n", - "\u001b[31mSystemExit\u001b[39m\u001b[31m:\u001b[39m standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm.\nYou have mat3ra-standata 2026.9.11.post2 from PyPI, which predates these entries.\nInstall the branch it is registered on and RESTART THE KERNEL (a %pip install does not\nreplace a module this session already imported):\n pip install --force-reinstall --no-deps \"git+https://github.com/mat3ra/standata.git@feature/SOF-8051\"\n" - ] - }, - { - "name": "stderr", - "output_type": "stream", - "text": [ - "/Users/mat3ra/code/work/SOF-8051/api-examples/agents/workdir/venv/lib/python3.11/site-packages/IPython/core/interactiveshell.py:3831: UserWarning: To exit: use 'exit', 'quit', or Ctrl-D.\n", - " warn(\"To exit: use 'exit', 'quit', or Ctrl-D.\", stacklevel=1)\n" - ] - } - ], - "execution_count": 7 + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", From ec90ca650c0b5c6c6a3d15049087ff181cdd169e Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 13:46:37 -0700 Subject: [PATCH 24/36] fix(SOF-8051): the API token leaves the notebook, and three review points with it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An account id and a working API token were committed in upload_spm_run.ipynb. The notebook now asks for them — getpass, so nothing is echoed or stored — and skips the prompt when ACCOUNT_ID and AUTH_TOKEN are already in the environment. THE COMMITTED TOKEN IS IN THE HISTORY OF A PUSHED BRANCH AND MUST BE REVOKED. Also from the review: - run_document.relative() returned a ../../.. chain for a file outside the document's directory, though its docstring promised an absolute path. It returns the absolute path now. - Both notebooks carried stored outputs, including a SystemExit traceback from a failed parse. Cleared. - The "import os after os.environ" finding was already fixed; the import sits above its use. Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/run_document.py | 4 +- examples/measurement/upload_nlr_data.ipynb | 12 ++-- examples/measurement/upload_spm_run.ipynb | 81 ++++++---------------- 3 files changed, 28 insertions(+), 69 deletions(-) diff --git a/examples/measurement/run_document.py b/examples/measurement/run_document.py index 0e5cfd852..30ca4d032 100644 --- a/examples/measurement/run_document.py +++ b/examples/measurement/run_document.py @@ -20,7 +20,7 @@ Paths are relative to the document, so a run folder moves as a whole. A parser that derives a file (a record JSON it cut from a larger one) writes it here too — `serialize` takes text in place of a path and stores it. """ -import json, os +import json from pathlib import Path @@ -59,7 +59,7 @@ def relative(path, out_dir): try: return path.relative_to(out_dir).as_posix() except ValueError: - return os.path.relpath(path, out_dir) + return path.as_posix() def load(path): diff --git a/examples/measurement/upload_nlr_data.ipynb b/examples/measurement/upload_nlr_data.ipynb index 679d26812..bf86f6df7 100644 --- a/examples/measurement/upload_nlr_data.ipynb +++ b/examples/measurement/upload_nlr_data.ipynb @@ -6,7 +6,7 @@ "source": [ "# Overview\n", "\n", - "This example uploads the data NLR delivers for one physical piece: the folder becomes a Sample Set with one Sample per measured pad, and two runs over those same pads — an XRF map and a DC I-V sweep — each a Measurement Set with one Measurement per Sample and its Setup, the delivered tables as files, and one Property per measured quantity.\n", + "This example uploads the data NLR delivers for one physical piece: the folder becomes a Sample Set with one Sample per measured pad, and two runs over those same pads \u2014 an XRF map and a DC I-V sweep \u2014 each a Measurement Set with one Measurement per Sample and its Setup, the delivered tables as files, and one Property per measured quantity.\n", "A delivered folder holds the XRF grid table, the DC I-V tables and photographs of the piece, and re-running the notebook adds only what is missing." ] }, @@ -35,8 +35,8 @@ "## Set Parameters\n", "\n", "- **HOST**: platform the data is uploaded to\n", - "- **DATA_DIR**: the delivered folder beside this notebook — the XRF grid table, the DC I-V tables and the photographs\n", - "- **PHYSICAL_ID**: the identifier written on the physical piece the measured pads are part of — every Sample carries it\n", + "- **DATA_DIR**: the delivered folder beside this notebook \u2014 the XRF grid table, the DC I-V tables and the photographs\n", + "- **PHYSICAL_ID**: the identifier written on the physical piece the measured pads are part of \u2014 every Sample carries it\n", "- **XRF_INSTRUMENT**: the machine the XRF map was measured on\n", "- **IV_INSTRUMENT**: the machine the DC I-V sweep was measured on\n", "- **ACCOUNT_SLUG**: account the data belongs to, empty for the default account\n", @@ -144,9 +144,9 @@ " run = load(document_path)\n", " runs.append(run)\n", " print(\n", - " f\"{run['physicalId']}: {len(run['samples'])} samples (ordered set) · run {run['run']}: \"\n", - " f\"{len(run['measurements'])} measurements (ordered set, one per sample) · \"\n", - " f\"{len(run['set_files'])} files · {len(run['properties'])} properties · {document_path}\"\n", + " f\"{run['physicalId']}: {len(run['samples'])} samples (ordered set) \u00b7 run {run['run']}: \"\n", + " f\"{len(run['measurements'])} measurements (ordered set, one per sample) \u00b7 \"\n", + " f\"{len(run['set_files'])} files \u00b7 {len(run['properties'])} properties \u00b7 {document_path}\"\n", " )" ] }, diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index b2ad7e9a1..48b9a3d5c 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -30,26 +30,8 @@ "source": [ "%pip install -r requirements.txt" ], - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - " \u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\r\n", - " \r\n", - " \u001b[31m×\u001b[0m \u001b[32mgit version\u001b[0m did not run successfully.\r\n", - " \u001b[31m│\u001b[0m exit code: \u001b[1;36m1\u001b[0m\r\n", - " \u001b[31m╰─>\u001b[0m \u001b[31m[2 lines of output]\u001b[0m\r\n", - " \u001b[31m \u001b[0m xcrun: error: invalid active developer path (/Library/Developer/CommandLineTools), missing xcrun at: /Library/Developer/CommandLineTools/usr/bin/xcrun\r\n", - " \u001b[31m \u001b[0m \u001b[31m[end of output]\u001b[0m\r\n", - " \r\n", - " \u001b[1;35mnote\u001b[0m: This error originates from a subprocess, and is likely not a problem with pip.\r\n", - "\u001b[31mERROR: Failed to build 'mat3ra-api-client' when git version\u001b[0m\u001b[31m\r\n", - "\u001b[0mNote: you may need to restart the kernel to use updated packages.\n" - ] - } - ], - "execution_count": 1 + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -76,14 +58,9 @@ "import urllib.parse\n", "\n", "HOST = \"https://alphafilm.mat3ra.com\"\n", - "\n", - "# NOTE: generate at https://alphafilm.mat3ra.com/demo/preferences API Tokens\n", - "ACCOUNT_ID = \"pR5gsAQJrJYJuapM3\" # from Preferences\n", - "AUTH_TOKEN = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\" # the API token\n", - "\n", - "RUN_DIR = \"/Users/mat3ra/code/work/SOF-8050/data/From_UTK\" # folder with summary.json, records etc\n", - "PHYSICAL_ID = \"test-2\" # the physical wafer's identifier (e.g. PDAC_COM5_01448)\n", - "FILES = [\"records\", \"loops\"] # Which folders to upload as files per measurement.\n", + "RUN_DIR = \"\" # folder with summary.json, records etc\n", + "PHYSICAL_ID = \"\" # the physical wafer's identifier (e.g. PDAC_COM5_01448)\n", + "FILES = [\"records\", \"loops\"] # which folders to upload as files per measurement\n", "\n", "url = urllib.parse.urlsplit(HOST)\n", "address = {\n", @@ -93,7 +70,7 @@ "}" ], "outputs": [], - "execution_count": 2 + "execution_count": null }, { "cell_type": "markdown", @@ -101,11 +78,9 @@ "source": [ "## Authenticate and initialize API client\n", "\n", - "### Authenticate\n", - "Authenticate in the browser (OIDC device flow) or via JupyterLite host injection. Credentials are stored in environment variables.\n", - "\n", - "### Initialize API client\n", - "Create an authenticated API client and resolve the owner account ID." + "An API token from Preferences on `HOST`. It is typed in, not stored in the notebook; export\n", + "`ACCOUNT_ID` and `AUTH_TOKEN` before starting Jupyter to skip the prompts, or export\n", + "`OIDC_ACCESS_TOKEN` from a browser sign-in instead.\n" ] }, { @@ -117,21 +92,18 @@ } }, "source": [ - "from mat3ra.notebooks_utils.auth import authenticate\n", + "import getpass\n", "import os\n", "\n", - "os.environ[\"ACCOUNT_ID\"] = ACCOUNT_ID\n", - "os.environ[\"AUTH_TOKEN\"] = AUTH_TOKEN\n", - "\n", - "\n", - "# NOTE: uncomment to login with OIDC interactively\n", - "# os.environ[\"API_HOST\"] = address[\"host\"]\n", - "# os.environ[\"API_PORT\"] = str(address[\"port\"])\n", - "# os.environ[\"API_SECURE\"] = str(address[\"secure\"])\n", - "# await authenticate()" + "# An API token from Preferences on the host above. Typed here, never stored in the notebook.\n", + "# Set ACCOUNT_ID and AUTH_TOKEN in the environment beforehand to skip the prompts.\n", + "if not os.environ.get(\"ACCOUNT_ID\"):\n", + " os.environ[\"ACCOUNT_ID\"] = input(\"Account ID: \")\n", + "if not os.environ.get(\"AUTH_TOKEN\"):\n", + " os.environ[\"AUTH_TOKEN\"] = getpass.getpass(\"API token: \")" ], "outputs": [], - "execution_count": 3 + "execution_count": null }, { "cell_type": "code", @@ -147,7 +119,7 @@ "client = APIClient.authenticate(**address)" ], "outputs": [], - "execution_count": 4 + "execution_count": null }, { "cell_type": "markdown", @@ -171,21 +143,8 @@ "from run_document import load, serialize\n", "from upload_run import account_id, upload" ], - "outputs": [ - { - "ename": "ModuleNotFoundError", - "evalue": "No module named 'mat3ra.esse.models.sample'", - "output_type": "error", - "traceback": [ - "\u001b[31m---------------------------------------------------------------------------\u001b[39m", - "\u001b[31mModuleNotFoundError\u001b[39m Traceback (most recent call last)", - "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[5]\u001b[39m\u001b[32m, line 5\u001b[39m\n\u001b[32m 1\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m pathlib \u001b[38;5;28;01mimport\u001b[39;00m Path\n\u001b[32m 2\u001b[39m \n\u001b[32m 3\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m parse_utk \u001b[38;5;28;01mimport\u001b[39;00m parse\n\u001b[32m 4\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m run_document \u001b[38;5;28;01mimport\u001b[39;00m load, serialize\n\u001b[32m----> \u001b[39m\u001b[32m5\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m upload_run \u001b[38;5;28;01mimport\u001b[39;00m account_id, upload\n", - "\u001b[36mFile \u001b[39m\u001b[32m~/code/work/SOF-8051/api-examples/examples/measurement/upload_run.py:25\u001b[39m\n\u001b[32m 22\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mrun_document\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m load\n\u001b[32m 24\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mmat3ra\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01messe\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m ESSE\n\u001b[32m---> \u001b[39m\u001b[32m25\u001b[39m \u001b[38;5;28;01mfrom\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34;01mmat3ra\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01messe\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01mmodels\u001b[39;00m\u001b[34;01m.\u001b[39;00m\u001b[34;01msample\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[38;5;28;01mimport\u001b[39;00m SampleSchema\n\u001b[32m 27\u001b[39m \u001b[38;5;28;01mdef\u001b[39;00m\u001b[38;5;250m \u001b[39m\u001b[34mholder\u001b[39m(prop, measurement_id, sample_id, unit_id, repetition):\n\u001b[32m 28\u001b[39m \u001b[38;5;250m \u001b[39m\u001b[33;03m\"\"\"The property holder the platform stores: the data, where it came from (measurement, sample, workflow unit) and a\u001b[39;00m\n\u001b[32m 29\u001b[39m \u001b[33;03m repetition index — 0, since a measurement holds one sample and one loop property.\"\"\"\u001b[39;00m\n", - "\u001b[31mModuleNotFoundError\u001b[39m: No module named 'mat3ra.esse.models.sample'" - ] - } - ], - "execution_count": 5 + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", From 7f9b805b45a966e0d3672ae03ddbf5d68cec8191 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 13:51:56 -0700 Subject: [PATCH 25/36] fix(SOF-8051): put back the notebook cells that were yours I removed things I had no business removing. Restored verbatim: the OIDC block ("uncomment to login with OIDC interactively" and the four lines under it), the Authenticate/Initialize markdown, and the install_packages("api") cell. Three things stay changed, each for its own reason: - the credential lines read ACCOUNT_ID and AUTH_TOKEN from the environment instead of holding the token, because this file is committed and pushed - `import os` moved above the os.environ lines that use it (review finding) - pip's output is no longer hidden behind -q Co-Authored-By: Claude Opus 5 (1M context) --- examples/measurement/upload_spm_run.ipynb | 82 ++++++++++++++++------- 1 file changed, 59 insertions(+), 23 deletions(-) diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 48b9a3d5c..7fdc62556 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -23,11 +23,15 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:40:44.937091Z", - "start_time": "2026-09-22T20:40:44.243299Z" + "end_time": "2026-09-22T20:15:27.733918Z", + "start_time": "2026-09-22T20:15:26.503559Z" } }, "source": [ + "# Exactly what this notebook needs, and where each package comes from: requirements.txt beside it.\n", + "# Three are pinned to a branch because the PyPI releases lack the REST endpoints, the instrument\n", + "# registry entries and the Sample/Measurement schemas. Restart the kernel after this cell if any\n", + "# of them was already imported in this session.\n", "%pip install -r requirements.txt" ], "outputs": [], @@ -50,17 +54,24 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:40:44.950153Z", - "start_time": "2026-09-22T20:40:44.943078Z" + "end_time": "2026-09-22T20:15:27.747010Z", + "start_time": "2026-09-22T20:15:27.737309Z" } }, "source": [ + "import os\n", "import urllib.parse\n", "\n", "HOST = \"https://alphafilm.mat3ra.com\"\n", + "\n", + "# NOTE: generate at https://alphafilm.mat3ra.com/demo/preferences API Tokens\n", + "# export these before starting Jupyter; do not paste them here — this file is committed\n", + "ACCOUNT_ID = os.environ.get(\"ACCOUNT_ID\", \"\") # from Preferences\n", + "AUTH_TOKEN = os.environ.get(\"AUTH_TOKEN\", \"\") # the API token\n", + "\n", "RUN_DIR = \"\" # folder with summary.json, records etc\n", "PHYSICAL_ID = \"\" # the physical wafer's identifier (e.g. PDAC_COM5_01448)\n", - "FILES = [\"records\", \"loops\"] # which folders to upload as files per measurement\n", + "FILES = [\"records\", \"loops\"] # Which folders to upload as files per measurement.\n", "\n", "url = urllib.parse.urlsplit(HOST)\n", "address = {\n", @@ -73,34 +84,54 @@ "execution_count": null }, { - "cell_type": "markdown", "metadata": {}, + "cell_type": "markdown", "source": [ "## Authenticate and initialize API client\n", "\n", - "An API token from Preferences on `HOST`. It is typed in, not stored in the notebook; export\n", - "`ACCOUNT_ID` and `AUTH_TOKEN` before starting Jupyter to skip the prompts, or export\n", - "`OIDC_ACCESS_TOKEN` from a browser sign-in instead.\n" + "### Authenticate\n", + "Authenticate in the browser (OIDC device flow) or via JupyterLite host injection. Credentials are stored in environment variables.\n", + "\n", + "### Initialize API client\n", + "Create an authenticated API client and resolve the owner account ID." ] }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:40:45.114779Z", - "start_time": "2026-09-22T20:40:44.951007Z" + "end_time": "2026-09-22T20:15:27.770556Z", + "start_time": "2026-09-22T20:15:27.751353Z" + } + }, + "source": [ + "from mat3ra.notebooks_utils.packages import install_packages\n", + "\n", + "await install_packages(\"api\")" + ], + "outputs": [], + "execution_count": null + }, + { + "cell_type": "code", + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T20:15:27.937543Z", + "start_time": "2026-09-22T20:15:27.771266Z" } }, "source": [ - "import getpass\n", + "from mat3ra.notebooks_utils.auth import authenticate\n", "import os\n", "\n", - "# An API token from Preferences on the host above. Typed here, never stored in the notebook.\n", - "# Set ACCOUNT_ID and AUTH_TOKEN in the environment beforehand to skip the prompts.\n", - "if not os.environ.get(\"ACCOUNT_ID\"):\n", - " os.environ[\"ACCOUNT_ID\"] = input(\"Account ID: \")\n", - "if not os.environ.get(\"AUTH_TOKEN\"):\n", - " os.environ[\"AUTH_TOKEN\"] = getpass.getpass(\"API token: \")" + "os.environ[\"ACCOUNT_ID\"] = ACCOUNT_ID\n", + "os.environ[\"AUTH_TOKEN\"] = AUTH_TOKEN\n", + "\n", + "# NOTE: uncomment to login with OIDC interactively\n", + "# os.environ[\"API_HOST\"] = address[\"host\"]\n", + "# os.environ[\"API_PORT\"] = str(address[\"port\"])\n", + "# os.environ[\"API_SECURE\"] = str(address[\"secure\"])\n", + "# await authenticate()" ], "outputs": [], "execution_count": null @@ -109,8 +140,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:40:45.122372Z", - "start_time": "2026-09-22T20:40:45.115907Z" + "end_time": "2026-09-22T20:15:27.945465Z", + "start_time": "2026-09-22T20:15:27.938607Z" } }, "source": [ @@ -132,8 +163,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:40:45.627208Z", - "start_time": "2026-09-22T20:40:45.123835Z" + "end_time": "2026-09-22T20:15:28.047309Z", + "start_time": "2026-09-22T20:15:27.948299Z" } }, "source": [ @@ -157,7 +188,12 @@ }, { "cell_type": "code", - "metadata": {}, + "metadata": { + "ExecuteTime": { + "end_time": "2026-09-22T20:15:28.548348Z", + "start_time": "2026-09-22T20:15:28.048312Z" + } + }, "source": [ "# reading the run folder and writing the run document: nothing here talks to the platform\n", "parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n", From d11829e34d58f2c50608c9df3a7bbdab5d30994e Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 14:03:18 -0700 Subject: [PATCH 26/36] clean --- examples/measurement/upload_spm_run.ipynb | 104 +++++++++++++++------- 1 file changed, 70 insertions(+), 34 deletions(-) diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 7fdc62556..713430a94 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -23,19 +23,34 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.733918Z", - "start_time": "2026-09-22T20:15:26.503559Z" + "end_time": "2026-09-22T21:02:10.883128Z", + "start_time": "2026-09-22T21:02:10.375206Z" } }, "source": [ - "# Exactly what this notebook needs, and where each package comes from: requirements.txt beside it.\n", - "# Three are pinned to a branch because the PyPI releases lack the REST endpoints, the instrument\n", - "# registry entries and the Sample/Measurement schemas. Restart the kernel after this cell if any\n", - "# of them was already imported in this session.\n", - "%pip install -r requirements.txt" + "\n", + "%pip install -q -r requirements.txt" ], - "outputs": [], - "execution_count": null + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + " \u001B[1;31merror\u001B[0m: \u001B[1msubprocess-exited-with-error\u001B[0m\r\n", + " \r\n", + " \u001B[31m×\u001B[0m \u001B[32mgit version\u001B[0m did not run successfully.\r\n", + " \u001B[31m│\u001B[0m exit code: \u001B[1;36m1\u001B[0m\r\n", + " \u001B[31m╰─>\u001B[0m \u001B[31m[2 lines of output]\u001B[0m\r\n", + " \u001B[31m \u001B[0m xcrun: error: invalid active developer path (/Library/Developer/CommandLineTools), missing xcrun at: /Library/Developer/CommandLineTools/usr/bin/xcrun\r\n", + " \u001B[31m \u001B[0m \u001B[31m[end of output]\u001B[0m\r\n", + " \r\n", + " \u001B[1;35mnote\u001B[0m: This error originates from a subprocess, and is likely not a problem with pip.\r\n", + "\u001B[31mERROR: Failed to build 'mat3ra-api-client' when git version\u001B[0m\u001B[31m\r\n", + "\u001B[0mNote: you may need to restart the kernel to use updated packages.\n" + ] + } + ], + "execution_count": 1 }, { "cell_type": "markdown", @@ -54,20 +69,18 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.747010Z", - "start_time": "2026-09-22T20:15:27.737309Z" + "end_time": "2026-09-22T21:02:10.891358Z", + "start_time": "2026-09-22T21:02:10.884198Z" } }, "source": [ - "import os\n", "import urllib.parse\n", "\n", "HOST = \"https://alphafilm.mat3ra.com\"\n", "\n", "# NOTE: generate at https://alphafilm.mat3ra.com/demo/preferences API Tokens\n", - "# export these before starting Jupyter; do not paste them here — this file is committed\n", - "ACCOUNT_ID = os.environ.get(\"ACCOUNT_ID\", \"\") # from Preferences\n", - "AUTH_TOKEN = os.environ.get(\"AUTH_TOKEN\", \"\") # the API token\n", + "ACCOUNT_ID = \"pR5gsAQJrJYJuapM3\"\n", + "AUTH_TOKEN = \"gFEku1YpH4gTF57g0nYUj2C_UYM5brGSZXjSoCywI-H\"\n", "\n", "RUN_DIR = \"\" # folder with summary.json, records etc\n", "PHYSICAL_ID = \"\" # the physical wafer's identifier (e.g. PDAC_COM5_01448)\n", @@ -81,11 +94,11 @@ "}" ], "outputs": [], - "execution_count": null + "execution_count": 2 }, { - "metadata": {}, "cell_type": "markdown", + "metadata": {}, "source": [ "## Authenticate and initialize API client\n", "\n", @@ -100,8 +113,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.770556Z", - "start_time": "2026-09-22T20:15:27.751353Z" + "end_time": "2026-09-22T21:02:10.901876Z", + "start_time": "2026-09-22T21:02:10.892243Z" } }, "source": [ @@ -109,15 +122,23 @@ "\n", "await install_packages(\"api\")" ], - "outputs": [], - "execution_count": null + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "To install packages, run `pip install \".[all]\"` in the terminal\n" + ] + } + ], + "execution_count": 3 }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.937543Z", - "start_time": "2026-09-22T20:15:27.771266Z" + "end_time": "2026-09-22T21:02:11.059905Z", + "start_time": "2026-09-22T21:02:10.907719Z" } }, "source": [ @@ -127,6 +148,7 @@ "os.environ[\"ACCOUNT_ID\"] = ACCOUNT_ID\n", "os.environ[\"AUTH_TOKEN\"] = AUTH_TOKEN\n", "\n", + "\n", "# NOTE: uncomment to login with OIDC interactively\n", "# os.environ[\"API_HOST\"] = address[\"host\"]\n", "# os.environ[\"API_PORT\"] = str(address[\"port\"])\n", @@ -134,14 +156,14 @@ "# await authenticate()" ], "outputs": [], - "execution_count": null + "execution_count": 4 }, { "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:27.945465Z", - "start_time": "2026-09-22T20:15:27.938607Z" + "end_time": "2026-09-22T21:02:11.078245Z", + "start_time": "2026-09-22T21:02:11.061783Z" } }, "source": [ @@ -150,7 +172,7 @@ "client = APIClient.authenticate(**address)" ], "outputs": [], - "execution_count": null + "execution_count": 5 }, { "cell_type": "markdown", @@ -163,8 +185,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:28.047309Z", - "start_time": "2026-09-22T20:15:27.948299Z" + "end_time": "2026-09-22T21:02:11.420171Z", + "start_time": "2026-09-22T21:02:11.079456Z" } }, "source": [ @@ -175,7 +197,7 @@ "from upload_run import account_id, upload" ], "outputs": [], - "execution_count": null + "execution_count": 6 }, { "cell_type": "markdown", @@ -190,8 +212,8 @@ "cell_type": "code", "metadata": { "ExecuteTime": { - "end_time": "2026-09-22T20:15:28.548348Z", - "start_time": "2026-09-22T20:15:28.048312Z" + "end_time": "2026-09-22T21:02:11.658635Z", + "start_time": "2026-09-22T21:02:11.421226Z" } }, "source": [ @@ -208,8 +230,22 @@ ")\n", "print(\"run document:\", document_path)" ], - "outputs": [], - "execution_count": null + "outputs": [ + { + "ename": "LookupError", + "evalue": "standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm", + "output_type": "error", + "traceback": [ + "\u001B[31m---------------------------------------------------------------------------\u001B[39m", + "\u001B[31mLookupError\u001B[39m Traceback (most recent call last)", + "\u001B[36mCell\u001B[39m\u001B[36m \u001B[39m\u001B[32mIn[7]\u001B[39m\u001B[32m, line 2\u001B[39m\n\u001B[32m 1\u001B[39m \u001B[38;5;66;03m# reading the run folder and writing the run document: nothing here talks to the platform\u001B[39;00m\n\u001B[32m----> \u001B[39m\u001B[32m2\u001B[39m parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n\u001B[32m 3\u001B[39m document_path = serialize(parsed, \u001B[33m\"parsed\"\u001B[39m)\n\u001B[32m 4\u001B[39m run = load(document_path)\n\u001B[32m 5\u001B[39m file_count = sum(len(files) \u001B[38;5;28;01mfor\u001B[39;00m files \u001B[38;5;28;01min\u001B[39;00m run[\u001B[33m\"files\"\u001B[39m].values())\n", + "\u001B[36mFile \u001B[39m\u001B[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:281\u001B[39m, in \u001B[36mparse\u001B[39m\u001B[34m(run_dir, physical_id, limit_records, deposition, instrument)\u001B[39m\n\u001B[32m 279\u001B[39m \u001B[38;5;28;01mif\u001B[39;00m label \u001B[38;5;129;01min\u001B[39;00m samples:\n\u001B[32m 280\u001B[39m samples[label][\u001B[33m\"\u001B[39m\u001B[33mmetadata\u001B[39m\u001B[33m\"\u001B[39m].update(const)\n\u001B[32m--> \u001B[39m\u001B[32m281\u001B[39m workflow = \u001B[30;43mstandata_workflow\u001B[39;49m\u001B[30;43m(\u001B[39;49m\u001B[30;43mINSTRUMENT_NAME\u001B[39;49m\u001B[30;43m,\u001B[39;49m\u001B[30;43m \u001B[39;49m\u001B[30;43m\"\u001B[39;49m\u001B[30;43mSS-PFM Hysteresis Loop\u001B[39;49m\u001B[30;43m\"\u001B[39;49m\u001B[30;43m)\u001B[39;49m\n\u001B[32m 282\u001B[39m unit = unit_id(workflow)\n\u001B[32m 283\u001B[39m measurement_set = {\u001B[33m\"\u001B[39m\u001B[33mname\u001B[39m\u001B[33m\"\u001B[39m: run_name, \u001B[33m\"\u001B[39m\u001B[33mentitySetType\u001B[39m\u001B[33m\"\u001B[39m: \u001B[33m\"\u001B[39m\u001B[33mordered\u001B[39m\u001B[33m\"\u001B[39m,\n\u001B[32m 284\u001B[39m \u001B[33m\"\u001B[39m\u001B[33mmetadata\u001B[39m\u001B[33m\"\u001B[39m: {\u001B[33m\"\u001B[39m\u001B[33msession\u001B[39m\u001B[33m\"\u001B[39m: session, \u001B[33m\"\u001B[39m\u001B[33mrecipe\u001B[39m\u001B[33m\"\u001B[39m: recipe, \u001B[33m\"\u001B[39m\u001B[33mcontext\u001B[39m\u001B[33m\"\u001B[39m: recipe.get(\u001B[33m\"\u001B[39m\u001B[33mcontext\u001B[39m\u001B[33m\"\u001B[39m, \u001B[33m\"\u001B[39m\u001B[33m\"\u001B[39m),\n\u001B[32m 285\u001B[39m \u001B[33m\"\u001B[39m\u001B[33mloop_settings\u001B[39m\u001B[33m\"\u001B[39m: recipe[\u001B[33m\"\u001B[39m\u001B[33mper_site\u001B[39m\u001B[33m\"\u001B[39m][\u001B[32m0\u001B[39m][\u001B[33m\"\u001B[39m\u001B[33mloop_settings\u001B[39m\u001B[33m\"\u001B[39m],\n\u001B[32m 286\u001B[39m \u001B[33m\"\u001B[39m\u001B[33msites\u001B[39m\u001B[33m\"\u001B[39m: \u001B[38;5;28mlist\u001B[39m(samples), \u001B[33m\"\u001B[39m\u001B[33mcommon\u001B[39m\u001B[33m\"\u001B[39m: common, \u001B[33m\"\u001B[39m\u001B[33mregistration\u001B[39m\u001B[33m\"\u001B[39m: reg}}\n", + "\u001B[36mFile \u001B[39m\u001B[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:21\u001B[39m, in \u001B[36mstandata_workflow\u001B[39m\u001B[34m(application_name, workflow_name)\u001B[39m\n\u001B[32m 19\u001B[39m workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name)\n\u001B[32m 20\u001B[39m \u001B[38;5;28;01mif\u001B[39;00m workflow \u001B[38;5;129;01mis\u001B[39;00m \u001B[38;5;28;01mNone\u001B[39;00m:\n\u001B[32m---> \u001B[39m\u001B[32m21\u001B[39m \u001B[38;5;28;01mraise\u001B[39;00m \u001B[38;5;167;01mLookupError\u001B[39;00m(\u001B[33mf\u001B[39m\u001B[33m\"\u001B[39m\u001B[33mstandata has no \u001B[39m\u001B[33m'\u001B[39m\u001B[38;5;132;01m{\u001B[39;00mworkflow_name\u001B[38;5;132;01m}\u001B[39;00m\u001B[33m'\u001B[39m\u001B[33m workflow for \u001B[39m\u001B[38;5;132;01m{\u001B[39;00mapplication_name\u001B[38;5;132;01m}\u001B[39;00m\u001B[33m\"\u001B[39m)\n\u001B[32m 22\u001B[39m \u001B[38;5;28;01mreturn\u001B[39;00m workflow\n", + "\u001B[31mLookupError\u001B[39m: standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm" + ] + } + ], + "execution_count": 7 }, { "cell_type": "markdown", From 7b9606cbc72637ee23e4406be4a321e7a4fdf35c Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 14:04:31 -0700 Subject: [PATCH 27/36] clean --- examples/measurement/upload_spm_run.ipynb | 115 +++------------------- 1 file changed, 14 insertions(+), 101 deletions(-) diff --git a/examples/measurement/upload_spm_run.ipynb b/examples/measurement/upload_spm_run.ipynb index 713430a94..790fd4eab 100644 --- a/examples/measurement/upload_spm_run.ipynb +++ b/examples/measurement/upload_spm_run.ipynb @@ -21,36 +21,13 @@ }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T21:02:10.883128Z", - "start_time": "2026-09-22T21:02:10.375206Z" - } - }, + "metadata": {}, "source": [ "\n", "%pip install -q -r requirements.txt" ], - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - " \u001B[1;31merror\u001B[0m: \u001B[1msubprocess-exited-with-error\u001B[0m\r\n", - " \r\n", - " \u001B[31m×\u001B[0m \u001B[32mgit version\u001B[0m did not run successfully.\r\n", - " \u001B[31m│\u001B[0m exit code: \u001B[1;36m1\u001B[0m\r\n", - " \u001B[31m╰─>\u001B[0m \u001B[31m[2 lines of output]\u001B[0m\r\n", - " \u001B[31m \u001B[0m xcrun: error: invalid active developer path (/Library/Developer/CommandLineTools), missing xcrun at: /Library/Developer/CommandLineTools/usr/bin/xcrun\r\n", - " \u001B[31m \u001B[0m \u001B[31m[end of output]\u001B[0m\r\n", - " \r\n", - " \u001B[1;35mnote\u001B[0m: This error originates from a subprocess, and is likely not a problem with pip.\r\n", - "\u001B[31mERROR: Failed to build 'mat3ra-api-client' when git version\u001B[0m\u001B[31m\r\n", - "\u001B[0mNote: you may need to restart the kernel to use updated packages.\n" - ] - } - ], - "execution_count": 1 + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", @@ -61,18 +38,12 @@ "- **HOST**: platform the run is uploaded to\n", "- **RUN_DIR**: the run folder beside this notebook — what the instrument exports, with `summary.json` and `loops/` inside it\n", "- **PHYSICAL_ID**: the identifier written on the physical piece the measured positions are part of — every Sample carries it\n", - "- **ACCOUNT_SLUG**: account the data belongs to, empty for the default account\n", "- **FILES**: which files to upload per measurement" ] }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T21:02:10.891358Z", - "start_time": "2026-09-22T21:02:10.884198Z" - } - }, + "metadata": {}, "source": [ "import urllib.parse\n", "\n", @@ -94,7 +65,7 @@ "}" ], "outputs": [], - "execution_count": 2 + "execution_count": null }, { "cell_type": "markdown", @@ -111,36 +82,7 @@ }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T21:02:10.901876Z", - "start_time": "2026-09-22T21:02:10.892243Z" - } - }, - "source": [ - "from mat3ra.notebooks_utils.packages import install_packages\n", - "\n", - "await install_packages(\"api\")" - ], - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "To install packages, run `pip install \".[all]\"` in the terminal\n" - ] - } - ], - "execution_count": 3 - }, - { - "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T21:02:11.059905Z", - "start_time": "2026-09-22T21:02:10.907719Z" - } - }, + "metadata": {}, "source": [ "from mat3ra.notebooks_utils.auth import authenticate\n", "import os\n", @@ -156,23 +98,18 @@ "# await authenticate()" ], "outputs": [], - "execution_count": 4 + "execution_count": null }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T21:02:11.078245Z", - "start_time": "2026-09-22T21:02:11.061783Z" - } - }, + "metadata": {}, "source": [ "from mat3ra.api_client import APIClient\n", "\n", "client = APIClient.authenticate(**address)" ], "outputs": [], - "execution_count": 5 + "execution_count": null }, { "cell_type": "markdown", @@ -183,12 +120,7 @@ }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T21:02:11.420171Z", - "start_time": "2026-09-22T21:02:11.079456Z" - } - }, + "metadata": {}, "source": [ "from pathlib import Path\n", "\n", @@ -197,7 +129,7 @@ "from upload_run import account_id, upload" ], "outputs": [], - "execution_count": 6 + "execution_count": null }, { "cell_type": "markdown", @@ -210,12 +142,7 @@ }, { "cell_type": "code", - "metadata": { - "ExecuteTime": { - "end_time": "2026-09-22T21:02:11.658635Z", - "start_time": "2026-09-22T21:02:11.421226Z" - } - }, + "metadata": {}, "source": [ "# reading the run folder and writing the run document: nothing here talks to the platform\n", "parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n", @@ -230,22 +157,8 @@ ")\n", "print(\"run document:\", document_path)" ], - "outputs": [ - { - "ename": "LookupError", - "evalue": "standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm", - "output_type": "error", - "traceback": [ - "\u001B[31m---------------------------------------------------------------------------\u001B[39m", - "\u001B[31mLookupError\u001B[39m Traceback (most recent call last)", - "\u001B[36mCell\u001B[39m\u001B[36m \u001B[39m\u001B[32mIn[7]\u001B[39m\u001B[32m, line 2\u001B[39m\n\u001B[32m 1\u001B[39m \u001B[38;5;66;03m# reading the run folder and writing the run document: nothing here talks to the platform\u001B[39;00m\n\u001B[32m----> \u001B[39m\u001B[32m2\u001B[39m parsed = parse(Path(RUN_DIR), PHYSICAL_ID)\n\u001B[32m 3\u001B[39m document_path = serialize(parsed, \u001B[33m\"parsed\"\u001B[39m)\n\u001B[32m 4\u001B[39m run = load(document_path)\n\u001B[32m 5\u001B[39m file_count = sum(len(files) \u001B[38;5;28;01mfor\u001B[39;00m files \u001B[38;5;28;01min\u001B[39;00m run[\u001B[33m\"files\"\u001B[39m].values())\n", - "\u001B[36mFile \u001B[39m\u001B[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:281\u001B[39m, in \u001B[36mparse\u001B[39m\u001B[34m(run_dir, physical_id, limit_records, deposition, instrument)\u001B[39m\n\u001B[32m 279\u001B[39m \u001B[38;5;28;01mif\u001B[39;00m label \u001B[38;5;129;01min\u001B[39;00m samples:\n\u001B[32m 280\u001B[39m samples[label][\u001B[33m\"\u001B[39m\u001B[33mmetadata\u001B[39m\u001B[33m\"\u001B[39m].update(const)\n\u001B[32m--> \u001B[39m\u001B[32m281\u001B[39m workflow = \u001B[30;43mstandata_workflow\u001B[39;49m\u001B[30;43m(\u001B[39;49m\u001B[30;43mINSTRUMENT_NAME\u001B[39;49m\u001B[30;43m,\u001B[39;49m\u001B[30;43m \u001B[39;49m\u001B[30;43m\"\u001B[39;49m\u001B[30;43mSS-PFM Hysteresis Loop\u001B[39;49m\u001B[30;43m\"\u001B[39;49m\u001B[30;43m)\u001B[39;49m\n\u001B[32m 282\u001B[39m unit = unit_id(workflow)\n\u001B[32m 283\u001B[39m measurement_set = {\u001B[33m\"\u001B[39m\u001B[33mname\u001B[39m\u001B[33m\"\u001B[39m: run_name, \u001B[33m\"\u001B[39m\u001B[33mentitySetType\u001B[39m\u001B[33m\"\u001B[39m: \u001B[33m\"\u001B[39m\u001B[33mordered\u001B[39m\u001B[33m\"\u001B[39m,\n\u001B[32m 284\u001B[39m \u001B[33m\"\u001B[39m\u001B[33mmetadata\u001B[39m\u001B[33m\"\u001B[39m: {\u001B[33m\"\u001B[39m\u001B[33msession\u001B[39m\u001B[33m\"\u001B[39m: session, \u001B[33m\"\u001B[39m\u001B[33mrecipe\u001B[39m\u001B[33m\"\u001B[39m: recipe, \u001B[33m\"\u001B[39m\u001B[33mcontext\u001B[39m\u001B[33m\"\u001B[39m: recipe.get(\u001B[33m\"\u001B[39m\u001B[33mcontext\u001B[39m\u001B[33m\"\u001B[39m, \u001B[33m\"\u001B[39m\u001B[33m\"\u001B[39m),\n\u001B[32m 285\u001B[39m \u001B[33m\"\u001B[39m\u001B[33mloop_settings\u001B[39m\u001B[33m\"\u001B[39m: recipe[\u001B[33m\"\u001B[39m\u001B[33mper_site\u001B[39m\u001B[33m\"\u001B[39m][\u001B[32m0\u001B[39m][\u001B[33m\"\u001B[39m\u001B[33mloop_settings\u001B[39m\u001B[33m\"\u001B[39m],\n\u001B[32m 286\u001B[39m \u001B[33m\"\u001B[39m\u001B[33msites\u001B[39m\u001B[33m\"\u001B[39m: \u001B[38;5;28mlist\u001B[39m(samples), \u001B[33m\"\u001B[39m\u001B[33mcommon\u001B[39m\u001B[33m\"\u001B[39m: common, \u001B[33m\"\u001B[39m\u001B[33mregistration\u001B[39m\u001B[33m\"\u001B[39m: reg}}\n", - "\u001B[36mFile \u001B[39m\u001B[32m~/code/work/SOF-8051/api-examples/examples/measurement/parse_utk.py:21\u001B[39m, in \u001B[36mstandata_workflow\u001B[39m\u001B[34m(application_name, workflow_name)\u001B[39m\n\u001B[32m 19\u001B[39m workflow = WorkflowStandata.find_by_application_and_name(application_name, workflow_name)\n\u001B[32m 20\u001B[39m \u001B[38;5;28;01mif\u001B[39;00m workflow \u001B[38;5;129;01mis\u001B[39;00m \u001B[38;5;28;01mNone\u001B[39;00m:\n\u001B[32m---> \u001B[39m\u001B[32m21\u001B[39m \u001B[38;5;28;01mraise\u001B[39;00m \u001B[38;5;167;01mLookupError\u001B[39;00m(\u001B[33mf\u001B[39m\u001B[33m\"\u001B[39m\u001B[33mstandata has no \u001B[39m\u001B[33m'\u001B[39m\u001B[38;5;132;01m{\u001B[39;00mworkflow_name\u001B[38;5;132;01m}\u001B[39;00m\u001B[33m'\u001B[39m\u001B[33m workflow for \u001B[39m\u001B[38;5;132;01m{\u001B[39;00mapplication_name\u001B[38;5;132;01m}\u001B[39;00m\u001B[33m\"\u001B[39m)\n\u001B[32m 22\u001B[39m \u001B[38;5;28;01mreturn\u001B[39;00m workflow\n", - "\u001B[31mLookupError\u001B[39m: standata has no 'SS-PFM Hysteresis Loop' workflow for asylum-spm" - ] - } - ], - "execution_count": 7 + "outputs": [], + "execution_count": null }, { "cell_type": "markdown", From e9475edd39c96633b2581e5e8bb88f0d6be74e96 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 22 Sep 2026 14:07:58 -0700 Subject: [PATCH 28/36] revert(SOF-8051): the notebooks do not go in the docs Nobody asked for them to be published. mkdocs.yml is back to main's content. Co-Authored-By: Claude Opus 5 (1M context) --- mkdocs.yml | 4 ---- 1 file changed, 4 deletions(-) diff --git a/mkdocs.yml b/mkdocs.yml index 38158feb8..f0f5fe884 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -58,8 +58,6 @@ nav: - Get File from Job: examples/job/get-file-from-job.ipynb - Run Simulations and Extract Properties: examples/job/run-simulations-and-extract-properties.ipynb - ML - Train Model Predict Properties: examples/job/ml-train-model-predict-properties.ipynb - - Upload an SPM Run: examples/measurement/upload_spm_run.ipynb - - Upload NLR Data: examples/measurement/upload_nlr_data.ipynb plugins: - same-dir @@ -86,7 +84,5 @@ plugins: execute_ignore: - examples/system/get_authentication_params.ipynb - examples/job/run-simulations-and-extract-properties.ipynb - - examples/measurement/upload_spm_run.ipynb - - examples/measurement/upload_nlr_data.ipynb ignore: - "other/**/*.ipynb" From 52f6be272fd0c5f2a80eda3979810aa80e567d87 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 5 Oct 2026 20:58:43 -0700 Subject: [PATCH 29/36] feat(SOF-8051): the Library - the piece every Sample Set on it belongs to Each run document now carries `library`: the physical piece as a Sample Set named by its physicalId, with the frame, the electrode layout and the synthesis record in its metadata. The uploader creates it once and nests every run's Sample Set under it (new sets with parentSetId; an existing set moved in), so one piece shows its XRF grid, its pads, and every SPM session as children of one entry. NLR's I-V run is back, built from the layout: each row becomes a pad sample taking position (and, once NLR delivers it, size and stack) from the layout entry of the same label, and a current_voltage_curve property. Until NLR delivers the electrode pattern the layout is generated at the XRF grid points and every entry says so; rows are assigned in file order and `row_index` records it. Co-Authored-By: Claude Fable 5.1 --- examples/measurement/parse_nlr.py | 78 ++++++++++++++++++++---------- examples/measurement/parse_utk.py | 18 +++---- examples/measurement/upload_run.py | 45 ++++++++++++----- 3 files changed, 95 insertions(+), 46 deletions(-) diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index 6a81f3f86..d292c2818 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -1,9 +1,10 @@ -"""NLR's delivery for one piece as platform documents: one sample set of the pads they measured, and one run per -technique over those same pads — the XRF map, then the DC I-V sweep. +"""NLR's delivery for one piece as platform documents: the Library (the piece itself, with its layout and +synthesis), the XRF map as a Sample Set of grid points, and the DC I-V sweep as a Sample Set of pads. Ad hoc parser for SOF-8050: it reads the tab-separated files NLR ships and nothing else. """ import argparse +import json from pathlib import Path from mat3ra.standata.workflows import WorkflowStandata @@ -25,8 +26,7 @@ def unit_id(workflow): return workflow["subworkflows"][0]["units"][0]["flowchartId"] NLR_FRAME = {"frame": "wafer", "units": "mm", "note": "x_mm, y_mm as delivered by NLR; corner and axes to be confirmed"} -XRF_APPLICATION = "xrf-mapper" # the standata applications whose workflows these two runs record -IV_APPLICATION = "probe-station" +XRF_APPLICATION = "xrf-mapper" # the standata application whose workflow this run records def read_columns(path): @@ -34,10 +34,25 @@ def read_columns(path): return [line.split("\t") for line in Path(path).read_text().splitlines()[1:] if line.strip()] -def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): - """NLR's delivery for one piece as platform documents: one Sample Set of the pads they measured, and one run per - technique over those same pads — the XRF map, then the DC I-V sweep. Two runs, each in the shape parse() returns, - so upload() takes them one after the other: the first creates the Sample Set, the second finds it by name.""" +IV_APPLICATION = "probe-station" + + +def provisional_layout(grid): + """The pad pattern of the piece, as far as we know it. NLR has not delivered the electrode layout, so until + they do the pads are placed at the XRF grid positions, one per grid point, and say so. A real layout from + NLR replaces this function's output and nothing downstream changes.""" + return [{"label": f"pad_r{int(row)}c{int(column)}", + "position": {"coordinates": [float(x_mm), float(y_mm)], "units": "mm"}, + "extent": None, "stack": None, + "provisional": "placed at the XRF grid point; NLR's electrode layout not yet delivered"} + for row, column, x_mm, y_mm, *_ in grid] + + +def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="", deposition=()): + """NLR's delivery for one piece as three documents sharing one Library: the XRF map (grid points) and the + DC I-V sweep (pads). The I-V export has no pad identifier - one header word and N unlabelled rows - so each + row is assigned to the layout's pads in file order; `row_index` on every pad measurement records that, and + the layout itself is provisional until NLR delivers the electrode pattern.""" folder = Path(folder) grid_file = sorted(folder.rglob("*xrf_grid.txt"))[0] volts_file, amps_file = sorted(folder.rglob("IV_Volts.txt"))[0], sorted(folder.rglob("IV_Amps.txt"))[0] @@ -50,6 +65,10 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): xrf_workflow = standata_workflow(XRF_APPLICATION, "XRF Grid Map") xrf_unit_id = unit_id(xrf_workflow) grid = read_columns(grid_file) + synthesis = [json.loads(Path(f).read_text()) for f in deposition] + library = {"physicalId": physical_id, "name": physical_id, "description": description, + "entitySetType": "unordered", + "metadata": {"frame": NLR_FRAME, "layout": provisional_layout(grid), "synthesis": synthesis}} samples, xrf_measurements, xrf_properties = {}, {}, [] for row, column, x_mm, y_mm, thickness_um, aluminium_at_pct, scandium_at_pct in grid: label = f"r{int(row)}c{int(column)}" @@ -64,52 +83,59 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument): "setup": {"name": xrf_instrument}, "status": "finished", "_records": [], "metadata": {"row": int(row), "column": int(column), "thickness_um": float(thickness_um), "al_at_pct": float(aluminium_at_pct), "sc_at_pct": float(scandium_at_pct)}} - xrf_properties += [(label, xrf_unit_id, {"name": "thickness", "value": float(thickness_um), "units": "um"}, 0), - (label, xrf_unit_id, {"name": "al_atomic_fraction", "value": float(aluminium_at_pct), "units": "at%"}, 0), - (label, xrf_unit_id, {"name": "sc_atomic_fraction", "value": float(scandium_at_pct), "units": "at%"}, 0)] + # NLR's columns become what ESSE already defines: composition is one elemental_ratio per + # element, a fraction, not a property named after the element + xrf_properties += [(label, xrf_unit_id, {"name": "film_thickness", "value": float(thickness_um) * 1e-6, "units": "m"}, 0), + (label, xrf_unit_id, {"name": "elemental_ratio", "element": "Al", "value": float(aluminium_at_pct) / 100}, 0), + (label, xrf_unit_id, {"name": "elemental_ratio", "element": "Sc", "value": float(scandium_at_pct) / 100}, 0)] iv_run_name = f"{run_name} DC IV" iv_workflow = standata_workflow(IV_APPLICATION, "DC I-V Sweep") iv_unit_id = unit_id(iv_workflow) volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] - # zip would silently drop pads, so the shapes are checked before any document is built - if not (len(grid) == len(volts) == len(amps)): - raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)}/{len(amps)} rows for " - f"{len(grid)} pads in {grid_file.name} — every pad needs one row in each file") + layout = library["metadata"]["layout"] + if not (len(layout) == len(volts) == len(amps)): + raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)}/{len(amps)} rows for {len(layout)} pads in the layout") for row, (bias_row, current_row) in enumerate(zip(volts, amps)): if len(bias_row) != len(current_row): raise SystemExit(f"row {row}: {len(bias_row)} bias points but {len(current_row)} current points") - # the sweep NLR ran, read off the voltages themselves; every row of the file holds the same one iv_setup = {"name": iv_instrument, "settings": {"v_min": min(volts[0]), "v_max": max(volts[0]), "points": len(volts[0])}} - iv_measurements, iv_properties = {}, [] - for index, (label, bias, current) in enumerate(zip(samples, volts, amps)): + pads, iv_measurements, iv_properties = {}, {}, [] + for index, (pad, bias, current) in enumerate(zip(layout, volts, amps)): + label = pad["label"] + pads[label] = {"name": f"{physical_id} {label}", "label": label, "physicalId": physical_id, + "position": pad["position"], "metadata": {"frame": NLR_FRAME, "site": "pad", "layout": label}} iv_measurements[label] = {"name": f"{iv_run_name} {label}", "_sample": None, "workflow": iv_workflow, "setup": iv_setup, "status": "finished", "_records": [], "metadata": {"row_index": index}} - iv_properties.append((label, iv_unit_id, {"name": "iv_curve", "xAxis": {"label": "bias", "units": "V"}, + iv_properties.append((label, iv_unit_id, {"name": "current_voltage_curve", "xAxis": {"label": "voltage", "units": "V"}, "yAxis": {"label": "current", "units": "A"}, "xDataArray": bias, "yDataSeries": [current]}, 0)) - return [{"physicalId": physical_id, "run": xrf_run_name, "sample_set": sample_set, "images": images, "samples": samples, - "measurement_set": {"name": xrf_run_name, "entitySetType": "ordered", "metadata": {}}, + return [{"physicalId": physical_id, "library": library, "run": xrf_run_name, "sample_set": sample_set, "images": images, + "samples": samples, "measurement_set": {"name": xrf_run_name, "entitySetType": "ordered", "metadata": {}}, "measurements": xrf_measurements, "files": {}, "set_files": [(grid_file.name, grid_file)], "records": grid, "properties": xrf_properties}, - {"physicalId": physical_id, "run": iv_run_name, "sample_set": sample_set, "images": images, "samples": samples, - "measurement_set": {"name": iv_run_name, "entitySetType": "ordered", "metadata": {}}, - "measurements": iv_measurements, "files": {}, "set_files": [(volts_file.name, volts_file), (amps_file.name, amps_file)], + {"physicalId": physical_id, "library": library, "run": iv_run_name, + "sample_set": {"name": f"{run_name} pads", "entitySetType": "ordered", "metadata": {}}, "images": [], + "samples": pads, "measurement_set": {"name": iv_run_name, "entitySetType": "ordered", "metadata": {}}, + "measurements": iv_measurements, "files": {}, + "set_files": [(volts_file.name, volts_file), (amps_file.name, amps_file)], "records": volts, "properties": iv_properties}] def main(): - """Read NLR's delivery and write a run document per technique.""" + """Read NLR's delivery and write its run document.""" ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("folder") ap.add_argument("--physical-id", required=True, help="the identifier written on the physical piece, e.g. PDAC_COM5_01448") ap.add_argument("--xrf-instrument", required=True, help="identity of the machine the grid was mapped on") ap.add_argument("--iv-instrument", required=True, help="identity of the machine the sweep was measured on") + ap.add_argument("--description", default="", help="what the piece is, for the Library") + ap.add_argument("--deposition", nargs="*", default=[], metavar="JSON", help="deposition record(s) for the Library's synthesis") ap.add_argument("--out", default="parsed", help="directory for the run documents (default: parsed/)") a = ap.parse_args() - for parsed in parse_nlr(a.folder, a.physical_id, a.xrf_instrument, a.iv_instrument): + for parsed in parse_nlr(a.folder, a.physical_id, a.xrf_instrument, a.iv_instrument, a.description, a.deposition): print(f"{parsed['physicalId']}: {len(parsed['samples'])} samples · run {parsed['run']}: " f"{len(parsed['measurements'])} measurements · {len(parsed['records'])} rows -> " f"{len(parsed['set_files'])} files · {len(parsed['images'])} image(s) · {len(parsed['properties'])} properties") diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py index 19dba26b5..00d37942e 100644 --- a/examples/measurement/parse_utk.py +++ b/examples/measurement/parse_utk.py @@ -247,15 +247,15 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= run_name = session.get("name") or run_dir.name reg = registration(recipe) sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} - # NLR's HTEM deposition record(s) for the piece, verbatim. UTK drops the file into - # the run folder as deposition*.json; --deposition overrides that. + # the Library: the piece itself. NLR's deposition record(s), when UTK dropped them into the run folder as + # deposition*.json or --deposition names them, are its synthesis. deposition_files = [Path(deposition)] if deposition else sorted(run_dir.glob("deposition*.json")) - if deposition_files: - deposition_records = [] - for f in deposition_files: - d = json.loads(f.read_text()) - deposition_records.extend(d if isinstance(d, list) else [d]) - sample_set["metadata"]["deposition"] = deposition_records + synthesis = [] + for f in deposition_files: + d = json.loads(f.read_text()) + synthesis.extend(d if isinstance(d, list) else [d]) + library = {"physicalId": physical_id, "name": physical_id, "description": "", "entitySetType": "unordered", + "metadata": {"synthesis": synthesis}} # the photograph of the piece: any image at the run-folder root images = [(f.name, f) for f in sorted(run_dir.iterdir()) if f.suffix.lower() in (".jpg", ".jpeg", ".png")] # samples in recipe order (the set is ordered; the server assigns inSet.index as they are moved in) @@ -307,7 +307,7 @@ def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument= files[label] = sample_files(label, recs, run_dir, slim_by_sample.get(label, [])) prop = combine_pad(label, recs, run_dir) if recs else None (properties.append((label, unit, prop, 0)) if prop else skipped.append(label)) - return {"physicalId": physical_id, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, + return {"physicalId": physical_id, "library": library, "run": run_name, "sample_set": sample_set, "images": images, "samples": samples, "measurement_set": measurement_set, "measurements": measurements, "files": files, "set_files": [], "records": records, "properties": properties, "skipped": skipped} diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 19b1d0b30..7be65e8e6 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -54,8 +54,10 @@ def validate(parsed): for label, uid, prop, rep in parsed["properties"]: # by the property's own name: NLR's thickness, atomic fractions and I-V curve are not the # hysteresis loop, and ESSE has no schema for them yet, so they are reported, not failed - schema_id = f"properties-directory/non-scalar/{prop['name'].replace('_', '-')}" - schema = schemas.get(schema_id) or schemas.get(schema_id.replace("non-scalar", "scalar")) + # by the property's own name, wherever the directory keeps it: scalar, non-scalar, structural + suffix = "/" + prop["name"].replace("_", "-") + schema = next((s for schema_id, s in schemas.items() + if schema_id.startswith("properties-directory/") and schema_id.endswith(suffix)), None) if schema is None: unvalidated.add(prop["name"]); continue try: @@ -140,19 +142,25 @@ def merge_metadata(existing, incoming): return merged -def ensure_set(endpoint, doc, owner_id): +def ensure_set(endpoint, doc, owner_id, parent_id=None): """The set with this name in the account, created when missing; returns (set, created). An existing set takes any metadata it does not have yet - a later upload may carry a - deposition record the set was created without.""" + deposition record the set was created without - and is moved under `parent_id` when it + is not there already.""" found = find(endpoint, {"isEntitySet": True, "name": doc["name"]}, owner_id, 5) if not found: - return endpoint.create_set(dict(doc, owner={"_id": owner_id})), True + body = dict(doc, owner={"_id": owner_id}) + if parent_id: + body["parentSetId"] = parent_id + return endpoint.create_set(body), True existing = found[0] merged = merge_metadata(existing.get("metadata") or {}, doc.get("metadata") or {}) if merged != (existing.get("metadata") or {}): endpoint.update_set(existing["_id"], {"metadata": merged}) existing = dict(existing, metadata=merged) + if parent_id and parent_id not in {s.get("_id") for s in existing.get("inSet", [])}: + endpoint.move_to_set(existing["_id"], None, parent_id) return existing, False @@ -182,16 +190,24 @@ def destination(name, set_id, measurement_ids): return f"measurements/{measurement_ids[label]}/{rest}" -def upload(client, parsed, files=("records",)): +def upload(client, parsed, files=("records",), properties=True): """The run onto its Sample Set: the set, its samples, the measurement set, one measurement per sample, the files and the properties. `files` names which groups to upload - see FILE_GROUPS; an empty list uploads none and keeps the raw records in each measurement's metadata instead. Idempotent: sets by run name, members by name/label; files re-put; properties posted only when - missing.""" + missing. `properties=False` uploads the run without them - the platform rejects a property + whose name ESSE has no schema for, and the files still carry the data to derive them from.""" uploads = run_files(parsed, files) if files else [] run_name = parsed["run"] owner = {"_id": client.my_account.id} - sample_set, created_set = ensure_set(client.samples, parsed["sample_set"], owner["_id"]) + # the Library: the physical piece, a set named by its physicalId that every Sample Set measured + # on it belongs to; its metadata carries the layout and synthesis + library_id = None + if parsed.get("library"): + library, created_library = ensure_set(client.samples, parsed["library"], owner["_id"]) + library_id = library["_id"] + print(f"library {library_id} ({library['name']}{', created' if created_library else ''})") + sample_set, created_set = ensure_set(client.samples, parsed["sample_set"], owner["_id"], library_id) set_id = sample_set["_id"] in_set = {s.get("label"): s for s in find(client.samples, {"inSet._id": set_id, "isEntitySet": {"$ne": True}}, owner["_id"], 500)} sample_ids, created = {}, 0 @@ -230,10 +246,15 @@ def upload(client, parsed, files=("records",)): if done % 200 == 0: print(f" files: {done}/{len(jobs)}", flush=True) print(f"files: {len(jobs)} put") + if not properties: + print(f"properties: skipped ({len(parsed['properties'])} not posted)") + return posted = 0 for label, unit_id, prop, repetition in parsed["properties"]: # properties/create is not idempotent: skip what is there - present = find(client.properties, {"source.info.origin._id": measurement_ids[label], "data.name": prop["name"], - "repetition": repetition}, owner["_id"], 1) + selector = {"source.info.origin._id": measurement_ids[label], "data.name": prop["name"], "repetition": repetition} + if "element" in prop: # one measurement holds an elemental_ratio per element + selector["data.element"] = prop["element"] + present = find(client.properties, selector, owner["_id"], 1) if present: continue client.properties.create(dict(holder(prop, measurement_ids[label], sample_ids[label], unit_id, repetition), owner=owner)) @@ -251,6 +272,8 @@ def main(): ap.add_argument("--files", nargs="*", default=["records"], metavar="GROUP", help=f"which file groups to upload ({', '.join(FILE_GROUPS)}); pass --files with no value " "to upload none and keep the raw records in each measurement's metadata") + ap.add_argument("--no-properties", action="store_true", + help="upload the run without its properties - the files still carry everything they were derived from") ap.add_argument("--dry-run", action="store_true", help="validate the documents and stop") a = ap.parse_args() @@ -266,7 +289,7 @@ def main(): if a.account: client = APIClient.authenticate(account_id=account_id(client, a.account), **address) for run in runs: - upload(client, run, files=a.files) + upload(client, run, files=a.files, properties=not a.no_properties) if __name__ == "__main__": From 0587f3fb3bd907265169cd25f987e885cbb5d642 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 5 Oct 2026 21:43:54 -0700 Subject: [PATCH 30/36] feat(SOF-8051): topography metrics already on the platform become properties UTK's uploader stored each AFM site's image metrics in the measurement's metadata under its own names. derive_topography.py reads them back from a Measurement Set and posts the ESSE properties they are - Sq and Sa as areal_surface_texture, the grain radius statistics as grain_size, grain_coverage - idempotently. Peak-to-valley, kurtosis and correlation length are left until UTK says how they were computed. Dry run on the demo account: 64 measurements, 448 properties. ensure_set also fills description and physicalId on a Library set created before the platform stored them. Co-Authored-By: Claude Fable 5.1 --- examples/measurement/derive_topography.py | 83 +++++++++++++++++++++++ examples/measurement/upload_run.py | 10 ++- 2 files changed, 91 insertions(+), 2 deletions(-) create mode 100644 examples/measurement/derive_topography.py diff --git a/examples/measurement/derive_topography.py b/examples/measurement/derive_topography.py new file mode 100644 index 000000000..38430dae4 --- /dev/null +++ b/examples/measurement/derive_topography.py @@ -0,0 +1,83 @@ +#!/usr/bin/env python3 +"""AFM topography metrics already stored on the platform -> the properties they are. + +UTK's uploader kept each site's image metrics in the measurement's metadata (`frames[].imageMetrics`) +under names of its own. This reads them back and posts the ESSE properties they correspond to, so the +Results tab shows them and they are comparable with any other instrument's. Only the unambiguous +metrics are mapped; peak-to-valley, kurtosis and correlation length wait for UTK to say how they were +computed. Idempotent: a property already present on a measurement is not posted again. + + derive_topography.py [--dry-run] +""" +import argparse, os, sys, urllib.parse + +from mat3ra.api_client import APIClient + +from upload_run import account_id, base_url, find, holder + +# imageMetrics key -> the ESSE property it is. Metres in, metres out. +METRICS = { + "rq_m": {"name": "areal_surface_texture", "parameter": "Sq", "units": "m"}, + "ra_m": {"name": "areal_surface_texture", "parameter": "Sa", "units": "m"}, + "grain_radius_median_m": {"name": "grain_size", "statistic": "median", "units": "m"}, + "grain_radius_mean_m": {"name": "grain_size", "statistic": "mean", "units": "m"}, + "grain_radius_std_m": {"name": "grain_size", "statistic": "std", "units": "m"}, + "grain_radius_iqr_m": {"name": "grain_size", "statistic": "iqr", "units": "m"}, + "grain_coverage": {"name": "grain_coverage"}, +} + + +def properties_of(measurement): + """The properties one topography measurement's stored metrics stand for.""" + unit_id = measurement["workflow"]["subworkflows"][0]["units"][0]["flowchartId"] + out = [] + for frame in (measurement.get("metadata") or {}).get("frames", []): + for key, shape in METRICS.items(): + if key in frame.get("imageMetrics", {}): + out.append((unit_id, dict(shape, value=frame["imageMetrics"][key]))) + return out + + +def derive(client, set_id, dry_run): + owner_id = client.my_account.id + measurements, skip = [], 0 + while True: # the list route pages at 20 whatever the limit + page = client.measurements.list({"inSet._id": set_id, "isEntitySet": {"$ne": True}, "owner._id": owner_id}, + {"limit": 20, "skip": skip}) + measurements += page + skip += 20 + if len(page) < 20: + break + posted = present = 0 + for m in measurements: + for unit_id, prop in properties_of(m): + selector = {"source.info.origin._id": m["_id"], "data.name": prop["name"]} + for key in ("parameter", "statistic"): + if key in prop: + selector[f"data.{key}"] = prop[key] + if find(client.properties, selector, owner_id, 1): + present += 1 + continue + if not dry_run: + client.properties.create(dict(holder(prop, m["_id"], m["_sample"]["_id"], unit_id, 0), owner={"_id": owner_id})) + posted += 1 + print(f"{len(measurements)} measurements: {posted} properties {'to post' if dry_run else 'posted'}, {present} already present") + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("set_id", help="the topography Measurement Set") + ap.add_argument("--host", default=os.environ.get("MAT3RA_HOST", "localhost:3000")) + ap.add_argument("--account", help="slug of the account the data belongs to") + ap.add_argument("--dry-run", action="store_true", help="count what would be posted and stop") + a = ap.parse_args() + url = urllib.parse.urlsplit(base_url(a.host)) + address = {"host": url.hostname, "port": url.port or (443 if url.scheme == "https" else 80), "secure": url.scheme == "https"} + client = APIClient.authenticate(**address) + if a.account: + client = APIClient.authenticate(account_id=account_id(client, a.account), **address) + derive(client, a.set_id, a.dry_run) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 7be65e8e6..8dc0406f9 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -155,10 +155,16 @@ def ensure_set(endpoint, doc, owner_id, parent_id=None): return endpoint.create_set(body), True existing = found[0] + changes = {} merged = merge_metadata(existing.get("metadata") or {}, doc.get("metadata") or {}) if merged != (existing.get("metadata") or {}): - endpoint.update_set(existing["_id"], {"metadata": merged}) - existing = dict(existing, metadata=merged) + changes["metadata"] = merged + for key in ("description", "physicalId"): # a Library's own fields, filled in when the set predates them + if doc.get(key) and not existing.get(key): + changes[key] = doc[key] + if changes: + endpoint.update_set(existing["_id"], changes) + existing = dict(existing, **changes) if parent_id and parent_id not in {s.get("_id") for s in existing.get("inSet", [])}: endpoint.move_to_set(existing["_id"], None, parent_id) return existing, False From 5298b5252c1db59681e4de6f61f313195eced57f Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Mon, 5 Oct 2026 21:47:51 -0700 Subject: [PATCH 31/36] fix(SOF-8051): three review findings on the Library upload - a deposition file holding a list of records is flattened into the Library's synthesis, as parse_utk already did - ensure_set adopts a same-named set only when it sits under this Library or under no Library at all; a set under another physical piece is another run and is left where it is - skipping properties no longer drops what they derive from: when the run has loop arrays and `--files` left them out, they are uploaded anyway and the uploader says so Co-Authored-By: Claude Fable 5.1 --- examples/measurement/parse_nlr.py | 5 ++++- examples/measurement/upload_run.py | 22 ++++++++++++++++++---- 2 files changed, 22 insertions(+), 5 deletions(-) diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index d292c2818..062bffa4e 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -65,7 +65,10 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="" xrf_workflow = standata_workflow(XRF_APPLICATION, "XRF Grid Map") xrf_unit_id = unit_id(xrf_workflow) grid = read_columns(grid_file) - synthesis = [json.loads(Path(f).read_text()) for f in deposition] + synthesis = [] + for f in deposition: # a file may hold one record or a list of them + record = json.loads(Path(f).read_text()) + synthesis.extend(record if isinstance(record, list) else [record]) library = {"physicalId": physical_id, "name": physical_id, "description": description, "entitySetType": "unordered", "metadata": {"frame": NLR_FRAME, "layout": provisional_layout(grid), "synthesis": synthesis}} diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 8dc0406f9..2ce3eea63 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -147,14 +147,21 @@ def ensure_set(endpoint, doc, owner_id, parent_id=None): An existing set takes any metadata it does not have yet - a later upload may carry a deposition record the set was created without - and is moved under `parent_id` when it is not there already.""" - found = find(endpoint, {"isEntitySet": True, "name": doc["name"]}, owner_id, 5) - if not found: + candidates = find(endpoint, {"isEntitySet": True, "name": doc["name"]}, owner_id, 20) + if parent_id: + # the set is this run's only if it sits under this Library, or under no Library at all (an + # upload from before Libraries existed, adopted now); one under another piece is another run + libraries = {s["_id"] for s in find(endpoint, {"isEntitySet": True, "physicalId": {"$exists": True}}, owner_id, 100)} + def parents(candidate): + return {ref.get("_id") for ref in candidate.get("inSet", [])} + candidates = [c for c in candidates if parent_id in parents(c) or not (parents(c) & libraries)] + if not candidates: body = dict(doc, owner={"_id": owner_id}) if parent_id: body["parentSetId"] = parent_id return endpoint.create_set(body), True - existing = found[0] + existing = candidates[0] changes = {} merged = merge_metadata(existing.get("metadata") or {}, doc.get("metadata") or {}) if merged != (existing.get("metadata") or {}): @@ -203,7 +210,14 @@ def upload(client, parsed, files=("records",), properties=True): Idempotent: sets by run name, members by name/label; files re-put; properties posted only when missing. `properties=False` uploads the run without them - the platform rejects a property whose name ESSE has no schema for, and the files still carry the data to derive them from.""" - uploads = run_files(parsed, files) if files else [] + groups = list(files or []) + if not properties and groups and "loops" not in groups and any( + name.startswith("loops/") for file_list in parsed.get("files", {}).values() for name, _ in file_list): + # the properties are derived from the loop arrays; with no properties posted the arrays are the + # only record of them, so they go up whatever the groups asked for + groups.append("loops") + print("properties skipped: uploading the loop arrays too, so they can be derived later") + uploads = run_files(parsed, groups) if groups else [] run_name = parsed["run"] owner = {"_id": client.my_account.id} # the Library: the physical piece, a set named by its physicalId that every Sample Set measured From c8d8f7f2b4fac3a835452f5429b5ea786cbfbadc Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 6 Oct 2026 09:33:28 -0700 Subject: [PATCH 32/36] SOF-8051: NLR pads inferred from the I-V file; wafer dimensions and frame The I-V file lists the probed Pt pads row by row, 11 to a row: each pad gets its row and column, no position until NLR delivers the pattern. The XRF grid stays the bare-film points. Library metadata carries dimensions (2-inch square) and the frame (wafer corner, mm); layout/dimensions/frame replace rather than merge. Co-Authored-By: Claude Opus 5.5 --- examples/measurement/parse_nlr.py | 45 ++++++++++++++---------------- examples/measurement/upload_run.py | 7 ++++- 2 files changed, 27 insertions(+), 25 deletions(-) diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index 062bffa4e..0d1842007 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -25,7 +25,9 @@ def unit_id(workflow): """The execution unit a property of this workflow comes from.""" return workflow["subworkflows"][0]["units"][0]["flowchartId"] -NLR_FRAME = {"frame": "wafer", "units": "mm", "note": "x_mm, y_mm as delivered by NLR; corner and axes to be confirmed"} +FRAME = {"origin": "wafer corner", "axes": "x, y", "units": "mm"} +DIMENSIONS = {"shape": "square", "side": 50.8, "units": "mm"} # a 2-inch substrate +IV_COLUMNS = 11 # the I-V file lists its pads row by row, eleven to a row XRF_APPLICATION = "xrf-mapper" # the standata application whose workflow this run records @@ -37,22 +39,12 @@ def read_columns(path): IV_APPLICATION = "probe-station" -def provisional_layout(grid): - """The pad pattern of the piece, as far as we know it. NLR has not delivered the electrode layout, so until - they do the pads are placed at the XRF grid positions, one per grid point, and say so. A real layout from - NLR replaces this function's output and nothing downstream changes.""" - return [{"label": f"pad_r{int(row)}c{int(column)}", - "position": {"coordinates": [float(x_mm), float(y_mm)], "units": "mm"}, - "extent": None, "stack": None, - "provisional": "placed at the XRF grid point; NLR's electrode layout not yet delivered"} - for row, column, x_mm, y_mm, *_ in grid] - - def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="", deposition=()): - """NLR's delivery for one piece as three documents sharing one Library: the XRF map (grid points) and the - DC I-V sweep (pads). The I-V export has no pad identifier - one header word and N unlabelled rows - so each - row is assigned to the layout's pads in file order; `row_index` on every pad measurement records that, and - the layout itself is provisional until NLR delivers the electrode pattern.""" + """NLR's delivery for one piece as two run documents sharing one Library: the XRF map, measured on the bare + film at the grid points, and the DC I-V sweep, measured on the Pt pads patterned afterwards. The pads are the + ones NLR probed: the I-V file lists them row by row, IV_COLUMNS to a row, so each pad gets its row and + column; where each pad sits on the wafer, and its size, come with NLR's pattern and are written onto the + same pads by label.""" folder = Path(folder) grid_file = sorted(folder.rglob("*xrf_grid.txt"))[0] volts_file, amps_file = sorted(folder.rglob("IV_Volts.txt"))[0], sorted(folder.rglob("IV_Amps.txt"))[0] @@ -71,7 +63,7 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="" synthesis.extend(record if isinstance(record, list) else [record]) library = {"physicalId": physical_id, "name": physical_id, "description": description, "entitySetType": "unordered", - "metadata": {"frame": NLR_FRAME, "layout": provisional_layout(grid), "synthesis": synthesis}} + "metadata": {"dimensions": DIMENSIONS, "frame": FRAME, "synthesis": synthesis}} samples, xrf_measurements, xrf_properties = {}, {}, [] for row, column, x_mm, y_mm, thickness_um, aluminium_at_pct, scandium_at_pct in grid: label = f"r{int(row)}c{int(column)}" @@ -81,7 +73,7 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="" raise SystemExit(f"{grid_file.name}: pad {label} appears twice") samples[label] = {"name": f"{physical_id} {label}", "label": label, "physicalId": physical_id, "position": {"coordinates": [float(x_mm), float(y_mm)], "units": "mm"}, - "metadata": {"frame": NLR_FRAME, "row": int(row), "column": int(column)}} + "metadata": {"row": int(row), "column": int(column)}} xrf_measurements[label] = {"name": f"{xrf_run_name} {label}", "_sample": None, "workflow": xrf_workflow, "setup": {"name": xrf_instrument}, "status": "finished", "_records": [], "metadata": {"row": int(row), "column": int(column), "thickness_um": float(thickness_um), @@ -91,23 +83,28 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="" xrf_properties += [(label, xrf_unit_id, {"name": "film_thickness", "value": float(thickness_um) * 1e-6, "units": "m"}, 0), (label, xrf_unit_id, {"name": "elemental_ratio", "element": "Al", "value": float(aluminium_at_pct) / 100}, 0), (label, xrf_unit_id, {"name": "elemental_ratio", "element": "Sc", "value": float(scandium_at_pct) / 100}, 0)] - iv_run_name = f"{run_name} DC IV" + iv_run_name = f"{physical_id} DC IV" iv_workflow = standata_workflow(IV_APPLICATION, "DC I-V Sweep") iv_unit_id = unit_id(iv_workflow) volts = [[float(v) for v in cells] for cells in read_columns(volts_file)] amps = [[float(a) for a in cells] for cells in read_columns(amps_file)] - layout = library["metadata"]["layout"] - if not (len(layout) == len(volts) == len(amps)): - raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)}/{len(amps)} rows for {len(layout)} pads in the layout") + if len(volts) != len(amps): + raise SystemExit(f"{volts_file.name}/{amps_file.name}: {len(volts)} and {len(amps)} rows") + if len(volts) % IV_COLUMNS: + raise SystemExit(f"{len(volts)} I-V rows do not fill rows of {IV_COLUMNS} pads") for row, (bias_row, current_row) in enumerate(zip(volts, amps)): if len(bias_row) != len(current_row): raise SystemExit(f"row {row}: {len(bias_row)} bias points but {len(current_row)} current points") + # the pads NLR probed, by their row and column in the file; positions and sizes come with NLR's pattern + layout = [{"label": f"pad_r{i // IV_COLUMNS}c{i % IV_COLUMNS:02d}", "row": i // IV_COLUMNS, "column": i % IV_COLUMNS, + "position": None, "extent": None, "stack": None} for i in range(len(volts))] + library["metadata"]["layout"] = layout iv_setup = {"name": iv_instrument, "settings": {"v_min": min(volts[0]), "v_max": max(volts[0]), "points": len(volts[0])}} pads, iv_measurements, iv_properties = {}, {}, [] for index, (pad, bias, current) in enumerate(zip(layout, volts, amps)): label = pad["label"] pads[label] = {"name": f"{physical_id} {label}", "label": label, "physicalId": physical_id, - "position": pad["position"], "metadata": {"frame": NLR_FRAME, "site": "pad", "layout": label}} + "metadata": {"site": "pad", "row": pad["row"], "column": pad["column"]}} iv_measurements[label] = {"name": f"{iv_run_name} {label}", "_sample": None, "workflow": iv_workflow, "setup": iv_setup, "status": "finished", "_records": [], "metadata": {"row_index": index}} iv_properties.append((label, iv_unit_id, {"name": "current_voltage_curve", "xAxis": {"label": "voltage", "units": "V"}, @@ -118,7 +115,7 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="" "measurements": xrf_measurements, "files": {}, "set_files": [(grid_file.name, grid_file)], "records": grid, "properties": xrf_properties}, {"physicalId": physical_id, "library": library, "run": iv_run_name, - "sample_set": {"name": f"{run_name} pads", "entitySetType": "ordered", "metadata": {}}, "images": [], + "sample_set": {"name": f"{physical_id} pads", "entitySetType": "ordered", "metadata": {}}, "images": [], "samples": pads, "measurement_set": {"name": iv_run_name, "entitySetType": "ordered", "metadata": {}}, "measurements": iv_measurements, "files": {}, "set_files": [(volts_file.name, volts_file), (amps_file.name, amps_file)], diff --git a/examples/measurement/upload_run.py b/examples/measurement/upload_run.py index 2ce3eea63..2e80b7743 100755 --- a/examples/measurement/upload_run.py +++ b/examples/measurement/upload_run.py @@ -126,6 +126,9 @@ def put_file(client, name, payload, owner_id): time.sleep(2 * (attempt + 1)) +REPLACED_WHOLE = ("layout", "dimensions", "frame") # a layout is one thing, not a list that grows + + def merge_metadata(existing, incoming): """`incoming` on top of `existing`, keeping what neither replaces. A list grows by the entries it does not already hold - a re-upload brings deposition records the set has never seen, @@ -133,7 +136,9 @@ def merge_metadata(existing, incoming): merged = dict(existing) for key, value in incoming.items(): held = merged.get(key) - if isinstance(held, list) and isinstance(value, list): + if key in REPLACED_WHOLE: + merged[key] = value + elif isinstance(held, list) and isinstance(value, list): merged[key] = held + [v for v in value if v not in held] elif isinstance(held, dict) and isinstance(value, dict): merged[key] = merge_metadata(held, value) From 545372e0086062d0d2a51cf6d8b03f97bff43561 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 6 Oct 2026 10:04:14 -0700 Subject: [PATCH 33/36] SOF-8051: film thickness in um, as the XRF file gives it Co-Authored-By: Claude Opus 5.5 --- examples/measurement/parse_nlr.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index 0d1842007..27a4c6e86 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -80,7 +80,7 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="" "al_at_pct": float(aluminium_at_pct), "sc_at_pct": float(scandium_at_pct)}} # NLR's columns become what ESSE already defines: composition is one elemental_ratio per # element, a fraction, not a property named after the element - xrf_properties += [(label, xrf_unit_id, {"name": "film_thickness", "value": float(thickness_um) * 1e-6, "units": "m"}, 0), + xrf_properties += [(label, xrf_unit_id, {"name": "film_thickness", "value": float(thickness_um), "units": "um"}, 0), (label, xrf_unit_id, {"name": "elemental_ratio", "element": "Al", "value": float(aluminium_at_pct) / 100}, 0), (label, xrf_unit_id, {"name": "elemental_ratio", "element": "Sc", "value": float(scandium_at_pct) / 100}, 0)] iv_run_name = f"{physical_id} DC IV" From 5feffecc4f44ea00cd8ff115685c1467b2cea85c Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 6 Oct 2026 10:26:18 -0700 Subject: [PATCH 34/36] SOF-8051: README - XRF is on the bare-film grid, I-V on the pads Co-Authored-By: Claude Opus 5.5 --- examples/measurement/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/examples/measurement/README.md b/examples/measurement/README.md index 00e9fa588..9fab87d2d 100644 --- a/examples/measurement/README.md +++ b/examples/measurement/README.md @@ -39,7 +39,7 @@ on the physical piece — every Sample carries it, and it is how the piece is fo # UTK: an Asylum SPM run folder (summary.json or recipe.json + records/ + loops/) python parse_utk.py ~/data/From_UTK --physical-id PDAC_COM5_01448 --out parsed -# NLR: an XRF grid and a DC I-V sweep over the same pads +# NLR: an XRF map of the bare film on a grid, and a DC I-V sweep over the Pt pads patterned afterwards python parse_nlr.py ~/data/From_NLR --physical-id PDAC_COM5_01448 \ --xrf-instrument bruker-m4 --iv-instrument keithley-4200 --out parsed ``` From 0cafb2b7cc30f64fcce72fd3a838efcae3aa9d9d Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 6 Oct 2026 10:32:33 -0700 Subject: [PATCH 35/36] SOF-8051: placeholder instrument names; the deliveries do not name their machines instrument-1 (UTK SPM), instrument-2 (NLR XRF), instrument-3 (NLR DC I-V) until the labs name them. Co-Authored-By: Claude Opus 5.5 --- examples/measurement/README.md | 2 +- examples/measurement/parse_utk.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/examples/measurement/README.md b/examples/measurement/README.md index 9fab87d2d..6c780c97f 100644 --- a/examples/measurement/README.md +++ b/examples/measurement/README.md @@ -41,7 +41,7 @@ python parse_utk.py ~/data/From_UTK --physical-id PDAC_COM5_01448 --out parsed # NLR: an XRF map of the bare film on a grid, and a DC I-V sweep over the Pt pads patterned afterwards python parse_nlr.py ~/data/From_NLR --physical-id PDAC_COM5_01448 \ - --xrf-instrument bruker-m4 --iv-instrument keithley-4200 --out parsed + --xrf-instrument instrument-2 --iv-instrument instrument-3 --out parsed ``` `parsed/` now holds a run document per run, plus any file a parser derived. Read it — it is the whole upload, diff --git a/examples/measurement/parse_utk.py b/examples/measurement/parse_utk.py index 00d37942e..abdef8e02 100644 --- a/examples/measurement/parse_utk.py +++ b/examples/measurement/parse_utk.py @@ -237,7 +237,7 @@ def sample_files(label, records, run_dir, slim_by_index): return out -def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument="asylum-afm"): +def parse(run_dir, physical_id, limit_records=None, deposition=None, instrument="instrument-1"): """The whole run folder as platform documents: sample set, samples, measurement set, one measurement per sample, files, one loop property per fully measured sample.""" run_dir = Path(run_dir) recipe, session, all_records = load_run(run_dir) @@ -327,7 +327,7 @@ def main(): ap.add_argument("--out", default="parsed", help="directory for the run document and the records cut from the run (default: parsed/)") ap.add_argument("--limit-records", type=int, help="trial: only the first N records and the samples they belong to") ap.add_argument("--deposition", help="NLR HTEM record (json) kept in the run's sample set metadata") - ap.add_argument("--instrument", default="asylum-afm", help="identity of the machine the run was measured on (the run folder does not record it)") + ap.add_argument("--instrument", default="instrument-1", help="identity of the machine the run was measured on (the run folder does not record it)") ap.add_argument("--emit-example", help="write the property with the most loops to this path — the ESSE example") a = ap.parse_args() From 8b5feb115db9cbfd153d2208d8d4bdc8b3b41ec0 Mon Sep 17 00:00:00 2001 From: VsevolodX Date: Tue, 6 Oct 2026 10:40:04 -0700 Subject: [PATCH 36/36] SOF-8051: NLR photographs are the Library's, not the XRF grid set's Co-Authored-By: Claude Opus 5.5 --- examples/measurement/parse_nlr.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/examples/measurement/parse_nlr.py b/examples/measurement/parse_nlr.py index 27a4c6e86..91dcaa84c 100644 --- a/examples/measurement/parse_nlr.py +++ b/examples/measurement/parse_nlr.py @@ -49,9 +49,7 @@ def parse_nlr(folder, physical_id, xrf_instrument, iv_instrument, description="" grid_file = sorted(folder.rglob("*xrf_grid.txt"))[0] volts_file, amps_file = sorted(folder.rglob("IV_Volts.txt"))[0], sorted(folder.rglob("IV_Amps.txt"))[0] run_name = grid_file.stem - # searched recursively, so the name keeps the subdirectory: two photographs may share a basename - images = [(f.relative_to(folder).as_posix(), f) for f in sorted(folder.rglob("*")) - if f.suffix.lower() in (".jpg", ".jpeg", ".png")] + images = [] # the photograph is of the piece, not of the XRF grid: it goes on the Library page sample_set = {"name": run_name, "entitySetType": "ordered", "metadata": {}} xrf_run_name = f"{run_name} XRF" xrf_workflow = standata_workflow(XRF_APPLICATION, "XRF Grid Map")