#!/usr/bin/env python3 """ make_summary.py VASP benchmark/test result summarizer. - Does not depend on tklib. - Recursively finds VASP calculation directories by INCAR. - Reads condition.md from the calculation directory or its parents. - Extracts host/CPU/parallelization settings and selected VASP results. - Writes one row per calculation to CSV. Usage: python make_summary.py DIR [DIR ...] --target /NaCl_diel/scf/ python make_summary.py "SLS_*" --target /NaCl_diel/scf/ python make_summary.py "SLS_*" --target /NaCl_diel/scf/ -o summary.csv The directory arguments may be directory paths or glob patterns. When --target is given, only calculation directories whose paths contain the target path fragment are summarized. If -o/--output is omitted, the target is converted to a filename such as vasp_summary_NaCl_diel_scf.csv. """ from __future__ import annotations import argparse import csv import glob import math import os import re import sys from pathlib import Path from typing import Any # ---------------------------------------------------------------------- # Utility # ---------------------------------------------------------------------- _FLOAT = r"[-+]?(?:\d+(?:\.\d*)?|\.\d+)(?:[EeDd][-+]?\d+)?" def as_number(value: str) -> Any: """Convert a string to int/float where possible; otherwise return string.""" s = value.strip() try: return int(s) except ValueError: pass try: return float(s.replace("D", "E").replace("d", "e")) except ValueError: return s def read_text(path: Path) -> str: return path.read_text(encoding="utf-8", errors="replace") def find_in_parents(start_dir: Path, filename: str) -> Path | None: """Search start_dir and its parents for filename.""" current = start_dir.resolve() while True: candidate = current / filename if candidate.is_file(): return candidate if current.parent == current: return None current = current.parent def expand_input_dirs(patterns: list[str]) -> list[Path]: """Expand input paths/globs while preserving order and removing duplicates.""" if not patterns: patterns = ["."] result: list[Path] = [] seen: set[Path] = set() for pattern in patterns: matches = glob.glob(pattern) if not matches and Path(pattern).is_dir(): matches = [pattern] for item in matches: path = Path(item).resolve() if path.is_dir() and path not in seen: seen.add(path) result.append(path) return result def normalize_path_fragment(value: str) -> str: """Normalize a path/path fragment for portable substring matching.""" value = value.strip().replace("\\", "/") value = re.sub(r"/+", "/", value) return value def path_contains_target(path: Path, target: str | None) -> bool: """Return True when target occurs in path using '/' as separator.""" if not target: return True path_text = normalize_path_fragment(path.as_posix()) target_text = normalize_path_fragment(target) # Add a virtual trailing slash so target="/foo/bar/" also matches a # calculation directory whose path itself ends at "/foo/bar". if not path_text.endswith("/"): path_text += "/" return target_text in path_text def default_output_name(target: str | None) -> str: """Build the default CSV filename from --target.""" if not target: return "vasp_summary.csv" name = normalize_path_fragment(target).strip("/") name = re.sub(r"[^0-9A-Za-z._-]+", "_", name) name = name.replace("/", "_") name = re.sub(r"_+", "_", name).strip("_.-") return f"vasp_summary_{name}.csv" if name else "vasp_summary.csv" # ---------------------------------------------------------------------- # condition.md # ---------------------------------------------------------------------- def normalize_condition_key(key: str) -> str: """Map condition.md labels to stable CSV field names.""" table = { "hostname": "hostname", "account": "account", "date": "date", "work_dir": "work_dir", "vasp_path": "vasp_path", "vasp_dir": "vasp_dir", "os": "os", "os name": "os_name", "os version": "os_version", "execution environment": "execution_environment", "kernel release": "kernel_release", "architecture": "architecture", "cpu model": "cpu_model", "sockets": "cpu_sockets", "physical cores": "cpu_physical_cores", "threads per core": "cpu_threads_per_core", "logical processors": "cpu_logical_processors", "mpi processes": "n_mpirun_cores", "omp_num_threads": "num_openmp", "mkl_num_threads": "num_mkl", # Backward-compatible names from older condition.md files "ncpus": "cpu_sockets", "ncores": "cpu_physical_cores", "nlogicalprocessors": "cpu_logical_processors", } k = key.strip().lower() return table.get(k, "") def read_condition_md(path: Path | None) -> dict[str, Any]: """ Read structured host/run information and linked-library information from condition.md. Known ``key: value`` metadata are imported, while the full Environment section is deliberately ignored. ``uses_oneapi`` and ``uses_mkl`` are determined only from the ``# VASP linked libraries`` (ldd) section, so an Intel oneAPI environment being loaded does not by itself count as VASP using oneAPI/MKL. """ result: dict[str, Any] = {} if path is None or not path.is_file(): return result text = read_text(path) lines = text.splitlines() section = "" linked_library_lines: list[str] = [] for raw in lines: line = raw.strip() if not line: continue if line.startswith("#"): section = line.lstrip("#").strip().lower() continue if section == "vasp linked libraries": linked_library_lines.append(raw) # Import known metadata, but never copy Environment variables. if section != "environment" and ":" in line: key, value = line.split(":", 1) outkey = normalize_condition_key(key) if outkey: result[outkey] = as_number(value) if linked_library_lines: ldd_text = "\n".join(linked_library_lines).lower() # oneAPI: linked library path/runtime explicitly comes from Intel oneAPI. result["uses_oneapi"] = "yes" if ( "/oneapi/" in ldd_text or "\\oneapi\\" in ldd_text ) else "no" # MKL: any dynamically linked MKL library is sufficient. result["uses_mkl"] = "yes" if re.search(r"\blibmkl[^\s]*", ldd_text) else "no" return result # ---------------------------------------------------------------------- # CONTCAR/POSCAR lattice # ---------------------------------------------------------------------- def dot(a: tuple[float, float, float], b: tuple[float, float, float]) -> float: return sum(x * y for x, y in zip(a, b)) def norm(a: tuple[float, float, float]) -> float: return math.sqrt(dot(a, a)) def cross(a: tuple[float, float, float], b: tuple[float, float, float]) -> tuple[float, float, float]: return ( a[1] * b[2] - a[2] * b[1], a[2] * b[0] - a[0] * b[2], a[0] * b[1] - a[1] * b[0], ) def angle_deg(a: tuple[float, float, float], b: tuple[float, float, float]) -> float: denom = norm(a) * norm(b) if denom == 0: return float("nan") x = max(-1.0, min(1.0, dot(a, b) / denom)) return math.degrees(math.acos(x)) def read_lattice(path: Path) -> dict[str, float]: """Read lattice parameters from POSCAR/CONTCAR without external packages.""" lines = read_text(path).splitlines() if len(lines) < 5: return {} try: scale = float(lines[1].split()[0]) raw = [ tuple(float(x) for x in lines[i].split()[:3]) for i in range(2, 5) ] except (ValueError, IndexError): return {} # POSCAR convention: # scale > 0 : direct scale factor. # scale < 0 : absolute target volume in A^3. if scale > 0: factor = scale elif scale < 0: target_volume = abs(scale) raw_volume = abs(dot(raw[0], cross(raw[1], raw[2]))) if raw_volume == 0: return {} factor = (target_volume / raw_volume) ** (1.0 / 3.0) else: return {} a_vec = tuple(x * factor for x in raw[0]) b_vec = tuple(x * factor for x in raw[1]) c_vec = tuple(x * factor for x in raw[2]) a = norm(a_vec) b = norm(b_vec) c = norm(c_vec) alpha = angle_deg(b_vec, c_vec) beta = angle_deg(c_vec, a_vec) gamma = angle_deg(a_vec, b_vec) volume = abs(dot(a_vec, cross(b_vec, c_vec))) return { "a_A": a, "b_A": b, "c_A": c, "alpha_deg": alpha, "beta_deg": beta, "gamma_deg": gamma, "volume_A3": volume, } # ---------------------------------------------------------------------- # OUTCAR # ---------------------------------------------------------------------- def last_float_match(text: str, pattern: str) -> float | None: matches = re.findall(pattern, text, flags=re.MULTILINE | re.IGNORECASE) if not matches: return None value = matches[-1] if isinstance(value, tuple): value = value[-1] try: return float(str(value).replace("D", "E").replace("d", "e")) except ValueError: return None def parse_outcar(path: Path) -> dict[str, Any]: """Extract benchmark-relevant scalar data from OUTCAR.""" text = read_text(path) result: dict[str, Any] = {} # VASP version, usually appears near the top as "vasp.6.x.x ..." m = re.search(r"\b(vasp\.\S+)", text, flags=re.IGNORECASE) if m: result["vasp_version"] = m.group(1) # Parallelization information printed by VASP. m = re.search(r"running on\s+(\d+)\s+total cores", text, flags=re.IGNORECASE) if m: result["outcar_total_cores"] = int(m.group(1)) m = re.search( r"distrk:\s*each k-point on\s+(\d+)\s+cores,\s*(\d+)\s+groups", text, flags=re.IGNORECASE ) if m: result["ncores_k"] = int(m.group(1)) result["kpoint_groups"] = int(m.group(2)) m = re.search( r"distr:\s*one band on\s+(\d+)\s+cores,\s*(\d+)\s+groups", text, flags=re.IGNORECASE ) if m: result["ncores_band"] = int(m.group(1)) result["band_groups"] = int(m.group(2)) # Common scalar results. patterns = { "ISPIN": rf"^\s*ISPIN\s*=\s*(\d+)", "EF_eV": rf"E-fermi\s*:\s*({_FLOAT})", "TOTEN_eV": rf"free\s+energy\s+TOTEN\s*=\s*({_FLOAT})", "total_cpu_time_s": rf"Total CPU time used \(sec\):\s*({_FLOAT})", "user_time_s": rf"User time \(sec\):\s*({_FLOAT})", "system_time_s": rf"System time \(sec\):\s*({_FLOAT})", "elapsed_time_s": rf"Elapsed time \(sec\):\s*({_FLOAT})", "maximum_memory_kb": rf"Maximum memory used \(kb\):\s*({_FLOAT})", } for key, pattern in patterns.items(): if key == "ISPIN": values = re.findall(pattern, text, flags=re.MULTILINE | re.IGNORECASE) if values: result[key] = int(values[-1]) else: value = last_float_match(text, pattern) if value is not None: result[key] = value # NCORE/KPAR are useful when present in OUTCAR. for key in ("NCORE", "KPAR"): vals = re.findall( rf"^\s*{key}\s*=\s*(\d+)", text, flags=re.MULTILINE | re.IGNORECASE ) if vals: result[key.lower()] = int(vals[-1]) return result # ---------------------------------------------------------------------- # EIGENVAL: simple band-edge extraction # ---------------------------------------------------------------------- def parse_eigenval_band_edges(path: Path, ef: float | None) -> dict[str, float]: """ Estimate EV, EC and Eg from EIGENVAL occupancies. This parser handles ordinary VASP EIGENVAL files with one or two spin channels. Metallic/partially occupied systems may not have meaningful EV/EC/Eg, so those fields are omitted when a clean separation is not found. """ lines = read_text(path).splitlines() if len(lines) < 8: return {} try: nelect, nkpts, nbands = (int(float(x)) for x in lines[5].split()[:3]) except (ValueError, IndexError): return {} occupied: list[float] = [] empty: list[float] = [] partial_found = False i = 6 for _ in range(nkpts): while i < len(lines) and not lines[i].strip(): i += 1 if i >= len(lines): break # k-point header i += 1 for _band in range(nbands): if i >= len(lines): break parts = lines[i].split() i += 1 if len(parts) < 3: continue try: nums = [float(x.replace("D", "E")) for x in parts] except ValueError: continue # Typical forms: # non-spin: band, energy, occupancy # spin: band, Eup, Edn, occ_up, occ_dn if len(nums) >= 5: channels = [(nums[1], nums[3]), (nums[2], nums[4])] else: channels = [(nums[1], nums[2])] for energy, occ in channels: # Occupancy convention is normally 2/0 (ISPIN=1) or 1/0 (ISPIN=2). if occ > 1.0e-5: occupied.append(energy) else: empty.append(energy) # Flag fractional occupations; a sharp band gap is then ambiguous. nearest = min(abs(occ - x) for x in (0.0, 1.0, 2.0)) if nearest > 1.0e-3: partial_found = True if not occupied or not empty or partial_found: return {} ev = max(occupied) ec = min(empty) if ec < ev: return {} result = { "EV_eV": ev, "EC_eV": ec, "Eg_eV": ec - ev, } if ef is not None: result["EV_minus_EF_eV"] = ev - ef result["EC_minus_EF_eV"] = ec - ef return result # ---------------------------------------------------------------------- # One calculation / one benchmark tree # ---------------------------------------------------------------------- def summarize_calculation(calc_dir: Path, base_dir: Path) -> dict[str, Any]: row: dict[str, Any] = { "source_dir": str(base_dir), "task": os.path.relpath(calc_dir, base_dir), "calc_dir": str(calc_dir), } condition_path = find_in_parents(calc_dir, "condition.md") if condition_path: row["condition_md"] = str(condition_path) row.update(read_condition_md(condition_path)) outcar = calc_dir / "OUTCAR" if outcar.is_file(): row.update(parse_outcar(outcar)) # Prefer relaxed structure, then input structure. structure = calc_dir / "CONTCAR" if not structure.is_file() or structure.stat().st_size == 0: structure = calc_dir / "POSCAR" if structure.is_file(): row.update(read_lattice(structure)) eigenval = calc_dir / "EIGENVAL" if eigenval.is_file(): ef = row.get("EF_eV") row.update(parse_eigenval_band_edges( eigenval, float(ef) if isinstance(ef, (int, float)) else None )) return row def find_calculations(base_dir: Path) -> list[Path]: """Find directories containing INCAR under base_dir.""" calc_dirs = { p.parent.resolve() for p in base_dir.rglob("INCAR") if p.is_file() } return sorted(calc_dirs, key=lambda p: str(p)) # ---------------------------------------------------------------------- # CSV # ---------------------------------------------------------------------- PREFERRED_COLUMNS = [ "source_dir", "task", "calc_dir", "condition_md", "hostname", "account", "date", "os", "os_name", "os_version", "execution_environment", "kernel_release", "architecture", "cpu_model", "cpu_sockets", "cpu_physical_cores", "cpu_threads_per_core", "cpu_logical_processors", "n_mpirun_cores", "num_openmp", "num_mkl", "vasp_path", "vasp_dir", "uses_oneapi", "uses_mkl", "vasp_version", "outcar_total_cores", "ncore", "kpar", "ncores_k", "kpoint_groups", "ncores_band", "band_groups", "ISPIN", "TOTEN_eV", "EF_eV", "EV_eV", "EC_eV", "Eg_eV", "EV_minus_EF_eV", "EC_minus_EF_eV", "a_A", "b_A", "c_A", "alpha_deg", "beta_deg", "gamma_deg", "volume_A3", "total_cpu_time_s", "user_time_s", "system_time_s", "elapsed_time_s", "maximum_memory_kb", ] def write_csv(rows: list[dict[str, Any]], output: Path) -> None: keys = set() for row in rows: keys.update(row) columns = [c for c in PREFERRED_COLUMNS if c in keys] columns += sorted(keys - set(columns)) # utf-8-sig opens cleanly in Japanese Excel. with output.open("w", newline="", encoding="utf-8-sig") as f: writer = csv.DictWriter(f, fieldnames=columns, extrasaction="ignore") writer.writeheader() writer.writerows(rows) # ---------------------------------------------------------------------- # CLI # ---------------------------------------------------------------------- def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description="Summarize VASP benchmark/test results into CSV." ) parser.add_argument( "dirs", nargs="*", help="Benchmark/result directories or glob patterns (default: current directory)", ) parser.add_argument( "-t", "--target", default=None, help=( "Only summarize calculation directories whose path contains this " "fragment, e.g. /NaCl_diel/scf/" ), ) parser.add_argument( "-o", "--output", default=None, help=( "Output CSV path. Default: vasp_summary_.csv when --target " "is given; otherwise vasp_summary.csv" ), ) return parser.parse_args() def main() -> int: args = parse_args() base_dirs = expand_input_dirs(args.dirs) if not base_dirs: print("ERROR: no input directories found.", file=sys.stderr) return 2 rows: list[dict[str, Any]] = [] for base_dir in base_dirs: all_calc_dirs = find_calculations(base_dir) calc_dirs = [ calc_dir for calc_dir in all_calc_dirs if path_contains_target(calc_dir, args.target) ] if args.target: print( f"[{base_dir}] {len(calc_dirs)} / {len(all_calc_dirs)} " f"VASP calculation(s) matched target={args.target!r}." ) else: print(f"[{base_dir}] {len(calc_dirs)} VASP calculation(s) found.") for calc_dir in calc_dirs: row = summarize_calculation(calc_dir, base_dir) rows.append(row) elapsed = row.get("elapsed_time_s", "") toten = row.get("TOTEN_eV", "") print( f" {row['task']}" f" elapsed={elapsed}" f" TOTEN={toten}" ) if not rows: if args.target: print( f"ERROR: no INCAR-containing calculation directories matched " f"target={args.target!r}.", file=sys.stderr, ) else: print("ERROR: no INCAR-containing calculation directories found.", file=sys.stderr) return 1 output = Path(args.output or default_output_name(args.target)).resolve() write_csv(rows, output) print() print(f"Wrote {len(rows)} row(s) to: {output}") return 0 if __name__ == "__main__": raise SystemExit(main())