#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ tkcif_normalize.py A small, dependency-free CIF normalizer/tester intended as the first step of "tkcif_base". It is deliberately a syntax-level normalizer, not a crystal structure interpreter. Default behavior under __main__: - recursively search *.cif below the current directory - read each CIF - parse and normalize it in memory - re-parse the normalized text as a self-check - print failed paths only, plus a summary - do not write any output files The parser is lenient for common broken CIFs: - key without inline value followed by unquoted continuation lines - semicolon text fields - quoted values - loops with quoted values - missing initial data_ header, which is repaired in memory Author: tkCIF cleanup helper """ from __future__ import annotations import argparse import re import sys from dataclasses import dataclass, field from pathlib import Path from typing import Iterable, Literal ItemKind = Literal["comment", "scalar", "loop", "unknown"] class CIFNormalizeError(Exception): """Raised when CIF normalization fails.""" @dataclass class CIFValue: raw: str value: str compact: str source: str = "normal" # normal / quoted / text_field / inferred_continuation / inferred_inline_multitoken multiline: bool = False warnings: list[str] = field(default_factory=list) @dataclass class CIFLoop: keys: list[str] rows: list[list[CIFValue]] = field(default_factory=list) warnings: list[str] = field(default_factory=list) @dataclass class CIFItem: kind: ItemKind key: str | None = None value: CIFValue | None = None loop: CIFLoop | None = None text: str | None = None @dataclass class CIFBlock: name: str items: list[CIFItem] = field(default_factory=list) warnings: list[str] = field(default_factory=list) @dataclass class CIFDocument: blocks: list[CIFBlock] = field(default_factory=list) warnings: list[str] = field(default_factory=list) def to_canonical_text(self, *, compact_multiline: bool = True) -> str: out: list[str] = [] for ib, block in enumerate(self.blocks): if ib: out.append("") name = block.name or "unknown" out.append(f"data_{safe_data_name(name)}") for item in block.items: if item.kind == "comment": text = item.text or "#" out.append(text if text.lstrip().startswith("#") else f"# {text}") elif item.kind == "scalar": if item.key is None or item.value is None: continue val = item.value.compact if compact_multiline else item.value.value out.append(f"{item.key:<35s} {quote_cif_value(val)}") elif item.kind == "loop": if item.loop is None or not item.loop.keys: continue out.append("loop_") for key in item.loop.keys: out.append(key) for row in item.loop.rows: vals = [quote_cif_value(v.compact if compact_multiline else v.value) for v in row] out.append(" " + " ".join(vals)) elif item.kind == "unknown": if item.text: out.append(f"# [ignored] {item.text}") return "\n".join(out).rstrip() + "\n" def normalize_whitespace(value: str) -> str: """Compact any whitespace sequence to a single space.""" return re.sub(r"\s+", " ", value.strip()) def safe_data_name(name: str) -> str: """Make a simple data_ block name.""" name = normalize_whitespace(name) if not name: return "unknown" # data_ names are safer without whitespace. return re.sub(r"\s+", "_", name) def is_reserved_word(token: str) -> bool: return token.lower() in {"data_", "loop_", "global_", "save_", "stop_"} def quote_cif_value(value: str | None) -> str: """ Quote a value so that the result can be written as CIF-like text. This is conservative. It does not attempt semantic repairs; it only makes scalar tokens safe for later parsers such as pymatgen. """ if value is None: return "?" value = str(value) if value in {"", ".", "?"}: return value if value else "?" if "\n" in value: return ";\n" + value.rstrip("\n") + "\n;" low = value.lower() needs_quote = ( any(ch.isspace() for ch in value) or value[0] in {"_", "#", "$", ";", "[", "]"} or low.startswith("data_") or low in {"loop_", "global_", "save_", "stop_"} ) if not needs_quote: return value if "'" not in value: return f"'{value}'" if '"' not in value: return f'"{value}"' # No escaping game here; use a text field. return ";\n" + value + "\n;" def read_text_guess(path: Path) -> tuple[str, str]: """Read a text file with a few practical encodings.""" data = path.read_bytes() for enc in ("utf-8-sig", "utf-8", "cp932", "latin-1"): try: return data.decode(enc), enc except UnicodeDecodeError: continue # latin-1 should always succeed, but keep a final fallback. return data.decode("utf-8", errors="replace"), "utf-8-replace" def normalize_newlines(text: str) -> str: text = text.replace("\r\n", "\n").replace("\r", "\n") text = text.lstrip("\ufeff") return text def split_key_value_line(line: str) -> tuple[str, str]: """ Split a CIF scalar line into key and rest. Example: _cell_length_a 4.0 -> ('_cell_length_a', '4.0') _key -> ('_key', '') """ s = line.strip() if not s.startswith("_"): raise CIFNormalizeError(f"Not a CIF key line: {line!r}") m = re.match(r"(\S+)(?:\s+(.*))?$", s) if not m: raise CIFNormalizeError(f"Cannot split key/value line: {line!r}") return m.group(1), (m.group(2) or "").strip() def line_starts_text_field(line: str) -> bool: # Strict CIF requires semicolon in column 1; accept indentation as a rescue. return line.startswith(";") or line.lstrip().startswith(";") def is_comment_line(line: str) -> bool: return line.lstrip().startswith("#") def is_blank_line(line: str) -> bool: return line.strip() == "" def is_data_line(line: str) -> bool: return line.lstrip().lower().startswith("data_") def is_loop_line(line: str) -> bool: return line.strip().lower() == "loop_" def is_key_line(line: str) -> bool: return line.lstrip().startswith("_") def is_new_item_line(line: str) -> bool: s = line.lstrip() low = s.lower() return s.startswith("_") or low.startswith("data_") or low == "loop_" def strip_inline_comment_unquoted(rest: str) -> str: """ Strip an inline comment from an unquoted scalar value. We only treat ' #' or leading '#' as comment start. This avoids damaging tokens that contain # without whitespace. """ if rest.startswith("#"): return "" m = re.search(r"\s#", rest) if m: return rest[: m.start()].rstrip() return rest.rstrip() def parse_quoted_or_token_line(line: str) -> tuple[list[CIFValue], list[str]]: """ Tokenize one CIF data line for loop rows. This is intentionally lenient. It understands single/double quoted values and treats '#' outside quotes as comment start. It does not implement the full CIF grammar; semicolon text fields are handled outside this function. """ values: list[CIFValue] = [] warnings: list[str] = [] n = len(line) i = 0 while i < n: while i < n and line[i].isspace(): i += 1 if i >= n: break if line[i] == "#": break if line[i] in {"'", '"'}: quote = line[i] start = i i += 1 chars: list[str] = [] closed = False while i < n: ch = line[i] if ch == quote: # CIF quoted string termination is normally quote followed # by whitespace/end. Accept any matching quote here. closed = True i += 1 break chars.append(ch) i += 1 raw = line[start:i] val = "".join(chars) warns: list[str] = [] if not closed: warns.append("Unclosed quoted value was accepted as a value.") warnings.extend(warns) values.append( CIFValue( raw=raw, value=val, compact=normalize_whitespace(val), source="quoted", multiline=False, warnings=warns, ) ) continue start = i while i < n and not line[i].isspace(): if line[i] == "#": break i += 1 raw = line[start:i] if raw: values.append( CIFValue( raw=raw, value=raw, compact=normalize_whitespace(raw), source="normal", multiline=False, ) ) return values, warnings def parse_scalar_inline_value(rest: str) -> CIFValue: raw = rest.strip() if raw == "": return CIFValue(raw="", value="?", compact="?", source="missing") # Quoted scalar: use tokenizer. If extra tokens exist, join them but warn. if raw[0] in {"'", '"'}: vals, warns = parse_quoted_or_token_line(raw) if not vals: return CIFValue(raw=raw, value="?", compact="?", source="quoted", warnings=warns) if len(vals) == 1: v = vals[0] v.warnings.extend(warns) return v joined = " ".join(v.value for v in vals) warns.append("Multiple tokens after a quoted scalar were joined.") return CIFValue( raw=raw, value=joined, compact=normalize_whitespace(joined), source="quoted", warnings=warns, ) # Lenient scalar: keep the full rest of the line as one value. # This rescues e.g. _space_group_name_H-M P 63 m c s = strip_inline_comment_unquoted(raw) source = "normal" if not re.search(r"\s", s) else "inferred_inline_multitoken" warns = [] if source == "inferred_inline_multitoken": warns.append("Unquoted scalar with whitespace was accepted as one value.") return CIFValue(raw=raw, value=s, compact=normalize_whitespace(s), source=source, warnings=warns) def read_text_field(lines: list[str], i: int) -> tuple[CIFValue, int, list[str]]: """Read a semicolon-delimited CIF text field starting at line i.""" warnings: list[str] = [] first = lines[i] if not first.startswith(";"): warnings.append("Indented semicolon text field was accepted.") stripped_first = first.lstrip() first_content = stripped_first[1:] if stripped_first.startswith(";") else "" content_lines: list[str] = [] if first_content.strip(): content_lines.append(first_content.rstrip("\n")) j = i + 1 while j < len(lines): line = lines[j] if line.startswith(";") or line.lstrip().startswith(";"): if not line.startswith(";"): warnings.append("Indented semicolon terminator was accepted.") raw = "\n".join(content_lines) return ( CIFValue( raw=raw, value=raw, compact=normalize_whitespace(raw), source="text_field", multiline=True, warnings=warnings.copy(), ), j + 1, warnings, ) content_lines.append(line.rstrip("\n")) j += 1 warnings.append("Unclosed semicolon text field was accepted until EOF.") raw = "\n".join(content_lines) return ( CIFValue( raw=raw, value=raw, compact=normalize_whitespace(raw), source="text_field", multiline=True, warnings=warnings.copy(), ), j, warnings, ) def read_inferred_continuation(lines: list[str], i: int) -> tuple[CIFValue, int, list[str]]: """ Read non-standard unquoted continuation lines until the next CIF item. This reproduces the useful tkCIF behavior: if a key has no inline value and the next line is not clearly another key, read lines as the value until the next key/loop/data line. """ raw_lines: list[str] = [] warnings: list[str] = [] j = i while j < len(lines): line = lines[j] if is_new_item_line(line): break # Comments and blanks are skipped in the compact value, but raw_lines # keeps ordinary non-comment continuation text. if is_comment_line(line) or is_blank_line(line): j += 1 continue raw_lines.append(line.strip()) j += 1 if not raw_lines: warnings.append("Missing scalar value was replaced by '?'.") return CIFValue(raw="", value="?", compact="?", source="missing", warnings=warnings), j, warnings raw = "\n".join(raw_lines) warnings.append("Unquoted multiline scalar was compacted until next CIF item.") return ( CIFValue( raw=raw, value=raw, compact=normalize_whitespace(raw), source="inferred_continuation", multiline=True, warnings=warnings.copy(), ), j, warnings, ) def parse_loop(lines: list[str], i: int) -> tuple[CIFItem, int, list[str]]: """Parse loop_ beginning at line i.""" warnings: list[str] = [] j = i + 1 keys: list[str] = [] while j < len(lines): line = lines[j] if is_blank_line(line) or is_comment_line(line): j += 1 continue if is_key_line(line): key, rest = split_key_value_line(line) if rest: warnings.append(f"Loop key line has extra text and it was ignored: {line.strip()}") keys.append(key) j += 1 continue break if not keys: raise CIFNormalizeError(f"loop_ at line {i + 1} has no keys") values: list[CIFValue] = [] while j < len(lines): line = lines[j] if is_blank_line(line) or is_comment_line(line): j += 1 continue if is_new_item_line(line): break if line_starts_text_field(line): v, j2, w = read_text_field(lines, j) values.append(v) warnings.extend(w) j = j2 continue vals, w = parse_quoted_or_token_line(line.strip()) values.extend(vals) warnings.extend(w) j += 1 ncol = len(keys) rows: list[list[CIFValue]] = [] if values: k = 0 while k < len(values): row = values[k : k + ncol] if len(row) < ncol: warnings.append( f"Loop with {ncol} columns ended with {len(row)} leftover values; padded by '?'." ) while len(row) < ncol: row.append(CIFValue(raw="", value="?", compact="?", source="padded")) rows.append(row) k += ncol else: warnings.append("Loop has keys but no values.") loop = CIFLoop(keys=keys, rows=rows, warnings=warnings.copy()) return CIFItem(kind="loop", loop=loop), j, warnings def parse_cif_text(text: str, *, source: str = "") -> CIFDocument: """Parse CIF text into a lenient normalized document model.""" text = normalize_newlines(text) lines = text.split("\n") # Drop final artificial empty line from split if present. if lines and lines[-1] == "": lines.pop() doc = CIFDocument() block: CIFBlock | None = None i = 0 def ensure_block() -> CIFBlock: nonlocal block if block is None: block = CIFBlock(name="unknown") block.warnings.append("Missing initial data_ header; data_unknown was inserted in memory.") doc.blocks.append(block) return block while i < len(lines): line = lines[i] s = line.strip() if is_blank_line(line): i += 1 continue if is_comment_line(line): ensure_block().items.append(CIFItem(kind="comment", text=s)) i += 1 continue if is_data_line(line): name = line.lstrip()[5:].strip() or "unknown" block = CIFBlock(name=normalize_whitespace(name)) doc.blocks.append(block) i += 1 continue if is_loop_line(line): item, i2, w = parse_loop(lines, i) b = ensure_block() b.items.append(item) b.warnings.extend(w) i = i2 continue if is_key_line(line): key, rest = split_key_value_line(line) if rest: val = parse_scalar_inline_value(rest) i += 1 else: j = i + 1 while j < len(lines) and is_blank_line(lines[j]): j += 1 if j < len(lines) and line_starts_text_field(lines[j]): val, i2, w = read_text_field(lines, j) val.warnings.extend(w) i = i2 else: val, i2, w = read_inferred_continuation(lines, j) val.warnings.extend(w) i = i2 ensure_block().items.append(CIFItem(kind="scalar", key=key, value=val)) continue # Lenient behavior: preserve unknown lines as commented-out ignored lines. ensure_block().items.append(CIFItem(kind="unknown", text=s)) ensure_block().warnings.append(f"Ignored non-CIF line at {source}:{i + 1}: {s!r}") i += 1 if not doc.blocks: raise CIFNormalizeError(f"No CIF data found in {source}") return doc def collect_warnings(doc: CIFDocument) -> list[str]: warnings: list[str] = [] warnings.extend(doc.warnings) for block in doc.blocks: for w in block.warnings: warnings.append(f"data_{block.name}: {w}") for item in block.items: if item.value: for w in item.value.warnings: warnings.append(f"data_{block.name}:{item.key}: {w}") if item.loop: for w in item.loop.warnings: warnings.append(f"data_{block.name}:loop: {w}") for ir, row in enumerate(item.loop.rows): for ic, value in enumerate(row): for w in value.warnings: key = item.loop.keys[ic] if ic < len(item.loop.keys) else f"col{ic}" warnings.append(f"data_{block.name}:loop[{ir}].{key}: {w}") return warnings def normalize_cif_file(path: Path, *, self_check: bool = True) -> tuple[str, CIFDocument, list[str], str]: text, encoding = read_text_guess(path) doc = parse_cif_text(text, source=str(path)) normalized = doc.to_canonical_text(compact_multiline=True) if self_check: # Re-parse normalized text. This does not prove chemical correctness, # but catches normalization output that our own parser cannot read. parse_cif_text(normalized, source=f"normalized:{path}") return normalized, doc, collect_warnings(doc), encoding def iter_cif_paths(root: Path, patterns: Iterable[str]) -> list[Path]: paths: list[Path] = [] for pat in patterns: paths.extend(root.rglob(pat)) return sorted(set(p for p in paths if p.is_file())) def main() -> int: parser = argparse.ArgumentParser( description="Recursively test CIF reading and in-memory normalization. No files are written by default." ) parser.add_argument("--root", type=str, default=".", help="Root directory to search. Default: current directory") parser.add_argument("--pattern", type=str, default="*.cif", help="Glob pattern. Default: *.cif") parser.add_argument("--verbose", type=int, default=0, choices=[0, 1], help="Print OK files too. Default: 0") parser.add_argument( "--show-warnings", type=int, default=0, choices=[0, 1], help="Print normalization warnings. Default: 0", ) parser.add_argument( "--self-check", type=int, default=1, choices=[0, 1], help="Re-parse normalized text in memory. Default: 1", ) args = parser.parse_args() root = Path(args.root).resolve() paths = iter_cif_paths(root, [args.pattern]) failed: list[tuple[Path, Exception]] = [] nwarn = 0 ok = 0 for path in paths: try: _normalized, doc, warnings, encoding = normalize_cif_file(path, self_check=bool(args.self_check)) ok += 1 nwarn += len(warnings) if args.verbose: print(f"OK: {path} blocks={len(doc.blocks)} encoding={encoding} warnings={len(warnings)}") if args.show_warnings and warnings: print(f"\nWarnings: {path}") for w in warnings: print(f" - {w}") except Exception as exc: failed.append((path, exc)) print("\n===== CIF normalization test summary =====") print(f"root : {root}") print(f"pattern : {args.pattern}") print(f"files : {len(paths)}") print(f"ok : {ok}") print(f"failed : {len(failed)}") print(f"warnings : {nwarn}") if failed: print("\nFailed paths:") for path, exc in failed: print(f" {path}") print(f" {type(exc).__name__}: {exc}") return 1 print("\nNo failed CIF files.") return 0 if __name__ == "__main__": raise SystemExit(main())