cp2k/tools/precommit/check_file_properties.py

Ignoring revisions in .git-blame-ignore-revs. Click here to bypass and see the normal blame view.

359 lines
12 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
# author: Ole Schuett & Tiziano Müller
import argparse
import os
import pathlib
import re
import sys
2023-11-04 13:14:11 +01:00
from datetime import datetime, timezone
from functools import lru_cache
import itertools
from typing import Tuple, List, TypeVar, Iterable
T = TypeVar("T")
# We assume this script is in tools/precommit/
CP2K_DIR = pathlib.Path(__file__).resolve().parents[2]
FLAG_EXCEPTIONS = (
r"\$\{.+\}\$",
r"__.+__",
r"_M_.+",
2024-09-06 12:37:22 +02:00
r"__ARM_ARCH",
r"__ARM_FEATURE_.+",
r"CUDA_VERSION",
r"DBM_.+",
r"OPENMP_TRACE_SYMBOL",
r"OPENCL_.+",
r"ACC_OPENCL_.+",
r"FD_DEBUG",
r"GRID_DO_COLLOCATE",
r"GRID_GPU.*_H",
r"INTEL_MKL_VERSION",
r"LIBINT2_MAX_AM_eri",
r"LIBGRPP",
r"M_",
r"LIBINT_CONTRACTED_INTS",
r"XC_MAJOR_VERSION",
r"XC_MINOR_VERSION",
2023-12-13 21:49:22 +01:00
r"NDEBUG",
r"M_PI",
2023-02-03 16:14:16 +01:00
r"OMP_DEFAULT_NONE_WITH_OOP",
r"FTN_NO_DEFAULT_INIT",
r"_OPENMP",
r"__COMPILE_ARCH",
r"__COMPILE_DATE",
r"__COMPILE_HOST",
r"__COMPILE_REVISION",
r"__CRAY_PM_FAKE_ENERGY",
r"__DATA_DIR",
r"__FORCE_USE_FAST_MATH",
r"__INTEL_LLVM_COMPILER",
r"__INTEL_COMPILER",
2025-10-04 20:14:50 +02:00
r"OFFLOAD_BUFFER_MEMPOOL",
r"OFFLOAD_MEMPOOL_.+",
r"OFFLOAD_CHECK",
r"__OFFLOAD_CUDA",
r"__OFFLOAD_HIP",
r"__PILAENV_BLOCKSIZE",
r"__PW_CUDA_NO_HOSTALLOC",
r"__T_C_G0",
r"__YUKAWA",
r"__cplusplus",
r"HIP_VERSION",
r"LIBXSMM_GEMM_PREFETCH_NONE",
r"LIBXSMM_.*VERSION_MAJOR",
r"LIBXSMM_.*VERSION_MINOR",
r"LIBXSMM_.*VERSION_UPDATE",
r"LIBXSMM_.*VERSION_PATCH",
r"LIBXSMM_VERSION_NUMBER",
2022-01-24 14:20:04 +01:00
r"LIBXSMM_VERSION2",
r"LIBXSMM_VERSION3",
r"LIBXSMM_VERSION4",
r"LIBGRPP_.+",
r"TEST_LIBGRPP_.+",
r"__LIBXSMM2",
2024-07-02 10:12:17 +02:00
r"CPVERSION",
r"_WIN32",
2026-01-06 12:57:32 +01:00
r"OPENPMDAPI_VERSION_GE",
r"openPMD_HAVE_MPI",
# TODO: Add CMake support for the following flags or remove the corresponding code.
# See also https://github.com/cp2k/cp2k/issues/4611
r"__PW_FPGA",
r"__PW_FPGA_SP",
r"__NO_SOCKETS",
r"__SCALAPACK_NO_WA",
r"__STATM_RESIDENT",
r"__STATM_TOTAL",
)
FLAG_EXCEPTIONS_RE = re.compile(r"|".join(FLAG_EXCEPTIONS))
PORTABLE_FILENAME_RE = re.compile(r"^[a-zA-Z0-9._/#~=+-]*$")
OP_RE = re.compile(r"[\\|()!&><=*/+-]")
2023-11-04 13:14:11 +01:00
NUM_RE = re.compile(r"[0-9]+[ulUL]*")
NO_DEFAULT_PATH_RE = re.compile(r"\bNO_DEFAULT_PATH\b")
CP2K_FLAGS_RE = re.compile(
r"FUNCTION cp2k_flags\(\)(.*)END FUNCTION cp2k_flags", re.DOTALL
)
CMAKE_OPTION_RE = re.compile(r"option\(\s*(\w+)", re.DOTALL)
STR_END_NOSPACE_RE = re.compile(r'[^ ]"\s*//\s*&')
STR_BEGIN_NOSPACE_RE = re.compile(r'^\s*"[^ ]')
STR_END_SPACE_RE = re.compile(r' "\s*//\s*&')
STR_BEGIN_SPACE_RE = re.compile(r'^\s*" ')
BANNER_F = """\
!--------------------------------------------------------------------------------------------------!
! CP2K: A general program to perform molecular dynamics simulations !
! Copyright 2000-{:d} CP2K developers group <https://cp2k.org> !
! !
2021-08-30 21:03:26 +02:00
! SPDX-License-Identifier: {:s} !
!--------------------------------------------------------------------------------------------------!
"""
BANNER_SHELL = """\
#!-------------------------------------------------------------------------------------------------!
#! CP2K: A general program to perform molecular dynamics simulations !
#! Copyright 2000-{:d} CP2K developers group <https://cp2k.org> !
#! !
2021-08-30 21:03:26 +02:00
#! SPDX-License-Identifier: {:s} !
#!-------------------------------------------------------------------------------------------------!
"""
BANNER_C = """\
/*----------------------------------------------------------------------------*/
/* CP2K: A general program to perform molecular dynamics simulations */
/* Copyright 2000-{:d} CP2K developers group <https://cp2k.org> */
/* */
2021-08-30 21:03:26 +02:00
/* SPDX-License-Identifier: {:s} */
/*----------------------------------------------------------------------------*/
"""
2021-08-30 21:03:26 +02:00
C_EXTENSIONS = (".c", ".cu", ".cpp", ".cc", ".h", ".hpp")
# Non-GPL licenses (directory, file, basename, or generally "startswith")
BSD_PATHS = (
"src/base/openmp_trace.c",
"src/mpiwrap/cp_mpi.",
"src/offload/",
"src/grid/",
"src/dbm/",
)
MIT_PATHS = ("src/grpp/",)
@lru_cache(maxsize=None)
def get_src_cmakelists_txt() -> str:
return "\n".join(
(CP2K_DIR / fn).read_text(encoding="utf8")
for fn in ["src/CMakeLists.txt", "cmake/CompilerConfiguration.cmake"]
)
@lru_cache(maxsize=None)
def get_build_docs() -> str:
files = list((CP2K_DIR / "docs/technologies").glob("**/*.md"))
files.append(CP2K_DIR / "docs/getting-started/build-from-source.md")
return "\n".join(fn.read_text(encoding="utf8") for fn in files)
@lru_cache(maxsize=None)
def get_flags_src() -> str:
cp2k_info = (CP2K_DIR / "src/cp2k_info.F").read_text(encoding="utf8")
match = CP2K_FLAGS_RE.search(cp2k_info)
assert match
return match.group(1)
@lru_cache(maxsize=None)
def get_bibliography_dois() -> List[str]:
bib = (CP2K_DIR / "src/common/bibliography.F").read_text(encoding="utf8")
2024-12-15 13:35:40 +01:00
matches = re.findall(r'doi="([^"]+)"', bib, flags=re.IGNORECASE)
assert len(matches) > 260 and "10.1016/j.cpc.2004.12.014" in matches
return matches
def check_file(path: pathlib.Path) -> List[str]:
"""
Check the given source file for convention violations, like:
- correct copyright headers
- undocumented preprocessor flags
- stray unicode characters
"""
warnings: List[str] = []
fn_ext = path.suffix
abspath = path.resolve()
basefn = path.name
is_executable = os.access(abspath, os.X_OK)
if not PORTABLE_FILENAME_RE.match(str(path)):
warnings += [f"Filename '{path}' not portable"]
if not abspath.exists():
return warnings # skip broken symlinks
raw_content = abspath.read_bytes()
if b"\0" in raw_content:
return warnings # skip binary files
content = raw_content.decode("utf8")
if "\r\n" in content:
warnings += [f"{path}: contains DOS linebreaks"]
if fn_ext not in (".pot", ".patch") and basefn != "Makefile" and "\t" in content:
warnings += [f"{path}: contains tab character"]
if fn_ext == ".cu" and "#if defined(_OMP_H)\n#error" not in content:
warnings += [f"{path}: misses check against OpenMP usage"]
# Check spaces in Fortran multi-line strings.
if fn_ext == ".F":
for i, (a, b) in enumerate(pairwise(content.split("\n"))):
if STR_END_NOSPACE_RE.search(a) and STR_BEGIN_NOSPACE_RE.search(b):
warnings += [f"{path}:{i+1} Missing space in multi-line string"]
if STR_END_SPACE_RE.search(a) and STR_BEGIN_SPACE_RE.search(b):
warnings += [f"{path}:{i+1} Double space in multi-line string"]
# Check CPASSERT(.FALSE.) and empty CPABORT() messages
suppress_cpabort = ["semi_empirical_int_debug.F"]
if fn_ext == ".F" and basefn not in suppress_cpabort:
if "CPASSERT(.FALSE.)" in content:
warnings += [
f"{path}: Found CPASSERT(.FALSE.) - please use CPABORT() with messages"
]
if "CPABORT('')" in content or 'CPABORT("")' in content:
warnings += [
f"{path}: Found CPABORT() with empty message - please fill in the reason"
]
# check banner
2023-11-04 13:14:11 +01:00
year = datetime.now(timezone.utc).year
bsd_licensed = any(str(path).startswith(p) for p in BSD_PATHS)
mit_licensed = any(str(path).startswith(p) for p in MIT_PATHS)
if bsd_licensed:
spdx = "BSD-3-Clause "
elif mit_licensed:
spdx = "MIT "
else:
spdx = "GPL-2.0-or-later"
2021-08-30 21:03:26 +02:00
if fn_ext == ".F" and not content.startswith(BANNER_F.format(year, spdx)):
warnings += [f"{path}: Copyright banner malformed"]
if fn_ext == ".fypp" and not content.startswith(BANNER_SHELL.format(year, spdx)):
warnings += [f"{path}: Copyright banner malformed"]
if fn_ext == ".cmake" or path.name == "CMakeLists.txt":
if not content.startswith(BANNER_SHELL.format(year, spdx)):
warnings += [f"{path}: Copyright banner malformed"]
2021-08-30 21:03:26 +02:00
if fn_ext in C_EXTENSIONS and not content.startswith(BANNER_C.format(year, spdx)):
warnings += [f"{path}: Copyright banner malformed"]
if path.name == "LICENSE" and bsd_licensed and f"2000-{year}" not in content:
warnings += [f"{path}: Copyright banner malformed"]
2023-01-01 13:10:18 +01:00
if path.name == "cp2k_info.F" and f'cp2k_year = "{year}"' not in content:
warnings += [f"{path}: Wrong year."]
# check shebang
PY_SHEBANG = "#!/usr/bin/env python3"
if fn_ext == ".py" and is_executable and not content.startswith(f"{PY_SHEBANG}\n"):
warnings += [f"{path}: Wrong shebang, please use '{PY_SHEBANG}'"]
# find all flags
flags = set()
line_continuation = False
for line in content.splitlines():
line = line.lstrip()
if not line_continuation:
if not line or line[0] != "#":
continue
if line.split()[0] not in ("#if", "#ifdef", "#ifndef", "#elif"):
continue
DBM: OpenCL implementation (#3375) * There are three main implementations 1. DBM_MULTIPLY_SPLIT=1: default kernel with smallest LOC-number uses private/register-backed buffer. This implementation is simple and performs best even on different vendor's GPUs. There is moderate unrolling/code-bloat leveraging macros. Though, there is no "old school GPU" like copy-into SLM; if there is re-use, it leverage existing caches between registers and global memory. 2. DBM_MULTIPLY_SPLIT=2..N with DBM_MULTIPLY_SPLIT=8 matching the CUDA implementation using shared/local memory. This implementation matches the existing CUDA implementation and perform equal and reasonable across GPUs. There is moderate unrolling/code-bloat leveraging macros; heuristics are revised/different from CUDA implementation (less cases). 3. DBM_MULTIPLY_GEN=1: uses https://github.com/intel/tiny-tensor-compiler and IR describing DBM's input format. The IR-file (https://github.com/hfp/cp2k/blob/master/src/dbm/dbm_multiply_opencl.ir) is not part of PR. The generated kernel-code is not yet competitive. * The amount of code in the OpenCL-kernel and and C based glue-code can be *significantly* reduced (fraction of LOC) when removing for instance implementation 2 and 3 (above mentioned). Reducing the number of LOC can be part of a future cleanup (and once code is settled). * check_file_properties.py: support C-comments, and omit checking OpenCL source code (it is usually compiled at runtime, i.e., machine-driven rather than user-controlled).
2024-04-29 12:23:24 +02:00
line = line.split("/*", 1)[0] # C comment
line = line.split("//", 1)[0] # C++ comment
line_continuation = line.rstrip().endswith("\\")
line = OP_RE.sub(" ", line)
line = line.replace("defined", " ")
for word in line.split()[1:]:
if NUM_RE.match(word):
continue # skip numbers
if fn_ext in (".h", ".hpp") and word == basefn.upper().replace(".", "_"):
continue # ignore aptly named inclusion guards
flags.add(word)
flags = {flag for flag in flags if not FLAG_EXCEPTIONS_RE.match(flag)}
for flag in sorted(flags):
DBM: OpenCL implementation (#3375) * There are three main implementations 1. DBM_MULTIPLY_SPLIT=1: default kernel with smallest LOC-number uses private/register-backed buffer. This implementation is simple and performs best even on different vendor's GPUs. There is moderate unrolling/code-bloat leveraging macros. Though, there is no "old school GPU" like copy-into SLM; if there is re-use, it leverage existing caches between registers and global memory. 2. DBM_MULTIPLY_SPLIT=2..N with DBM_MULTIPLY_SPLIT=8 matching the CUDA implementation using shared/local memory. This implementation matches the existing CUDA implementation and perform equal and reasonable across GPUs. There is moderate unrolling/code-bloat leveraging macros; heuristics are revised/different from CUDA implementation (less cases). 3. DBM_MULTIPLY_GEN=1: uses https://github.com/intel/tiny-tensor-compiler and IR describing DBM's input format. The IR-file (https://github.com/hfp/cp2k/blob/master/src/dbm/dbm_multiply_opencl.ir) is not part of PR. The generated kernel-code is not yet competitive. * The amount of code in the OpenCL-kernel and and C based glue-code can be *significantly* reduced (fraction of LOC) when removing for instance implementation 2 and 3 (above mentioned). Reducing the number of LOC can be part of a future cleanup (and once code is settled). * check_file_properties.py: support C-comments, and omit checking OpenCL source code (it is usually compiled at runtime, i.e., machine-driven rather than user-controlled).
2024-04-29 12:23:24 +02:00
if fn_ext == ".cl": # usually compiled at RT (no direct user-control)
continue
if flag == "_OMP_H" and fn_ext == ".cu":
continue
if flag not in get_src_cmakelists_txt():
warnings += [
f"{path}: Flag '{flag}' not mentioned in src/CMakeLists.txt nor cmake/CompilerConfiguration.cmake"
]
if flag not in get_flags_src():
warnings += [f"{path}: Flag '{flag}' not mentioned in cp2k_flags()"]
if "cmake" in str(path).lower():
options = CMAKE_OPTION_RE.findall(content)
for opt in options:
if opt not in get_build_docs():
warnings += [
f"{path}: CMake option {opt} not mentioned in docs/technologies section nor build-from-source.md"
]
# NO_DEFAULT_PATH disables searching CMAKE_PREFIX_PATH and other default
# locations, which breaks Spack, system-package, and HPC-module installs
# unless every possible layout is hand-enumerated in PATHS/HINTS - don't use it.
if (
fn_ext == ".cmake" or path.name == "CMakeLists.txt"
) and NO_DEFAULT_PATH_RE.search(content):
warnings += [f"{path}: Found NO_DEFAULT_PATH, please remove it"]
# Check for DOIs that could be a bibliography reference.
if re.match(r"docs/[^/]+/.*\.md", str(path)) and "docs/CP2K_INPUT" not in str(path):
for line in content.splitlines():
for doi in get_bibliography_dois():
if doi.lower() in line:
warnings += [f"{path}: Please replace doi:{doi} with biblio ref."]
return warnings
# ======================================================================================
def pairwise(iterable: Iterable[T]) -> Iterable[Tuple[T, T]]:
"""itertools.pairwise is not available before Python 3.10."""
# pairwise('ABCDEFG') --> AB BC CD DE EF FG
a, b = itertools.tee(iterable)
next(b, None)
return zip(a, b)
# ======================================================================================
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Check the given FILENAME for conventions"
)
parser.add_argument("files", metavar="FILENAME", type=pathlib.Path, nargs="+")
args = parser.parse_args()
all_warnings = []
for fpath in args.files:
all_warnings += check_file(fpath)
for warning in all_warnings:
print(warning)
if all_warnings:
sys.exit(1)