Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
345 changes: 130 additions & 215 deletions checksit/check.py

Large diffs are not rendered by default.

8 changes: 4 additions & 4 deletions checksit/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -81,8 +81,8 @@ def check_files(
ignore_all_variables: Not implemented yet.
ignore_all_variable_attrs: Not implemented yet.
auto_cache: Store the file in the template cache for future use as a template.
log_mode: How the output should be printed. Options are "standard" (default)
and "compact".
log_mode: How the output should be printed. Options are "standard" (default),
"compact" and "json".
verbose: Print additional information to the console.
template: Template to use for checking. Options are "auto" (default), "off", or
`<template file>`. File location is relative to the top level of the checksit
Expand Down Expand Up @@ -187,8 +187,8 @@ def check(
ignore_all_variables: Not implemented yet.
ignore_all_variable_attrs: Not implemented yet.
auto_cache: Store the file in the template cache for future use as a template.
log_mode: How the output should be printed. Options are "standard" (default)
and "compact".
log_mode: How the output should be printed. Options are "standard" (default),
"compact" and "json".
verbose: Print additional information to the console.
template: Template to use for checking. Options are "auto" (default), "off", or
`<template file>`. File location is relative to the top level of the checksit
Expand Down
30 changes: 27 additions & 3 deletions checksit/readers/badc_csv.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
#https://github.com/cedadev/badc-csv/blob/main/badctextfile.py
from .badctextfile import BADCTextFile

from .base import BaseReader
from typing import List, Dict
"""
req_dicts = "dimensions", "variables", "global_attributes"


Expand All @@ -21,5 +23,27 @@ def read(fpath: str, verbose: bool = False) -> BADCCSVHeader:
d = {"global_attributes": dict(bm.globalRecords)}
# "variables": bm.varRecords}
return BADCCSVHeader(fpath, d)


"""
class BADCCSVHeader(BaseReader):
def __init__(
self,
inpt: str,
verbose: bool = False,
) -> None:
"""Initialise the BADCCSVHeader.

Args:
inpt: The input file path.
verbose: Print verbose output during parsing
"""
self.inpt = inpt
self.verbose = verbose
self.fmt_errors: List[str] = []
self.global_attrs: Dict[str, str] = {}
self.dimensions: Dict[str, str] = {}
self.variables: Dict[str, Dict[str, str]] = {}

def read(self) -> None:
"""Read BADC CSV file"""
content = BADCTextFile(open(self.inpt))._metadata
self.global_attrs = dict(content.globalRecords)
28 changes: 28 additions & 0 deletions checksit/readers/base.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
from abc import ABC, abstractmethod
from typing import List, Dict, Union

class BaseReader(ABC):
inpt: str
verbose: bool
fmt_errors: List[str]
global_attrs: Dict[str, str]
dimensions: Dict[str, str]
variables: Dict[str, Dict[str, str]]

@abstractmethod
def read(self) -> None:
"""Read file"""
pass

def to_dict(self) -> Dict[str, Union[Dict[str, str], Dict[str, Dict[str, str]], str]]:
"""Convert parsed data into dict

Returns:
dictionary mess
"""
return {
"dimensions": self.dimensions,
"variables": self.variables,
"global_attributes": self.global_attrs,
"inpt": self.inpt,
}
88 changes: 46 additions & 42 deletions checksit/readers/cdl.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,9 +5,10 @@
import yaml
import subprocess as sp
import sys
from typing import Tuple, List, Dict, Union
from typing import Tuple, List, Dict

from ..cvs import vocabs, vocabs_prefix
from .base import BaseReader


def get_output(cmd: str) -> str:
Expand All @@ -23,7 +24,7 @@ def get_output(cmd: str) -> str:
return subp.stdout.read().decode("utf-8")


class CDLParser:
class CDLParser(BaseReader):
"""Parse a CDL file or netCDF file into dictionaries.

Extract information from netCDF files or CDL files into a dictionaries for
Expand All @@ -47,32 +48,35 @@ def __init__(
inpt: str,
verbose: bool = False,
) -> None:
"""Initialise the CDLParser and parse the input file.
"""Initialise the CDLParser.

Args:
inpt: The input file path or CDL content.
verbose: Print verbose output during parsing
"""
self.inpt = inpt
self.verbose = verbose
self.fmt_errors = []
self._parse(inpt)
self._check_format()

def _parse(self, inpt: str) -> None:
self.fmt_errors: List[str] = []
self.global_attrs: Dict[str, str] = {}
self.dimensions: Dict[str, str] = {}
self.variables: Dict[str, Dict[str, str]] = {}
#self.read()
#self._check_format()

def read(self) -> None:
"""Parse the input file or CDL content into dictionaries.

Args:
inpt: The input file path or CDL content.
"""
if self.verbose:
print(f"[INFO] Parsing input: {inpt[:100]}...")
if inpt.endswith(".nc"):
self.cdl = get_output(f"ncdump -h {inpt}")
elif inpt.endswith(".cdl"):
self.cdl = open(inpt).read()
print(f"[INFO] Parsing input: {self.inpt[:100]}...")
if self.inpt.endswith(".nc"):
self.cdl = get_output(f"ncdump -h {self.inpt}")
elif self.inpt.endswith(".cdl"):
self.cdl = open(self.inpt).read()
else:
self.cdl = inpt
self.cdl = self.inpt

cdl_lines: List[str] = self.cdl.strip().split("\n")

Expand All @@ -86,7 +90,7 @@ def _parse(self, inpt: str) -> None:
for s in self.CDL_SPLITTERS:
if s not in cdl_lines:
print(
f"Please check your command - invalid file or CDL contents provided: '{inpt[:100]}...'"
f"Please check your command - invalid file or CDL contents provided: '{self.inpt[:100]}...'"
)
sys.exit(1)

Expand Down Expand Up @@ -343,30 +347,30 @@ def to_yaml(self) -> str:
sort_keys=False,
)

def to_dict(self) -> Dict[str, Union[Dict[str, str], Dict[str, Dict[str, str]], str, List[str]]]:
"""Return the parsed CDL content as a dictionary.

Returns:
A dictionary of the parsed CDL content, with keys "dimensions",
"variables", "global_attributes" and "inpt", where "inpt" is the input
file path or CDL content.
"""
return {
"dimensions": self.dimensions,
"variables": self.variables,
"global_attributes": self.global_attrs,
"inpt": self.inpt,
}


def read(fpath: str, verbose: bool = False) -> CDLParser:
"""Read a CDL file or netCDF file and parse it into a CDLParser object.

Args:
fpath: The file path to read.
verbose: Print verbose output during parsing.

Returns:
A CDLParser object containing the parsed CDL content.
"""
return CDLParser(fpath, verbose=verbose)
# def to_dict(self) -> Dict[str, Union[Dict[str, str], Dict[str, Dict[str, str]], str, List[str]]]:
# """Return the parsed CDL content as a dictionary.
#
# Returns:
# A dictionary of the parsed CDL content, with keys "dimensions",
# "variables", "global_attributes" and "inpt", where "inpt" is the input
# file path or CDL content.
# """
# return {
# "dimensions": self.dimensions,
# "variables": self.variables,
# "global_attributes": self.global_attrs,
# "inpt": self.inpt,
# }


#def read(fpath: str, verbose: bool = False) -> CDLParser:
# """Read a CDL file or netCDF file and parse it into a CDLParser object.
#
# Args:
# fpath: The file path to read.
# verbose: Print verbose output during parsing.
#
# Returns:
# A CDLParser object containing the parsed CDL content.
# """
# return CDLParser(fpath, verbose=verbose)
68 changes: 34 additions & 34 deletions checksit/readers/image.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,8 @@
"""
import subprocess as sp
import yaml
from typing import Tuple, Dict, Union
from typing import Tuple, Dict
from .base import BaseReader

def get_output(cmd: str) -> Tuple[str, str]:
"""Get the output of a shell command.
Expand All @@ -17,7 +18,7 @@ def get_output(cmd: str) -> Tuple[str, str]:
return subp.stdout.read().decode("charmap"), subp.stderr.read().decode("charmap")


class ImageParser:
class ImageParser(BaseReader):
"""Parse an image file into dictionaries.

Extract information from an image file into a dictionary for tags, labelled as
Expand All @@ -38,28 +39,27 @@ def __init__(
inpt: str,
verbose: bool = False
) -> None:
"""Initialise the ImageParser and parse the input file.
"""Initialise the ImageParser.

Args:
inpt: The input file path.
verbose: Print verbose output during parsing.
"""
self.inpt = inpt
self.verbose = verbose
self.base_exiftool_arguments = ["exiftool", "-G1", "-j", "-c", "%+.6f"]
self._base_exiftool_arguments = ["exiftool", "-G1", "-j", "-c", "%+.6f"]
self.global_attrs: Dict[str, str] = {}
self.dimensions: Dict[str, str] = {}
self.variables: Dict[str, Dict[str, str]] = {}
self._find_exiftool()
self._parse(inpt)
#self.read()

def _parse(self, inpt: str) -> None:
"""Parse the input file using exiftool.

Args:
inpt: The input file path.
"""
def read(self) -> None:
"""Parse the input file using exiftool."""
if self.verbose:
print(f"[INFO] Parsing input: {inpt[:100]}...")
print(f"[INFO] Parsing input: {self.inpt[:100]}...")
self.global_attrs = {}
exiftool_arguments = self.base_exiftool_arguments + [inpt]
exiftool_arguments = self._base_exiftool_arguments + [self.inpt]
exiftool_return_string = sp.check_output(exiftool_arguments)
raw_global_attrs = yaml.load(exiftool_return_string, Loader=yaml.SafeLoader)[0]
for tag_name in raw_global_attrs.keys():
Expand Down Expand Up @@ -100,24 +100,24 @@ def _attrs_dict(self, content_lines):
attr_dict[key] = value
return attr_dict

def to_dict(self) -> Dict[str, Union[str, Dict[str, str]]]:
"""Convert the ImageParser object data to a dictionary.

Returns:
Dictionary containing metadata tags and values as "global_attributes", and
the input file path as "inpt".
"""
return {"global_attributes": self.global_attrs, "inpt": self.inpt}


def read(fpath: str, verbose: bool = False) -> ImageParser:
"""Read an image file and return an ImageParser object.

Args:
fpath: The path to the image file.
verbose: Print verbose output during parsing.

Returns:
An ImageParser object containing the metadata tags and values.
"""
return ImageParser(fpath, verbose=verbose)
# def to_dict(self) -> Dict[str, Union[str, Dict[str, str]]]:
# """Convert the ImageParser object data to a dictionary.
#
# Returns:
# Dictionary containing metadata tags and values as "global_attributes", and
# the input file path as "inpt".
# """
# return {"global_attributes": self.global_attrs, "inpt": self.inpt}


#def read(fpath: str, verbose: bool = False) -> ImageParser:
# """Read an image file and return an ImageParser object.
#
# Args:
# fpath: The path to the image file.
# verbose: Print verbose output during parsing.
#
# Returns:
# An ImageParser object containing the metadata tags and values.
# """
# return ImageParser(fpath, verbose=verbose)
41 changes: 40 additions & 1 deletion checksit/readers/pp.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
import sys
import cf
from .base import BaseReader
from typing import List, Dict

"""
req_dicts = "dimensions", "variables", "global_attributes"

class PPHeader:
Expand All @@ -26,5 +28,42 @@ def read(fpath: str, verbose: bool = False) -> PPHeader:
d["variables"][sn] = {"shape": sh}

return PPHeader(fpath, d)
"""

class PPHeader(BaseReader):
def __init__(
self,
inpt: str,
verbose: bool = False,
) -> None:
"""Initialise the PPHeader.

Args:
inpt: The input file path.
verbose: Print verbose output during parsing
"""
self.inpt = inpt
self.verbose = verbose
self.fmt_errors: List[str] = []
self.global_attrs: Dict[str, str] = {}
self.dimensions: Dict[str, str] = {}
self.variables: Dict[str, Dict[str, str]] = {}

def read(self) -> None:
"""Read YAML file"""
fieldlist = cf.read(self.inpt)
content = {"variables": {}}

for field in fieldlist:
sn = field.standard_name
sh = list(field.shape)

content["variables"][sn] = {"shape": sh}

if "global_attributes" in content:
self.global_attrs = content["global_attributes"]
if "dimensions" in content:
self.dimensions = content["dimensions"]
if "variables" in content:
self.variables = content["variables"]

Loading