diff --git a/Readme.md b/Readme.md index d52a478..08266fc 100644 --- a/Readme.md +++ b/Readme.md @@ -31,6 +31,8 @@ Repositories to explore the use of Lexis in OWSEU ## Installation +Requires python 3.10 (3.11), but 3.12 makes problem with py4lexis + Usually `pip install .` on the cloned repo should do the trick. Alternative: `pip install git+https://opencode.it4i.eu/openwebsearcheu-public/owi-cli.git` diff --git a/owi/cli.py b/owi/cli.py index c241d93..a9a5f8a 100644 --- a/owi/cli.py +++ b/owi/cli.py @@ -2,7 +2,7 @@ import argparse from tabulate import tabulate from owi.core import OWSProject -import time, os +import time, os, re @@ -11,6 +11,24 @@ def _get_params(owi, args): return returns +def show_datasets(datasets, args): + # Split the patterns and compile them to regex objects for efficient matching + patterns = [re.compile(pattern) for pattern in args.fields.split(',')] + + def find_matching_columns(columns, pattern): + """Find columns that match a given regex pattern.""" + return [col for col in columns if pattern.match(col)] + + # Collect columns in the order of patterns + selected_columns = [] + for pattern in patterns: + matching_columns = find_matching_columns(datasets.columns, pattern) + selected_columns.extend(matching_columns) + + # Filter the DataFrame to only include matched columns + filtered_df = datasets[selected_columns] + print(tabulate(filtered_df, headers='keys', tablefmt=args.tablefmt)) + def _ask_yes_no(question, default='y'): """Prompt the user for a yes/no answer and return True/False. Assume 'yes' as default.""" valid_responses = { @@ -42,7 +60,7 @@ def project_stats(owi, args): def list_datasets(owi, args): datasets = owi.datasets.ls(**_get_params(owi,args)) - print(tabulate(datasets, headers='keys', tablefmt=args.tablefmt)) + show_datasets(datasets, args) def pull(owi, args): datasets = owi.datasets.ls(**_get_params(owi, args)) @@ -54,20 +72,21 @@ def pull(owi, args): datasets = datasets[(datasets["FileStatus"]=="*") | (datasets["FileStatus"]=="M")] if len(datasets)==0: print(f"Found the following {_remote_cnt} datasets, which also have local versions (content and metadata). Everything seems up-to-date") - print(tabulate(datasets_full, headers='keys', tablefmt=args.tablefmt)) + show_datasets(datasets_full, args) return accept = args.automatic_yes if not accept: print("Datasets found:") - print(tabulate(datasets, headers='keys', tablefmt=args.tablefmt)) + show_datasets(datasets, args) accept=_ask_yes_no(f"Download these datasets to {args.target} (directory will be created if not exist):") if accept: + print(f"Found {_remote_cnt} datasets, where {len(datasets)} require download") from tqdm import tqdm cols=["InternalID", "Title", "Access", "Zone","Date","DataCenter"] with tqdm(datasets[cols].values.tolist(), desc="Starting downloads") as pbar: for internal in pbar : - pbar.set_description(f"Downloading: {internal[1]}") + pbar.set_description(f"Downloading: {internal[1]} to {args.target}") results = {k: str(v) for k, v in zip(cols, internal)} try: if args.file_select is not None and args.file_select!="all" and args.file_select!="*": @@ -76,11 +95,13 @@ def pull(owi, args): print(tabulate(files, headers='keys', tablefmt=args.tablefmt)) else: owi.datasets.get(internal[0], internal[2], zone=internal[3], dest=args.target, meta =results) + if not args.no_ext: + print("Starting extraction") + owi.datasets.extract(internal[0], in_parallel=True) results["status"]="success" except Exception as e: results["status"]="fail" results["message"]=str(e) - print(f"Downloading of {internal[1]}to {args.target} stored under {internal[0]} done") owi.update_log(results) @@ -88,6 +109,7 @@ def pull(owi, args): + _DOCSTRING= """ - default path: `~/.owi` - default file name `{internalid}.tar.gz` for dataset and `{internalid}.json` for metadata @@ -110,6 +132,7 @@ Specifier Example """ def main(args=None): parser = argparse.ArgumentParser(description="CLI tool for handling OWI data."+_DOCSTRING) + parser.add_argument("--fields", type=str, default="Date,Title,ResourceType,DataCenter,Owner,PublicationYear,InternalID,Access,Zone", help='Comma separated list of fields to show') parser.add_argument('--tablefmt', type=str, default="psql", help='Output format') parser.add_argument('--target', type=str, default=os.path.expanduser(os.getenv("OWS_OWI_PATH","~/.owi")), help='Target directory') parser.add_argument('-y', '--yes', action='store_true', dest='automatic_yes', @@ -131,6 +154,9 @@ def main(args=None): parser_pull.add_argument('file_select', nargs="?", type=str, default="all", help='specifier to select files') parser_pull.add_argument('--no_sync', action='store_true', help='if true, remote datasets are not checked againts local ones') + parser_pull.add_argument('--no_ext', action='store_true', + help='if true, the dataset will not be extracted (but kept as tar.gz)') + parser_event = subparsers.add_parser('events', help='list logged events') diff --git a/owi/core/datasets.py b/owi/core/datasets.py index 532562e..281e3ef 100644 --- a/owi/core/datasets.py +++ b/owi/core/datasets.py @@ -1,4 +1,6 @@ import logging +import tarfile +import threading from typing import List from pandas import DataFrame @@ -101,7 +103,7 @@ class OWSData: def check_files(internal_id): tar_gz_path = os.path.join(local_dest, f"{internal_id}.tar.gz") json_path = os.path.join(local_dest, f"{internal_id}.json") - tar_exists = os.path.isfile(tar_gz_path) + tar_exists = os.path.isfile(tar_gz_path) or os.path.isdir(f"{internal_id}") json_exists = os.path.isfile(json_path) if not tar_exists: @@ -147,6 +149,29 @@ class OWSData: filtered_df = filtered_df[filtered_df['Date'] == latest_date] return filtered_df + def extract(self, target, in_parallel=True): + """ + extracts the tar file. + :param target: + :param in_parallel: if true, a new thread is started for extracting + :return: + """ + def _x(tar_file): + try: + print("Extracting tar file", tar_file) + with tarfile.open(tar_file, "r:gz") as tar: + tar.extractall(path=tar_file.rstrip(".tar.gz")) + except Exception as e: + print(f"An error occurred while extracting the file: {e}") + if in_parallel: + extractor_thread = threading.Thread(target=_x, args=(target,)) + extractor_thread.start() + print("Thread started for extracting", target) + else: + _x(target) + + + diff --git a/owi/core/project.py b/owi/core/project.py index 40f6649..f58ce91 100644 --- a/owi/core/project.py +++ b/owi/core/project.py @@ -1,4 +1,3 @@ -import json import logging import os from datetime import datetime @@ -9,6 +8,7 @@ import pandas as pd import re import json + def _flatten_and_count(df, column): # flatten a column [a,b,c] and count the occurences of values. if df[column].apply(lambda x: isinstance(x, list)).any(): @@ -23,12 +23,13 @@ class OWSProject: def __init__(self, logpath): self.name = os.getenv("OWI_LEXIS_PROJECT_NAME", "openwebsearch") - self.session = LexisSession(in_cli=True) - self.datasets = OWSData(self.name, self.session) self.logger = logging.getLogger(__name__) self.logpath = logpath if not os.path.exists(logpath): os.makedirs(logpath) - self.logfile = logpath+"/events.json" + self.logfile = os.path.join(logpath,"events.json") + self.session = LexisSession(in_cli=True, log_file=os.path.join(logpath, "lexis.log")) + self.datasets = OWSData(self.name, self.session) + def stats(self, day: str|datetime|None=None, duration:int = 0, public_only =True, data_center=None, query=None): df = self.datasets.ls(day, duration, public_only, data_center, query) diff --git a/requirements.txt b/requirements.txt index d6d1b7d..0fba5d3 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,6 +1,6 @@ tabulate fsspec pandas +tqdm --extra-index-url https://opencode.it4i.eu/api/v4/projects/107/packages/pypi/simple py4lexis -tqdm \ No newline at end of file