diff options
| author | Dennis Fink | 2020-12-11 13:58:14 +0100 |
|---|---|---|
| committer | Dennis Fink | 2020-12-11 13:58:14 +0100 |
| commit | 544fa972523de559446db8c5daf54d71d5a8d603 (patch) | |
| tree | c2ba3d244c95886329a3b3f5d9775af2626e6f9a | |
| parent | 93d9203a81075d9dc35d5377e881993d69afc282 (diff) | |
| download | patternutils-544fa972523de559446db8c5daf54d71d5a8d603.tar.gz patternutils-544fa972523de559446db8c5daf54d71d5a8d603.zip | |
Implement real streaming in prss and pmatch
Diffstat (limited to '')
| -rw-r--r-- | patternutils/commands/pmatch.py | 38 | ||||
| -rw-r--r-- | patternutils/commands/prss.py | 104 |
2 files changed, 79 insertions, 63 deletions
diff --git a/patternutils/commands/pmatch.py b/patternutils/commands/pmatch.py index 18df269..00c16e9 100644 --- a/patternutils/commands/pmatch.py +++ b/patternutils/commands/pmatch.py @@ -2,7 +2,7 @@ import functools import itertools import os.path import re -from typing import Any, Dict, Iterator, List, Pattern, Tuple +from typing import Any, Dict, Generator, Iterator, List, Optional, Pattern import click @@ -12,27 +12,25 @@ from .. import utils def apply_regex( - walk_function: Iterator[Any], regex_pattern: Pattern[str], match_full_path: bool, -) -> Tuple[List[Dict[str, str]], bool]: + walk_function: Iterator[Any], + regex_pattern: Pattern[str], + match_full_path: bool, +) -> Generator[Optional[Dict[str, str]], None, None]: - matches = [] - has_invalid_matches = False for f in walk_function: subject = f.path if match_full_path else f.name try: match = mat.apply(subject, regex_pattern) except ValueError: click.secho(f"{subject} did not match", fg="red", err=True) - has_invalid_matches = True + yield None else: match["_path"] = f.path match["_name"] = f.name match["_abspath"] = os.path.abspath(f.path) match["_realpath"] = os.path.realpath(f.path) match["_relpath"] = os.path.relpath(f.path) - matches.append(match) - - return matches, has_invalid_matches + yield match @click.command(context_settings={"help_option_names": ("-h", "--help", "-?")}) @@ -124,18 +122,22 @@ def pmatch( ) new_walk = itertools.chain.from_iterable(map(walk_function, directory)) - matches, has_invalid_matches = apply_regex( - new_walk, regex_pattern_c, match_full_path - ) + has_invalid_matches = False + matches = [] + for match in apply_regex(new_walk, regex_pattern_c, match_full_path): + if match is None: + has_invalid_matches = True + if stream and abort_on_no_match: + raise SystemExit + else: + if stream: + click.echo(utils.json_dumps(match, human_readable)) + else: + matches.append(match) if has_invalid_matches and abort_on_no_match: raise SystemExit - - if stream: - for m in matches: - click.echo(utils.json_dumps(m, human_readable)) - else: - click.echo(utils.json_dumps(matches, human_readable)) + click.echo(utils.json_dumps(matches, human_readable)) if __name__ == "__main__": diff --git a/patternutils/commands/prss.py b/patternutils/commands/prss.py index bdf35a4..cd3f820 100644 --- a/patternutils/commands/prss.py +++ b/patternutils/commands/prss.py @@ -1,6 +1,8 @@ import functools import itertools +import os import os.path +from typing import Any, Dict, Generator, Iterator, List, Optional import click @@ -13,6 +15,53 @@ except ImportError: from .. import utils +def match( + parsed_feed: feedparser.util.FeedParserDict, + walk_function: functools.partial[Generator[os.DirEntry[Any], None, None]], + directory: List[str], +) -> Generator[Optional[Dict[str, str]], None, None]: + + direntries = [] + filenames = [] + + for f in itertools.chain.from_iterable(map(walk_function, directory)): + direntries.append(f) + filenames.append(f.name) + + for entry in parsed_feed.entries: + href = entry.enclosures[0]["href"] + splitted_href = href.split("/") + file_to_search = splitted_href[-1] + try: + file_index = filenames.index(file_to_search) + except ValueError: + click.secho(f"{file_to_search} not found!", fg="red", err=True) + yield None + else: + filename = filenames[file_index] + direntry = direntries[file_index] + entry_values = { + "title": entry.get("title", None), + "link": entry.get("link", None), + "description": entry.get("description", None), + "id": entry.get("id", None), + "enclosure_href": href, + "published": entry.get("published", None), + "subtitle": entry.get("subtitle", None), + "summary": entry.get("summary", None), + "author": entry.get("author", None), + "_name": filename, + "_path": direntry.path, + "_abspath": os.path.abspath(direntry.path), + "_realpath": os.path.realpath(direntry.path), + "_relpath": os.path.relpath(direntry.path), + "feed_title": parsed_feed.feed.get("title", None), + "feed_link": parsed_feed.feed.get("link", None), + "feed_description": parsed_feed.feed.get("description", None), + } + yield entry_values + + @click.command(context_settings={"help_option_names": ("-h", "--help", "-?")}) @click.option("-d", "--directory", multiple=True, default=["./"]) @click.option( @@ -47,7 +96,7 @@ from .. import utils @click.argument("url") def prss( url: str, - directory: str, + directory: List[str], recursive: bool, abort_on_no_file: bool, stream: bool, @@ -70,58 +119,23 @@ def prss( utils.walk, recursive=recursive, matchdirectories=False ) - direntries = [] - filenames = [] - - for f in itertools.chain.from_iterable(map(walk_function, directory)): - direntries.append(f) - filenames.append(f.name) - matches = [] - has_no_file = False - for entry in parsed_feed.entries: - - href = entry.enclosures[0]["href"] - splitted_href = href.split("/") - file_to_search = splitted_href[-1] - try: - file_index = filenames.index(file_to_search) - except ValueError: - click.secho(f"{file_to_search} not found!", fg="red", err=True) + for m in match(parsed_feed, walk_function, directory): + if m is None: has_no_file = True + if abort_on_no_file and stream: + raise SystemExit else: - filename = filenames[file_index] - direntry = direntries[file_index] - entry_values = { - "title": entry.get("title", None), - "link": entry.get("link", None), - "description": entry.get("description", None), - "id": entry.get("id", None), - "enclosure_href": href, - "published": entry.get("published", None), - "subtitle": entry.get("subtitle", None), - "summary": entry.get("summary", None), - "author": entry.get("author", None), - "_name": filename, - "_path": direntry.path, - "_abspath": os.path.abspath(direntry.path), - "_realpath": os.path.realpath(direntry.path), - "_relpath": os.path.relpath(direntry.path), - "feed_title": parsed_feed.feed.get("title", None), - "feed_link": parsed_feed.feed.get("link", None), - "feed_description": parsed_feed.feed.get("description", None), - } - matches.append(entry_values) + if stream: + click.echo(utils.json_dumps(m, human_readable)) + else: + matches.append(m) if abort_on_no_file and has_no_file: raise SystemExit - if stream: - for m in matches: - click.echo(utils.json_dumps(m, human_readable)) - else: - click.echo(utils.json_dumps(matches, human_readable)) + click.echo(utils.json_dumps(matches, human_readable)) if __name__ == "__main__": |
