From 7a71b39e274aa308bfb2baba9d6606b9f2a5bc15 Mon Sep 17 00:00:00 2001 From: 24_OskinEA <24_OskinEA@iux.local> Date: Mon, 7 Sep 2026 09:19:25 +0300 Subject: [PATCH] Implement text analyzer core and CLI --- src/text_analyzer/__init__.py | 3 + src/text_analyzer/cli.py | 159 ++++++++++++++++++++++++++++++++++ src/text_analyzer/core.py | 74 ++++++++++++++++ 3 files changed, 236 insertions(+) create mode 100644 src/text_analyzer/__init__.py create mode 100644 src/text_analyzer/cli.py create mode 100644 src/text_analyzer/core.py diff --git a/src/text_analyzer/__init__.py b/src/text_analyzer/__init__.py new file mode 100644 index 0000000..2114d76 --- /dev/null +++ b/src/text_analyzer/__init__.py @@ -0,0 +1,3 @@ +"""Text analyzer package.""" + +__version__ = "1.0.0" diff --git a/src/text_analyzer/cli.py b/src/text_analyzer/cli.py new file mode 100644 index 0000000..d28dc8d --- /dev/null +++ b/src/text_analyzer/cli.py @@ -0,0 +1,159 @@ +"""Command-line interface for the text analyzer.""" + +import argparse +import json +import sys +from pathlib import Path +from typing import Final + +from text_analyzer.core import TextStatistics, analyze_file + +DEFAULT_TOP: Final[int] = 10 +DEFAULT_MIN_LEN: Final[int] = 1 +DEFAULT_ENCODING: Final[str] = "utf-8" + + +def build_parser() -> argparse.ArgumentParser: + """Build command-line parser.""" + parser = argparse.ArgumentParser( + prog="text-analyzer", + description="Analyze a text file.", + ) + + parser.add_argument( + "path", + type=Path, + help="Path to the text file.", + ) + + parser.add_argument( + "--top", + type=int, + default=DEFAULT_TOP, + help="Number of most frequent words.", + ) + + parser.add_argument( + "--min-len", + type=int, + default=DEFAULT_MIN_LEN, + help="Minimum word length.", + ) + + parser.add_argument( + "--format", + choices=("text", "json"), + default="text", + help="Output format.", + ) + + parser.add_argument( + "--encoding", + default=DEFAULT_ENCODING, + help="Input file encoding.", + ) + + return parser + + +def format_text(statistics: TextStatistics) -> str: + """Format statistics as text.""" + lines = [ + f"Lines: {statistics.lines}", + f"Words: {statistics.words}", + f"Characters: {statistics.characters}", + "Top words:", + ] + + if statistics.top_words: + for word, count in statistics.top_words: + lines.append(f" {word}: {count}") + else: + lines.append(" No words found.") + + return "\n".join(lines) + + +def format_json(statistics: TextStatistics) -> str: + """Format statistics as JSON.""" + data = { + "lines": statistics.lines, + "words": statistics.words, + "characters": statistics.characters, + "top_words": [ + { + "word": word, + "count": count, + } + for word, count in statistics.top_words + ], + } + + return json.dumps( + data, + ensure_ascii=True, + indent=2, + ) + + +def main() -> int: + """Run CLI application.""" + parser = build_parser() + args = parser.parse_args() + + if args.top < 0: + print( + "Error: --top must be non-negative.", + file=sys.stderr, + ) + return 2 + + if args.min_len < 1: + print( + "Error: --min-len must be at least 1.", + file=sys.stderr, + ) + return 2 + + try: + statistics = analyze_file( + args.path, + top=args.top, + min_len=args.min_len, + encoding=args.encoding, + ) + except FileNotFoundError: + print( + f"Error: file not found: {args.path}", + file=sys.stderr, + ) + return 1 + except UnicodeError: + print( + f"Error: cannot decode file using encoding: {args.encoding}", + file=sys.stderr, + ) + return 1 + except OSError as exc: + print( + f"Error: cannot read file: {exc}", + file=sys.stderr, + ) + return 1 + except ValueError as exc: + print( + f"Error: {exc}", + file=sys.stderr, + ) + return 2 + + if args.format == "json": + print(format_json(statistics)) + else: + print(format_text(statistics)) + + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/text_analyzer/core.py b/src/text_analyzer/core.py new file mode 100644 index 0000000..ac0a4eb --- /dev/null +++ b/src/text_analyzer/core.py @@ -0,0 +1,74 @@ +"""Core functionality for text analysis.""" + +import re +from collections import Counter +from dataclasses import dataclass +from pathlib import Path +from typing import Final + +WORD_PATTERN: Final[re.Pattern[str]] = re.compile( + r"[^\W\d_]+(?:['-][^\W\d_]+)*", + re.UNICODE, +) + + +@dataclass(frozen=True) +class TextStatistics: + """Statistics calculated from a text.""" + + lines: int + words: int + characters: int + top_words: list[tuple[str, int]] + + +def extract_words(text: str, min_len: int = 1) -> list[str]: + """Extract words from text.""" + if min_len < 1: + raise ValueError("min_len must be at least 1") + + words = WORD_PATTERN.findall(text) + return [word.lower() for word in words if len(word) >= min_len] + + +def count_words(text: str, min_len: int = 1) -> Counter[str]: + """Count word frequencies.""" + return Counter(extract_words(text, min_len)) + + +def analyze_text( + text: str, + top: int = 10, + min_len: int = 1, +) -> TextStatistics: + """Analyze supplied text.""" + if top < 0: + raise ValueError("top must be non-negative") + + if min_len < 1: + raise ValueError("min_len must be at least 1") + + frequencies = count_words(text, min_len) + + return TextStatistics( + lines=len(text.splitlines()), + words=sum(frequencies.values()), + characters=len(text), + top_words=frequencies.most_common(top), + ) + + +def analyze_file( + path: Path, + top: int = 10, + min_len: int = 1, + encoding: str = "utf-8", +) -> TextStatistics: + """Read and analyze a text file.""" + text = path.read_text(encoding=encoding) + + return analyze_text( + text, + top=top, + min_len=min_len, + )