Implement text analyzer core and CLI

This commit is contained in:
2026-09-07 09:19:25 +03:00
parent 39cf23be71
commit 7a71b39e27
3 changed files with 236 additions and 0 deletions

View File

@@ -0,0 +1,3 @@
"""Text analyzer package."""
__version__ = "1.0.0"

159
src/text_analyzer/cli.py Normal file
View File

@@ -0,0 +1,159 @@
"""Command-line interface for the text analyzer."""
import argparse
import json
import sys
from pathlib import Path
from typing import Final
from text_analyzer.core import TextStatistics, analyze_file
DEFAULT_TOP: Final[int] = 10
DEFAULT_MIN_LEN: Final[int] = 1
DEFAULT_ENCODING: Final[str] = "utf-8"
def build_parser() -> argparse.ArgumentParser:
"""Build command-line parser."""
parser = argparse.ArgumentParser(
prog="text-analyzer",
description="Analyze a text file.",
)
parser.add_argument(
"path",
type=Path,
help="Path to the text file.",
)
parser.add_argument(
"--top",
type=int,
default=DEFAULT_TOP,
help="Number of most frequent words.",
)
parser.add_argument(
"--min-len",
type=int,
default=DEFAULT_MIN_LEN,
help="Minimum word length.",
)
parser.add_argument(
"--format",
choices=("text", "json"),
default="text",
help="Output format.",
)
parser.add_argument(
"--encoding",
default=DEFAULT_ENCODING,
help="Input file encoding.",
)
return parser
def format_text(statistics: TextStatistics) -> str:
"""Format statistics as text."""
lines = [
f"Lines: {statistics.lines}",
f"Words: {statistics.words}",
f"Characters: {statistics.characters}",
"Top words:",
]
if statistics.top_words:
for word, count in statistics.top_words:
lines.append(f" {word}: {count}")
else:
lines.append(" No words found.")
return "\n".join(lines)
def format_json(statistics: TextStatistics) -> str:
"""Format statistics as JSON."""
data = {
"lines": statistics.lines,
"words": statistics.words,
"characters": statistics.characters,
"top_words": [
{
"word": word,
"count": count,
}
for word, count in statistics.top_words
],
}
return json.dumps(
data,
ensure_ascii=True,
indent=2,
)
def main() -> int:
"""Run CLI application."""
parser = build_parser()
args = parser.parse_args()
if args.top < 0:
print(
"Error: --top must be non-negative.",
file=sys.stderr,
)
return 2
if args.min_len < 1:
print(
"Error: --min-len must be at least 1.",
file=sys.stderr,
)
return 2
try:
statistics = analyze_file(
args.path,
top=args.top,
min_len=args.min_len,
encoding=args.encoding,
)
except FileNotFoundError:
print(
f"Error: file not found: {args.path}",
file=sys.stderr,
)
return 1
except UnicodeError:
print(
f"Error: cannot decode file using encoding: {args.encoding}",
file=sys.stderr,
)
return 1
except OSError as exc:
print(
f"Error: cannot read file: {exc}",
file=sys.stderr,
)
return 1
except ValueError as exc:
print(
f"Error: {exc}",
file=sys.stderr,
)
return 2
if args.format == "json":
print(format_json(statistics))
else:
print(format_text(statistics))
return 0
if __name__ == "__main__":
raise SystemExit(main())

74
src/text_analyzer/core.py Normal file
View File

@@ -0,0 +1,74 @@
"""Core functionality for text analysis."""
import re
from collections import Counter
from dataclasses import dataclass
from pathlib import Path
from typing import Final
WORD_PATTERN: Final[re.Pattern[str]] = re.compile(
r"[^\W\d_]+(?:['-][^\W\d_]+)*",
re.UNICODE,
)
@dataclass(frozen=True)
class TextStatistics:
"""Statistics calculated from a text."""
lines: int
words: int
characters: int
top_words: list[tuple[str, int]]
def extract_words(text: str, min_len: int = 1) -> list[str]:
"""Extract words from text."""
if min_len < 1:
raise ValueError("min_len must be at least 1")
words = WORD_PATTERN.findall(text)
return [word.lower() for word in words if len(word) >= min_len]
def count_words(text: str, min_len: int = 1) -> Counter[str]:
"""Count word frequencies."""
return Counter(extract_words(text, min_len))
def analyze_text(
text: str,
top: int = 10,
min_len: int = 1,
) -> TextStatistics:
"""Analyze supplied text."""
if top < 0:
raise ValueError("top must be non-negative")
if min_len < 1:
raise ValueError("min_len must be at least 1")
frequencies = count_words(text, min_len)
return TextStatistics(
lines=len(text.splitlines()),
words=sum(frequencies.values()),
characters=len(text),
top_words=frequencies.most_common(top),
)
def analyze_file(
path: Path,
top: int = 10,
min_len: int = 1,
encoding: str = "utf-8",
) -> TextStatistics:
"""Read and analyze a text file."""
text = path.read_text(encoding=encoding)
return analyze_text(
text,
top=top,
min_len=min_len,
)