diff --git a/.github/workflows/publish_docs.yml b/.github/workflows/publish_docs.yml new file mode 100644 index 0000000..9f46adf --- /dev/null +++ b/.github/workflows/publish_docs.yml @@ -0,0 +1,196 @@ +name: Publish Documentation + +on: + push: + branches: + - main + pull_request: + branches: + - main + +jobs: + build-and-deploy-docs: + runs-on: ubuntu-latest + permissions: + contents: write + pages: write + id-token: write + + steps: + - name: Checkout code + uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: Cache pip dependencies + uses: actions/cache@v4 + with: + path: ~/.cache/pip + key: ${{ runner.os }}-pip-docs-${{ hashFiles('docs/requirements.txt') }} + restore-keys: | + ${{ runner.os }}-pip-docs- + ${{ runner.os }}-pip- + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -e . + pip install -r docs/requirements.txt + + - name: Build Sphinx documentation + run: | + cd docs + python -m sphinx -b html . _build -W --keep-going + + - name: Upload documentation artifacts + uses: actions/upload-artifact@v4 + with: + name: documentation + path: docs/_build/ + retention-days: 30 + + - name: Deploy to GitHub Pages + if: github.ref == 'refs/heads/main' && github.event_name == 'push' + uses: peaceiris/actions-gh-pages@v4 + with: + github_token: ${{ secrets.GITHUB_TOKEN }} + publish_dir: docs/_build + force_orphan: true + user_name: 'github-actions[bot]' + user_email: 'github-actions[bot]@users.noreply.github.com' + commit_message: 'Deploy documentation from ${{ github.sha }}' + + docs-quality-check: + runs-on: ubuntu-latest + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -e . + pip install -r docs/requirements.txt + + - name: Check documentation links + run: | + cd docs + sphinx-build -b linkcheck . _build/linkcheck -W --keep-going + continue-on-error: true + + - name: Check documentation coverage + run: | + cd docs + sphinx-build -b coverage . _build/coverage + if [ -f _build/coverage/python.txt ]; then + echo "Documentation coverage report:" + cat _build/coverage/python.txt + fi + continue-on-error: true + branches: + - main + pull_request: + branches: + - main + +jobs: + build-and-deploy-docs: + runs-on: ubuntu-latest + permissions: + contents: write + pages: write + id-token: write + + steps: + - name: Checkout code + uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: Cache pip dependencies + uses: actions/cache@v4 + with: + path: ~/.cache/pip + key: ${{ runner.os }}-pip-docs-${{ hashFiles('docs/requirements.txt') }} + restore-keys: | + ${{ runner.os }}-pip-docs- + ${{ runner.os }}-pip- + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -e . + pip install -r docs/requirements.txt + + - name: Build Sphinx documentation + run: | + cd docs + python -m sphinx -b html . _build -W --keep-going + + - name: Upload documentation artifacts + uses: actions/upload-artifact@v4 + with: + name: documentation + path: docs/_build/ + retention-days: 30 + + - name: Deploy to GitHub Pages + if: github.ref == 'refs/heads/main' && github.event_name == 'push' + uses: peaceiris/actions-gh-pages@v4 + with: + github_token: ${{ secrets.GITHUB_TOKEN }} + publish_dir: docs/_build + force_orphan: true + user_name: 'github-actions[bot]' + user_email: 'github-actions[bot]@users.noreply.github.com' + commit_message: 'Deploy documentation from ${{ github.sha }}' + + # Job to check documentation links and quality + docs-quality-check: + runs-on: ubuntu-latest + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -e . + pip install -r docs/requirements.txt + pip install sphinx-linkcheck + + - name: Check documentation links + run: | + cd docs + sphinx-build -b linkcheck . _build/linkcheck -W --keep-going + continue-on-error: true + + - name: Check documentation coverage + run: | + cd docs + sphinx-build -b coverage . _build/coverage + if [ -f _build/coverage/python.txt ]; then + echo "Documentation coverage report:" + cat _build/coverage/python.txt + fi + continue-on-error: true diff --git a/docs/.gitignore b/docs/.gitignore new file mode 100644 index 0000000..7d00bc5 --- /dev/null +++ b/docs/.gitignore @@ -0,0 +1,21 @@ +# Sphinx build outputs +_build/ +_build_simple/ + +# Sphinx auto-generated files +_autosummary/ + +# Editor files +.vscode/ +*.swp +*.swo +*~ + +# OS files +.DS_Store +Thumbs.db + +# Python cache +__pycache__/ +*.pyc +*.pyo diff --git a/docs/Makefile b/docs/Makefile new file mode 100644 index 0000000..24548a3 --- /dev/null +++ b/docs/Makefile @@ -0,0 +1,33 @@ +# Minimal makefile for Sphinx documentation +# + +# You can set these variables from the command line, and also +# from the environment for the first two. +SPHINXOPTS ?= +SPHINXBUILD ?= sphinx-build +SOURCEDIR = . +BUILDDIR = _build + +# Put it first so that "make" without argument is like "make help". +help: + @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +.PHONY: help Makefile + +# Catch-all target: route all unknown targets to Sphinx using the new +# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS). +%: Makefile + @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +# Custom targets for development +clean-all: + rm -rf $(BUILDDIR)/* + +livehtml: + sphinx-autobuild "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +linkcheck: + @$(SPHINXBUILD) -b linkcheck "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +coverage: + @$(SPHINXBUILD) -b coverage "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 0000000..bac07cc --- /dev/null +++ b/docs/README.md @@ -0,0 +1,170 @@ +# Documentation + +This directory contains the Sphinx documentation for the tzst library. + +## Setup + +1. Install documentation dependencies: + ```bash + pip install -r requirements.txt + ``` + +2. Install the tzst package in development mode (required for autodoc): + ```bash + pip install -e .. + ``` + +## Building Documentation + +### Local Development + +Build the documentation locally: + +```bash +# On Unix/macOS +make html + +# On Windows +make.bat html +``` + +The built documentation will be in `_build/html/`. Open `_build/html/index.html` in your browser. + +### Live Reload (Recommended for Development) + +For automatic rebuilding when files change: + +```bash +# Install sphinx-autobuild if not already installed +pip install sphinx-autobuild + +# Start live reload server +make livehtml +# or +sphinx-autobuild . _build/html +``` + +This will start a local server (usually at http://localhost:8000) that automatically rebuilds and refreshes when you save changes. + +### Other Build Targets + +```bash +# Check documentation coverage +make coverage + +# Check for broken links +make linkcheck + +# Build PDF (requires LaTeX) +make latexpdf + +# Clean build directory +make clean +``` + +## Documentation Structure + +- `index.md` - Main documentation homepage +- `quickstart.md` - Quick start guide for new users +- `examples.md` - Practical examples and use cases +- `changelog.md` - Project changelog +- `api/` - API reference documentation + - `index.md` - API overview + - `core.md` - Core functionality documentation + - `cli.md` - CLI documentation + - `exceptions.md` - Exception classes documentation + +## Writing Documentation + +### Markdown vs reStructuredText + +This documentation uses MyST parser, which allows you to write in Markdown with some reStructuredText features. You can use either `.md` or `.rst` files. + +### Adding New Pages + +1. Create a new `.md` file in the appropriate directory +2. Add it to the relevant `toctree` directive in the parent index file +3. Use proper Markdown headers and cross-references + +### API Documentation + +API documentation is automatically generated from docstrings using Sphinx autodoc. To document a new module: + +1. Add the module to the appropriate API file (e.g., `api/core.md`) +2. Use autodoc directives like `automodule`, `autoclass`, `autofunction` + +### Code Examples + +Use fenced code blocks with language specification: + +````markdown +```python +from tzst import TzstArchive + +with TzstArchive("example.tzst", "w") as archive: + archive.add("file.txt") +``` +```` + +### Cross-References + +Link to other documentation pages: + +```markdown +See the {doc}`quickstart` guide for more information. +``` + +Link to API documentation: + +```markdown +Use the {class}`tzst.TzstArchive` class. +``` + +## Automated Deployment + +Documentation is automatically built and deployed to GitHub Pages when changes are pushed to the main branch. The workflow is defined in `.github/workflows/publish_docs.yml`. + +### Local Testing of Deployment + +To test the deployment process locally: + +1. Build the documentation: `make html` +2. Serve the built files: `python -m http.server 8000 -d _build/html` +3. Visit http://localhost:8000 + +## Troubleshooting + +### Import Errors + +If you get import errors when building documentation: + +1. Make sure the tzst package is installed: `pip install -e ..` +2. Check that all dependencies are installed: `pip install -r requirements.txt` +3. Verify your Python path includes the src directory + +### Theme Issues + +If the RTD theme isn't working: + +1. Install the theme: `pip install sphinx-rtd-theme` +2. Check that it's listed in `requirements.txt` +3. Verify the theme configuration in `conf.py` + +### Build Warnings + +Address all Sphinx warnings to ensure high-quality documentation: + +- Fix broken cross-references +- Add missing docstrings +- Resolve autodoc import issues +- Fix malformed markup + +## Contributing + +When contributing to documentation: + +1. Follow the existing style and structure +2. Test your changes locally before submitting +3. Add examples for new features +4. Update the changelog if appropriate +5. Ensure all links work correctly diff --git a/docs/api/cli.md b/docs/api/cli.md new file mode 100644 index 0000000..ea0b90d --- /dev/null +++ b/docs/api/cli.md @@ -0,0 +1,44 @@ +# CLI API + +The command-line interface module provides functions for the tzst CLI tool. + +```{eval-rst} +.. automodule:: tzst.cli + :members: + :undoc-members: + :show-inheritance: +``` + +## Main Functions + +### main + +```{eval-rst} +.. autofunction:: tzst.cli.main +``` + +### create_parser + +```{eval-rst} +.. autofunction:: tzst.cli.create_parser +``` + +## Utility Functions + +### print_banner + +```{eval-rst} +.. autofunction:: tzst.cli.print_banner +``` + +### format_size + +```{eval-rst} +.. autofunction:: tzst.cli.format_size +``` + +### validate_compression_level + +```{eval-rst} +.. autofunction:: tzst.cli.validate_compression_level +``` diff --git a/docs/api/core.md b/docs/api/core.md new file mode 100644 index 0000000..8bc128e --- /dev/null +++ b/docs/api/core.md @@ -0,0 +1,50 @@ +# Core API + +The core module provides the main functionality for working with tzst archives. + +```{eval-rst} +.. automodule:: tzst.core + :members: + :undoc-members: + :show-inheritance: +``` + +## TzstArchive Class + +The main class for handling `.tzst`/`.tar.zst` archives. + +```{eval-rst} +.. autoclass:: tzst.TzstArchive + :members: + :undoc-members: + :show-inheritance: + :special-members: __init__, __enter__, __exit__ +``` + +## Convenience Functions + +High-level functions for common archive operations. + +### create_archive + +```{eval-rst} +.. autofunction:: tzst.create_archive +``` + +### extract_archive + +```{eval-rst} +.. autofunction:: tzst.extract_archive +``` + +### list_archive + +```{eval-rst} +.. autofunction:: tzst.list_archive +``` + +### test_archive + +```{eval-rst} +.. autofunction:: tzst.test_archive +``` diff --git a/docs/api/exceptions.md b/docs/api/exceptions.md new file mode 100644 index 0000000..9e18785 --- /dev/null +++ b/docs/api/exceptions.md @@ -0,0 +1,22 @@ +# Exceptions + +Custom exception classes used by tzst. + +```{eval-rst} +.. automodule:: tzst.exceptions + :members: + :undoc-members: + :show-inheritance: +``` + +## Exception Hierarchy + +```{eval-rst} +.. autoexception:: tzst.exceptions.TzstArchiveError + :members: + :show-inheritance: + +.. autoexception:: tzst.exceptions.TzstDecompressionError + :members: + :show-inheritance: +``` diff --git a/docs/api/index.md b/docs/api/index.md new file mode 100644 index 0000000..1e9c515 --- /dev/null +++ b/docs/api/index.md @@ -0,0 +1,58 @@ +# API Reference + +This section contains the complete API documentation for tzst. + +```{toctree} +:maxdepth: 2 + +core +cli +exceptions +``` + +## Overview + +The tzst library provides both high-level convenience functions and a comprehensive class-based API for working with `.tzst`/`.tar.zst` archives. + +### Main Components + +- **{doc}`core`**: Core functionality including `TzstArchive` class and convenience functions +- **{doc}`cli`**: Command-line interface functions and utilities +- **{doc}`exceptions`**: Custom exception classes for error handling + +### Quick Reference + +#### Core Classes + +```{eval-rst} +.. currentmodule:: tzst + +.. autosummary:: + :nosignatures: + + TzstArchive +``` + +#### Convenience Functions + +```{eval-rst} +.. autosummary:: + :nosignatures: + + create_archive + extract_archive + list_archive + test_archive +``` + +#### Exceptions + +```{eval-rst} +.. currentmodule:: tzst.exceptions + +.. autosummary:: + :nosignatures: + + TzstArchiveError + TzstDecompressionError +``` diff --git a/docs/build_docs.py b/docs/build_docs.py new file mode 100644 index 0000000..21f155c --- /dev/null +++ b/docs/build_docs.py @@ -0,0 +1,158 @@ +#!/usr/bin/env python3 +"""Development script for building and serving documentation locally.""" + +import argparse +import os +import shutil +import subprocess +import sys +import webbrowser +from pathlib import Path + + +def run_command(cmd, cwd=None): + """Run a shell command and return the result.""" try: + result = subprocess.run( + cmd, shell=True, check=True, cwd=cwd, + capture_output=True, text=True + ) + return result.returncode == 0, result.stdout, result.stderr + except subprocess.CalledProcessError as e: + return False, e.stdout, e.stderr + + +def clean_build(build_dir): + """Clean the build directory.""" + if build_dir.exists(): + print(f"Cleaning {build_dir}") + shutil.rmtree(build_dir) + + +def build_docs(source_dir, build_dir, watch=False): + """Build the documentation.""" + if watch: + print("Starting live reload server...") + print("Visit http://localhost:8000 to view the documentation") + print("Press Ctrl+C to stop the server") + + cmd = f"sphinx-autobuild {source_dir} {build_dir} --host 0.0.0.0 --port 8000" + success, stdout, stderr = run_command(cmd) + + if not success: + print("Failed to start live reload server.") + print("Make sure sphinx-autobuild is installed: pip install sphinx-autobuild") + return False + else: + print(f"Building documentation: {source_dir} -> {build_dir}") + cmd = f"python -m sphinx -b html {source_dir} {build_dir}" + success, stdout, stderr = run_command(cmd) + + if success: + print("Documentation built successfully!") + index_file = build_dir / "index.html" + print(f"Open {index_file} in your browser to view the documentation") + return True + else: + print("Build failed!") + print("STDOUT:", stdout) + print("STDERR:", stderr) + return False + + +def serve_docs(build_dir, port=8000): + """Serve the built documentation locally.""" + if not build_dir.exists(): + print(f"Build directory {build_dir} does not exist. Build the docs first.") + return False + + print(f"Serving documentation at http://localhost:{port}") + print("Press Ctrl+C to stop the server") + + cmd = f"python -m http.server {port}" + success, stdout, stderr = run_command(cmd, cwd=build_dir) + + return success + + +def check_dependencies(): + """Check if required dependencies are installed.""" + try: + import sphinx + print(f"Sphinx version: {sphinx.__version__}") + except ImportError: + print("Sphinx is not installed. Install with: pip install sphinx") + return False + + try: + import tzst + print(f"tzst version: {tzst.__version__}") + except ImportError: + print("tzst package is not installed. Install with: pip install -e ..") + return False + + return True + + +def main(): + parser = argparse.ArgumentParser(description="Build and serve tzst documentation") + parser.add_argument( + "command", + choices=["build", "clean", "serve", "watch", "check"], + help="Command to execute" + ) + parser.add_argument( + "--port", "-p", + type=int, + default=8000, + help="Port for serving documentation (default: 8000)" + ) + parser.add_argument( + "--open", "-o", + action="store_true", + help="Open documentation in browser after building/serving" + ) + + args = parser.parse_args() + + # Get directories + script_dir = Path(__file__).parent + source_dir = script_dir + build_dir = script_dir / "_build" + + if args.command == "check": + success = check_dependencies() + sys.exit(0 if success else 1) + + elif args.command == "clean": + clean_build(build_dir) + + elif args.command == "build": + if not check_dependencies(): + sys.exit(1) + + success = build_docs(source_dir, build_dir) + + if success and args.open: + index_file = build_dir / "index.html" + webbrowser.open(f"file://{index_file.absolute()}") + + sys.exit(0 if success else 1) + + elif args.command == "watch": + if not check_dependencies(): + sys.exit(1) + + success = build_docs(source_dir, build_dir, watch=True) + sys.exit(0 if success else 1) + + elif args.command == "serve": + success = serve_docs(build_dir, args.port) + + if args.open: + webbrowser.open(f"http://localhost:{args.port}") + + sys.exit(0 if success else 1) + + +if __name__ == "__main__": + main() diff --git a/docs/build_docs_simple.py b/docs/build_docs_simple.py new file mode 100644 index 0000000..abe44fc --- /dev/null +++ b/docs/build_docs_simple.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +"""Simple development script for building documentation locally.""" + +import argparse +import shutil +import subprocess +import sys +from pathlib import Path + + +def clean_build(build_dir): + """Clean the build directory.""" + if build_dir.exists(): + print(f"Cleaning {build_dir}") + shutil.rmtree(build_dir) + + +def build_docs(source_dir, build_dir): + """Build the documentation.""" + print(f"Building documentation: {source_dir} -> {build_dir}") + cmd = [ + sys.executable, "-m", "sphinx", + "-b", "html", + str(source_dir), + str(build_dir) + ] + + try: + subprocess.run(cmd, check=True) + print("Documentation built successfully!") + index_file = build_dir / "index.html" + print(f"Open {index_file} in your browser to view the documentation") + return True + except subprocess.CalledProcessError: + print("Build failed!") + return False + + +def main(): + parser = argparse.ArgumentParser(description="Build tzst documentation") + parser.add_argument( + "command", + choices=["build", "clean"], + help="Command to execute" + ) + + args = parser.parse_args() + + # Get directories + script_dir = Path(__file__).parent + source_dir = script_dir + build_dir = script_dir / "_build" + + if args.command == "clean": + clean_build(build_dir) + elif args.command == "build": + success = build_docs(source_dir, build_dir) + sys.exit(0 if success else 1) + + +if __name__ == "__main__": + main() diff --git a/docs/changelog.md b/docs/changelog.md new file mode 100644 index 0000000..6bf6ba9 --- /dev/null +++ b/docs/changelog.md @@ -0,0 +1,45 @@ +# Changelog + +All notable changes to this project will be documented in this file. + +The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), +and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). + +## [1.1.0] - 2025-06-02 + +### Added +- Comprehensive Sphinx documentation with automated GitHub Pages deployment +- API documentation with autodoc integration +- Quickstart guide and examples +- Performance benchmarking examples +- Security-focused extraction examples + +### Changed +- Enhanced documentation structure +- Improved code examples and usage patterns + +### Documentation +- Added complete API reference +- Added practical examples for common use cases +- Added security best practices guide +- Added performance optimization tips + +## [1.0.0] - Previous Release + +### Added +- Initial release of tzst library +- TzstArchive class for archive management +- Convenience functions for common operations +- Command-line interface +- Streaming support for large archives +- Security filters for safe extraction +- Atomic operations for data integrity +- Comprehensive test suite + +### Features +- Support for .tzst/.tar.zst archives +- Zstandard compression with configurable levels +- Memory-efficient streaming mode +- Built-in security protections +- Cross-platform compatibility +- Python 3.12+ support diff --git a/docs/conf.py b/docs/conf.py new file mode 100644 index 0000000..38dbadc --- /dev/null +++ b/docs/conf.py @@ -0,0 +1,90 @@ +# Configuration file for the Sphinx documentation builder. + +import os +import sys + +# Add the source directory to the Python path +sys.path.insert(0, os.path.abspath('../src')) + +# Import version from the package +from tzst import __version__ + +# -- Project information ----------------------------------------------------- +project = 'tzst' +copyright = '2025, Xi Xu' +author = 'Xi Xu' +release = __version__ +version = __version__ + +# -- General configuration --------------------------------------------------- +extensions = [ + 'sphinx.ext.autodoc', + 'sphinx.ext.napoleon', + 'sphinx.ext.viewcode', + 'sphinx.ext.intersphinx', + 'sphinx.ext.autosummary', + 'myst_parser', +] + +templates_path = ['_templates'] +exclude_patterns = ['_build', 'Thumbs.db', '.DS_Store'] + +# -- Options for HTML output ------------------------------------------------- +html_theme = 'sphinx_rtd_theme' +html_static_path = ['_static'] +html_title = f'tzst {version} Documentation' +html_short_title = 'tzst' + +# Theme options +html_theme_options = { + 'canonical_url': 'https://xixu-me.github.io/tzst/', + 'logo_only': False, + 'display_version': True, + 'prev_next_buttons_location': 'bottom', + 'style_external_links': False, + 'style_nav_header_background': '#2980B9', + 'collapse_navigation': True, + 'sticky_navigation': True, + 'navigation_depth': 4, + 'includehidden': True, + 'titles_only': False +} + +# -- Extension configuration ------------------------------------------------- +autodoc_default_options = { + 'members': True, + 'undoc-members': True, + 'show-inheritance': True, + 'special-members': '__init__', + 'exclude-members': '__weakref__' +} + +napoleon_google_docstring = True +napoleon_numpy_docstring = True +napoleon_include_init_with_doc = False +napoleon_include_private_with_doc = False +napoleon_include_special_with_doc = True +napoleon_use_param = True +napoleon_use_rtype = True + +# Intersphinx mapping +intersphinx_mapping = { + 'python': ('https://docs.python.org/3', None), + 'zstandard': ('https://python-zstandard.readthedocs.io/en/latest/', None), +} + +# Autosummary +autosummary_generate = True + +myst_enable_extensions = [ + "colon_fence", + "deflist", + "fieldlist", + "html_admonition", + "linkify", + "replacements", + "smartquotes", + "strikethrough", + "substitution", + "tasklist", +] diff --git a/docs/examples.md b/docs/examples.md new file mode 100644 index 0000000..902294a --- /dev/null +++ b/docs/examples.md @@ -0,0 +1,523 @@ +# Examples + +This page provides practical examples of using tzst in various scenarios. + +## Basic Operations + +### Creating Your First Archive + +```python +from tzst import TzstArchive + +# Create a simple archive +with TzstArchive("my_first_archive.tzst", "w") as archive: + archive.add("important_file.txt") + archive.add("documents/", recursive=True) + +print("Archive created successfully!") +``` + +### Extracting an Archive + +```python +from tzst import TzstArchive + +# Extract everything safely +with TzstArchive("my_first_archive.tzst", "r") as archive: + archive.extract("extracted_files/", filter="data") + +print("Files extracted to extracted_files/") +``` + +## Advanced Usage + +### High-Compression Backup + +```python +from tzst import create_archive +import os + +# Create a highly compressed backup +def create_backup(source_dirs, backup_name): + create_archive( + archive_path=f"{backup_name}.tzst", + files=source_dirs, + compression_level=15, # High compression + ) + + # Check the compression ratio + original_size = sum( + os.path.getsize(os.path.join(dirpath, filename)) + for directory in source_dirs + if os.path.exists(directory) + for dirpath, dirnames, filenames in os.walk(directory) + for filename in filenames + ) + + compressed_size = os.path.getsize(f"{backup_name}.tzst") + ratio = (1 - compressed_size / original_size) * 100 + + print(f"Backup created: {backup_name}.tzst") + print(f"Compression ratio: {ratio:.1f}%") + print(f"Original size: {original_size:,} bytes") + print(f"Compressed size: {compressed_size:,} bytes") + +# Usage +create_backup(["documents/", "photos/", "projects/"], "full_backup") +``` + +### Processing Large Archives with Streaming + +```python +from tzst import TzstArchive + +def process_large_archive(archive_path, output_dir): + """Process a large archive efficiently using streaming mode.""" + + # Use streaming to handle large archives + with TzstArchive(archive_path, "r", streaming=True) as archive: + # First, list contents to understand what we're dealing with + print("Analyzing archive contents...") + contents = archive.list(verbose=True) + + total_files = sum(1 for item in contents if item['is_file']) + total_size = sum(item['size'] for item in contents if item['is_file']) + + print(f"Archive contains {total_files} files ({total_size:,} bytes)") + + # Extract only specific file types + text_files = [item['name'] for item in contents + if item['name'].endswith(('.txt', '.md', '.py'))] + + if text_files: + print(f"Extracting {len(text_files)} text files...") + archive.extract(output_dir, members=text_files, filter="data") + + print("Processing complete!") + +# Usage +process_large_archive("large_dataset.tzst", "extracted_text_files/") +``` + +### Batch Archive Operations + +```python +from tzst import test_archive, list_archive +import os +from pathlib import Path + +def verify_archive_collection(archive_dir): + """Verify integrity of all archives in a directory.""" + + archive_dir = Path(archive_dir) + archives = list(archive_dir.glob("*.tzst")) + + print(f"Found {len(archives)} archives to verify...") + + results = [] + for archive_path in archives: + print(f"Testing {archive_path.name}...") + + try: + # Test integrity + is_valid = test_archive(str(archive_path)) + + if is_valid: + # Get archive info + contents = list_archive(str(archive_path), verbose=True) + file_count = sum(1 for item in contents if item['is_file']) + total_size = sum(item['size'] for item in contents if item['is_file']) + + results.append({ + 'name': archive_path.name, + 'status': 'Valid', + 'file_count': file_count, + 'total_size': total_size + }) + else: + results.append({ + 'name': archive_path.name, + 'status': 'Corrupted', + 'file_count': 0, + 'total_size': 0 + }) + + except Exception as e: + results.append({ + 'name': archive_path.name, + 'status': f'Error: {e}', + 'file_count': 0, + 'total_size': 0 + }) + + # Print summary + print("\n" + "="*60) + print("VERIFICATION SUMMARY") + print("="*60) + + for result in results: + status = result['status'] + if status == 'Valid': + print(f"✓ {result['name']}: {result['file_count']} files, " + f"{result['total_size']:,} bytes") + else: + print(f"✗ {result['name']}: {status}") + + valid_count = sum(1 for r in results if r['status'] == 'Valid') + print(f"\nSummary: {valid_count}/{len(results)} archives are valid") + +# Usage +verify_archive_collection("backup_archives/") +``` + +## Security Examples + +### Safe Archive Extraction + +```python +from tzst import TzstArchive +from tzst.exceptions import TzstArchiveError + +def safe_extract(archive_path, output_dir, max_size_mb=100): + """Safely extract an archive with size limits and security filters.""" + + try: + with TzstArchive(archive_path, "r") as archive: + # First, analyze the archive + contents = archive.list(verbose=True) + + # Check total uncompressed size + total_size = sum(item['size'] for item in contents if item['is_file']) + max_size_bytes = max_size_mb * 1024 * 1024 + + if total_size > max_size_bytes: + print(f"Warning: Archive is {total_size:,} bytes when uncompressed") + print(f"This exceeds the limit of {max_size_bytes:,} bytes") + response = input("Continue anyway? (y/N): ") + if response.lower() != 'y': + return False + + # Check for suspicious files + suspicious_files = [] + for item in contents: + name = item['name'] + # Check for directory traversal attempts + if '..' in name or name.startswith('/'): + suspicious_files.append(name) + # Check for executable files + if name.endswith(('.exe', '.bat', '.sh', '.com')): + suspicious_files.append(name) + + if suspicious_files: + print(f"Warning: Found {len(suspicious_files)} suspicious files:") + for file in suspicious_files[:5]: # Show first 5 + print(f" - {file}") + if len(suspicious_files) > 5: + print(f" ... and {len(suspicious_files) - 5} more") + + response = input("Continue extraction? (y/N): ") + if response.lower() != 'y': + return False + + # Extract with the safest filter + print(f"Extracting {len(contents)} items to {output_dir}") + archive.extract(output_dir, filter="data") + print("Extraction completed safely!") + return True + + except TzstArchiveError as e: + print(f"Archive error: {e}") + return False + except Exception as e: + print(f"Unexpected error: {e}") + return False + +# Usage +safe_extract("untrusted_archive.tzst", "safe_output/", max_size_mb=50) +``` + +### Archive Validation Pipeline + +```python +from tzst import TzstArchive, test_archive +import hashlib +import json +from pathlib import Path + +def create_archive_manifest(archive_path): + """Create a manifest of archive contents for validation.""" + + manifest = { + 'archive_path': str(archive_path), + 'files': [], + 'created_at': str(Path(archive_path).stat().st_mtime) + } + + with TzstArchive(archive_path, "r") as archive: + contents = archive.list(verbose=True) + + for item in contents: + if item['is_file']: + manifest['files'].append({ + 'name': item['name'], + 'size': item['size'], + 'mtime': item['mtime'] + }) + + # Save manifest + manifest_path = Path(archive_path).with_suffix('.manifest.json') + with open(manifest_path, 'w') as f: + json.dump(manifest, f, indent=2) + + print(f"Manifest created: {manifest_path}") + return manifest + +def validate_archive_with_manifest(archive_path): + """Validate an archive against its manifest.""" + + manifest_path = Path(archive_path).with_suffix('.manifest.json') + + if not manifest_path.exists(): + print("No manifest found, creating new one...") + create_archive_manifest(archive_path) + return True + + # Load manifest + with open(manifest_path, 'r') as f: + manifest = json.load(f) + + print("Validating archive integrity...") + if not test_archive(archive_path): + print("❌ Archive integrity check failed!") + return False + + print("Validating against manifest...") + with TzstArchive(archive_path, "r") as archive: + contents = archive.list(verbose=True) + current_files = {item['name']: item for item in contents if item['is_file']} + + manifest_files = {item['name']: item for item in manifest['files']} + + # Check for missing files + missing = set(manifest_files.keys()) - set(current_files.keys()) + if missing: + print(f"❌ Missing files: {', '.join(missing)}") + return False + + # Check for extra files + extra = set(current_files.keys()) - set(manifest_files.keys()) + if extra: + print(f"⚠️ Extra files: {', '.join(extra)}") + + # Check file sizes + size_mismatches = [] + for name, manifest_file in manifest_files.items(): + if name in current_files: + if current_files[name]['size'] != manifest_file['size']: + size_mismatches.append(name) + + if size_mismatches: + print(f"❌ Size mismatches: {', '.join(size_mismatches)}") + return False + + print("✅ Archive validation passed!") + return True + +# Usage +archive_path = "important_backup.tzst" +if validate_archive_with_manifest(archive_path): + print("Archive is valid and matches manifest") +else: + print("Archive validation failed!") +``` + +## Performance Examples + +### Compression Level Comparison + +```python +from tzst import create_archive +import time +import os +from pathlib import Path + +def compression_benchmark(files, output_prefix="test"): + """Compare different compression levels for the same files.""" + + results = [] + levels = [1, 3, 6, 9, 15, 22] # Representative levels + + for level in levels: + archive_path = f"{output_prefix}_level_{level}.tzst" + + print(f"Testing compression level {level}...") + start_time = time.time() + + create_archive( + archive_path=archive_path, + files=files, + compression_level=level + ) + + compression_time = time.time() - start_time + archive_size = os.path.getsize(archive_path) + + results.append({ + 'level': level, + 'time': compression_time, + 'size': archive_size, + 'path': archive_path + }) + + print(f" Time: {compression_time:.2f}s, Size: {archive_size:,} bytes") + + # Print comparison table + print("\n" + "="*70) + print("COMPRESSION LEVEL COMPARISON") + print("="*70) + print(f"{'Level':<6} {'Time (s)':<10} {'Size (MB)':<12} {'Ratio':<8} {'Speed'}") + print("-" * 70) + + baseline_size = results[0]['size'] # Level 1 as baseline + baseline_time = results[0]['time'] + + for result in results: + size_mb = result['size'] / (1024 * 1024) + ratio = result['size'] / baseline_size + speed_factor = baseline_time / result['time'] + + print(f"{result['level']:<6} {result['time']:<10.2f} {size_mb:<12.1f} " + f"{ratio:<8.2f} {speed_factor:<.2f}x") + + # Clean up test files + for result in results: + os.remove(result['path']) + + return results + +# Usage +benchmark_files = ["large_directory/", "data_files/"] +results = compression_benchmark(benchmark_files, "benchmark") +``` + +## Error Handling Examples + +### Robust Archive Processing + +```python +from tzst import TzstArchive, extract_archive +from tzst.exceptions import TzstArchiveError, TzstDecompressionError +import logging + +# Set up logging +logging.basicConfig(level=logging.INFO) +logger = logging.getLogger(__name__) + +def robust_archive_processor(archive_paths, output_base_dir): + """Process multiple archives with comprehensive error handling.""" + + results = { + 'success': [], + 'failed': [], + 'skipped': [] + } + + for archive_path in archive_paths: + try: + logger.info(f"Processing {archive_path}") + + # Create output directory for this archive + archive_name = Path(archive_path).stem + output_dir = Path(output_base_dir) / archive_name + output_dir.mkdir(parents=True, exist_ok=True) + + # First, test the archive + logger.info(f"Testing integrity of {archive_path}") + with TzstArchive(archive_path, "r") as archive: + if not archive.test(): + raise TzstArchiveError(f"Archive {archive_path} failed integrity test") + + # Get archive info + contents = archive.list(verbose=True) + file_count = sum(1 for item in contents if item['is_file']) + total_size = sum(item['size'] for item in contents if item['is_file']) + + logger.info(f"Archive contains {file_count} files ({total_size:,} bytes)") + + # Extract with error handling + logger.info(f"Extracting to {output_dir}") + archive.extract(str(output_dir), filter="data") + + results['success'].append({ + 'path': archive_path, + 'file_count': file_count, + 'total_size': total_size, + 'output_dir': str(output_dir) + }) + + logger.info(f"Successfully processed {archive_path}") + + except TzstArchiveError as e: + logger.error(f"Archive error processing {archive_path}: {e}") + results['failed'].append({ + 'path': archive_path, + 'error': str(e), + 'error_type': 'TzstArchiveError' + }) + + except TzstDecompressionError as e: + logger.error(f"Decompression error processing {archive_path}: {e}") + results['failed'].append({ + 'path': archive_path, + 'error': str(e), + 'error_type': 'TzstDecompressionError' + }) + + except FileNotFoundError: + logger.warning(f"Archive not found: {archive_path}") + results['skipped'].append({ + 'path': archive_path, + 'reason': 'File not found' + }) + + except PermissionError as e: + logger.error(f"Permission error processing {archive_path}: {e}") + results['failed'].append({ + 'path': archive_path, + 'error': str(e), + 'error_type': 'PermissionError' + }) + + except Exception as e: + logger.error(f"Unexpected error processing {archive_path}: {e}") + results['failed'].append({ + 'path': archive_path, + 'error': str(e), + 'error_type': 'UnexpectedError' + }) + + # Print summary + print("\n" + "="*60) + print("PROCESSING SUMMARY") + print("="*60) + print(f"✅ Successfully processed: {len(results['success'])}") + print(f"❌ Failed: {len(results['failed'])}") + print(f"⏭️ Skipped: {len(results['skipped'])}") + + if results['failed']: + print("\nFailures:") + for failure in results['failed']: + print(f" - {failure['path']}: {failure['error_type']}") + + return results + +# Usage +archive_list = [ + "backup1.tzst", + "backup2.tzst", + "backup3.tzst", + "missing_file.tzst" # This will be skipped +] + +results = robust_archive_processor(archive_list, "extracted_archives/") +``` diff --git a/docs/index.md b/docs/index.md new file mode 100644 index 0000000..2172b58 --- /dev/null +++ b/docs/index.md @@ -0,0 +1,61 @@ +# tzst Documentation + +Welcome to **tzst**, the next-generation Python library engineered for modern archive management, leveraging cutting-edge Zstandard compression to deliver superior performance, security, and reliability. + +```{toctree} +:maxdepth: 2 +:caption: Contents: + +quickstart +api/index +examples +changelog +``` + +## What is tzst? + +**tzst** is a Python library built exclusively for Python 3.12+ that provides enterprise-grade solutions for handling `.tzst`/`.tar.zst` archives. It combines atomic operations, streaming efficiency, and a meticulously crafted API to redefine how developers handle compressed archives in production environments. + +## Key Features + +- **🚀 High Performance**: Leverages Zstandard compression for superior speed and compression ratios +- **🔒 Security First**: Built-in extraction filters protect against malicious archives +- **⚡ Streaming Support**: Memory-efficient handling of large archives +- **🛡️ Atomic Operations**: Ensures data integrity with fail-safe file operations +- **🎯 Modern API**: Clean, intuitive interface designed for Python 3.12+ +- **📦 CLI Tools**: Comprehensive command-line interface for everyday tasks + +## Quick Example + +```python +from tzst import TzstArchive + +# Create a new archive +with TzstArchive("backup.tzst", "w", compression_level=5) as archive: + archive.add("documents/") + archive.add("photos/", recursive=True) + +# Extract with security +with TzstArchive("backup.tzst", "r") as archive: + archive.extract("documents/", filter="data") +``` + +## Installation + +Install tzst from PyPI: + +```bash +pip install tzst +``` + +## Getting Started + +For a quick introduction to using tzst, see the {doc}`quickstart` guide. + +For detailed API documentation, browse the {doc}`api/index` section. + +## Indices and tables + +- {ref}`genindex` +- {ref}`modindex` +- {ref}`search` diff --git a/docs/make.bat b/docs/make.bat new file mode 100644 index 0000000..4643ded --- /dev/null +++ b/docs/make.bat @@ -0,0 +1,35 @@ +@ECHO OFF + +pushd %~dp0 + +REM Command file for Sphinx documentation + +if "%SPHINXBUILD%" == "" ( + set SPHINXBUILD=sphinx-build +) +set SOURCEDIR=. +set BUILDDIR=_build + +%SPHINXBUILD% >NUL 2>NUL +if errorlevel 9009 ( + echo. + echo.The 'sphinx-build' command was not found. Make sure you have Sphinx + echo.installed, then set the SPHINXBUILD environment variable to point + echo.to the full path of the 'sphinx-build' executable. Alternatively you + echo.may add the Sphinx directory to PATH. + echo. + echo.If you don't have Sphinx installed, grab it from + echo.https://sphinx-doc.org/ + exit /b 1 +) + +if "%1" == "" goto help + +%SPHINXBUILD% -M %1 %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O% +goto end + +:help +%SPHINXBUILD% -M help %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O% + +:end +popd diff --git a/docs/quickstart.md b/docs/quickstart.md new file mode 100644 index 0000000..6aff8d6 --- /dev/null +++ b/docs/quickstart.md @@ -0,0 +1,222 @@ +# Quick Start Guide + +This guide will help you get started with tzst quickly and efficiently. + +## Installation + +Install tzst using pip: + +```bash +pip install tzst +``` + +## Basic Usage + +### Creating Archives + +Use the `TzstArchive` class or convenience functions to create archives: + +```python +from tzst import TzstArchive, create_archive + +# Using TzstArchive class +with TzstArchive("my_archive.tzst", "w", compression_level=5) as archive: + archive.add("file.txt") + archive.add("directory/", recursive=True) + +# Using convenience function +create_archive( + archive_path="backup.tzst", + files=["documents/", "photos/", "config.txt"], + compression_level=10 +) +``` + +### Extracting Archives + +Extract archives safely with built-in security filters: + +```python +from tzst import TzstArchive, extract_archive + +# Using TzstArchive class +with TzstArchive("my_archive.tzst", "r") as archive: + # Extract all files with security filter + archive.extract("output/", filter="data") + + # Extract specific files + archive.extract("output/", members=["file.txt"], filter="data") + +# Using convenience function +extract_archive("backup.tzst", "restore/") +``` + +### Listing Archive Contents + +View what's inside an archive: + +```python +from tzst import TzstArchive, list_archive + +# Using TzstArchive class +with TzstArchive("my_archive.tzst", "r") as archive: + contents = archive.list(verbose=True) + for item in contents: + print(f"{item['name']} - {item['size']} bytes") + +# Using convenience function +files = list_archive("backup.tzst", verbose=True) +``` + +### Testing Archive Integrity + +Verify that an archive is valid: + +```python +from tzst import TzstArchive, test_archive + +# Using TzstArchive class +with TzstArchive("my_archive.tzst", "r") as archive: + is_valid = archive.test() + print(f"Archive is {'valid' if is_valid else 'corrupted'}") + +# Using convenience function +if test_archive("backup.tzst"): + print("Archive is valid") +``` + +## Command Line Interface + +tzst provides a comprehensive CLI for archive operations: + +### Creating Archives + +```bash +# Create an archive with multiple files +tzst a backup.tzst documents/ photos/ config.txt + +# Create with high compression +tzst a -l 15 backup.tzst large_files/ + +# Create without atomic operations (faster, less safe) +tzst a --no-atomic backup.tzst files/ +``` + +### Extracting Archives + +```bash +# Extract all files (default: safe extraction) +tzst x backup.tzst + +# Extract to specific directory +tzst x backup.tzst -o restore/ + +# Extract specific files only +tzst x backup.tzst config.txt documents/ + +# Extract with streaming (memory efficient) +tzst x backup.tzst --streaming +``` + +### Listing Contents + +```bash +# Simple listing +tzst l backup.tzst + +# Detailed listing with file info +tzst l backup.tzst -v + +# Streaming mode for large archives +tzst l backup.tzst --streaming +``` + +### Testing Archives + +```bash +# Test archive integrity +tzst t backup.tzst + +# Test with streaming +tzst t backup.tzst --streaming +``` + +## Security Considerations + +tzst includes built-in security features to protect against malicious archives: + +### Extraction Filters + +Always use appropriate filters when extracting archives from untrusted sources: + +- **`data`** (default): Safest option, only extracts regular files and directories +- **`tar`**: Honors most tar features but still secure +- **`fully_trusted`**: No restrictions (only use with completely trusted archives) + +```python +# Safe extraction (recommended) +archive.extract("output/", filter="data") + +# Command line +tzst x archive.tzst --filter=data +``` + +### Best Practices + +1. **Always use the default `data` filter** for untrusted archives +2. **Enable atomic operations** (default) for data integrity +3. **Use streaming mode** for very large archives to save memory +4. **Validate archives** with `test()` before processing +5. **Specify output directories** explicitly to avoid overwrites + +## Performance Tips + +### Memory Efficiency + +For large archives, use streaming mode: + +```python +# Streaming mode uses less memory +with TzstArchive("large.tzst", "r", streaming=True) as archive: + archive.extract("output/") +``` + +### Compression Levels + +Choose appropriate compression levels based on your needs: + +- **Level 1-3**: Fast compression, larger files +- **Level 3-6**: Balanced (default: 3) +- **Level 7-15**: Better compression, slower +- **Level 16-22**: Maximum compression, much slower + +```python +# Fast compression for temporary files +TzstArchive("temp.tzst", "w", compression_level=1) + +# Maximum compression for long-term storage +TzstArchive("backup.tzst", "w", compression_level=15) +``` + +## Error Handling + +tzst provides specific exceptions for different error conditions: + +```python +from tzst import TzstArchive +from tzst.exceptions import TzstArchiveError, TzstDecompressionError + +try: + with TzstArchive("archive.tzst", "r") as archive: + archive.extract("output/") +except TzstArchiveError as e: + print(f"Archive error: {e}") +except TzstDecompressionError as e: + print(f"Decompression error: {e}") +``` + +## Next Steps + +- Explore the complete {doc}`api/index` documentation +- Check out more {doc}`examples` and use cases +- Read about advanced features in the full documentation diff --git a/docs/requirements.txt b/docs/requirements.txt new file mode 100644 index 0000000..b405fb7 --- /dev/null +++ b/docs/requirements.txt @@ -0,0 +1,17 @@ +# Documentation requirements for Sphinx +sphinx>=7.1.0 +sphinx-rtd-theme>=2.0.0 +myst-parser>=3.0.0 +sphinxcontrib-napoleon>=0.7 + +# Additional Sphinx extensions +sphinx-autobuild>=2021.3.14 +sphinx-copybutton>=0.5.2 +sphinxext-opengraph>=0.9.0 +sphinx-autodoc-typehints>=1.25.0 + +# Alternative modern theme (optional) +furo>=2024.1.29 + +# Main package dependencies (needed for autodoc to import modules) +zstandard>=0.19.0,<1.0.0