diff --git a/dist/index.js b/dist/index.js
index 5b19419..5d8811d 100644
--- a/dist/index.js
+++ b/dist/index.js
@@ -1,14 +1,14 @@
(function() {
var modules = [
'modules/runtime.js',
- (params.get('lang') === 'ru' ? 'modules/packages/ru.js' : 'modules/packages/en.js'),
+ (currentLanguage === 'ru' ? 'modules/packages/ru.js' : 'modules/packages/en.js'),
'modules/loader.js',
'modules/fs.js',
'modules/audio.js',
'modules/graphics.js',
'modules/events.js',
'modules/fetch.js',
- (params.get('lang') === 'ru' ? 'modules/asm_consts/ru.js' : 'modules/asm_consts/en.js'),
+ (currentLanguage === 'ru' ? 'modules/asm_consts/ru.js' : 'modules/asm_consts/en.js'),
// 'modules/cheats.js',
'modules/main.js'
];
diff --git a/docker-compose.yml b/docker-compose.yml
index 389dc32..d5dd16b 100644
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -21,15 +21,21 @@ services:
- VCBR_URL=${VCBR_URL:-}
- VCSKY_CACHE=${VCSKY_CACHE:-}
- VCBR_CACHE=${VCBR_CACHE:-}
- command: >
+ - PACKED=${PACKED:-}
+ - UNPACKED=${UNPACKED:-}
+ - PACK=${PACK:-}
+ command: >
sh -c "python server.py
--port $${IN_PORT:-8000}
$$([ -n \"$$AUTH_LOGIN\" ] && [ -n \"$$AUTH_PASSWORD\" ] && echo \"--login $$AUTH_LOGIN --password $$AUTH_PASSWORD\" || echo '')
$$([ \"$$CUSTOM_SAVES\" = '1' ] && echo '--custom_saves' || echo '')
- $$([ \"$$VCSKY_LOCAL\" = '1' ] && echo '--vcsky_local' || echo '')
- $$([ \"$$VCBR_LOCAL\" = '1' ] && echo '--vcbr_local' || echo '')
+ $$(if [ \"$$VCSKY_LOCAL\" = '1' ]; then echo '--vcsky_local'; elif [ -n \"$$VCSKY_LOCAL\" ]; then echo \"--vcsky_local $$VCSKY_LOCAL\"; fi)
+ $$(if [ \"$$VCBR_LOCAL\" = '1' ]; then echo '--vcbr_local'; elif [ -n \"$$VCBR_LOCAL\" ]; then echo \"--vcbr_local $$VCBR_LOCAL\"; fi)
$$([ -n \"$$VCSKY_URL\" ] && echo \"--vcsky_url $$VCSKY_URL\" || echo '')
$$([ -n \"$$VCBR_URL\" ] && echo \"--vcbr_url $$VCBR_URL\" || echo '')
$$([ \"$$VCSKY_CACHE\" = '1' ] && echo '--vcsky_cache' || echo '')
- $$([ \"$$VCBR_CACHE\" = '1' ] && echo '--vcbr_cache' || echo '')"
+ $$([ \"$$VCBR_CACHE\" = '1' ] && echo '--vcbr_cache' || echo '')
+ $$([ -n \"$$PACKED\" ] && echo \"--packed $$PACKED\" || echo '')
+ $$([ -n \"$$UNPACKED\" ] && echo \"--unpacked $$UNPACKED\" || echo '')
+ $$([ -n \"$$PACK\" ] && echo \"--pack $$PACK\" || echo '')"
restart: unless-stopped
diff --git a/requirements.txt b/requirements.txt
index 5f6519a..abe6804 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -2,4 +2,5 @@ fastapi
httpx
uvicorn
brotli
-python-multipart
\ No newline at end of file
+python-multipart
+aiofiles
\ No newline at end of file
diff --git a/server.py b/server.py
index fe40f88..74c54c9 100644
--- a/server.py
+++ b/server.py
@@ -1,25 +1,276 @@
import os
+import sys
+import asyncio
import argparse
+import hashlib
+from typing import Optional
from fastapi import FastAPI, Request, HTTPException
from fastapi.responses import Response
from fastapi.staticfiles import StaticFiles
import additions.saves as saves
from additions.auth import BasicAuthMiddleware
from additions.cache import proxy_and_cache, get_local_file
+from additions.packed import init_packed_archive, get_packed_file, is_initialized as packed_is_initialized
+
+# Add utils path for imports
+sys.path.insert(0, os.path.join(os.path.dirname(__file__), 'utils'))
parser = argparse.ArgumentParser()
parser.add_argument("--port", type=int, default=8000)
parser.add_argument("--custom_saves", action="store_true")
parser.add_argument("--login", type=str)
parser.add_argument("--password", type=str)
-parser.add_argument("--vcsky_local", action="store_true", help="Serve vcsky from local directory instead of proxy")
-parser.add_argument("--vcbr_local", action="store_true", help="Serve vcbr from local directory instead of proxy")
+parser.add_argument("--vcsky_local", type=str, nargs='?', const='vcsky', default=None,
+ help="Serve vcsky from local directory instead of proxy. Optionally specify path (default: vcsky/)")
+parser.add_argument("--vcbr_local", type=str, nargs='?', const='vcbr', default=None,
+ help="Serve vcbr from local directory instead of proxy. Optionally specify path (default: vcbr/)")
parser.add_argument("--vcsky_url", type=str, default="https://cdn.dos.zone/vcsky/", help="Custom vcsky proxy URL")
parser.add_argument("--vcbr_url", type=str, default="https://br.cdn.dos.zone/vcsky/", help="Custom vcbr proxy URL")
parser.add_argument("--vcsky_cache", action="store_true", help="Cache vcsky files locally. If files are not found in the local directory, they will be downloaded from the specified URL and saved to the local directory.")
parser.add_argument("--vcbr_cache", action="store_true", help="Cache vcbr files locally. If files are not found in the local directory, they will be downloaded from the specified URL and saved to the local directory.")
+parser.add_argument("--packed", type=str, nargs='?', const='revcdos.bin', default=None,
+ help="Serve vcsky/ and vcbr/ from packed archive. Can be a local file path or URL. "
+ "If URL, downloads to local file if not present. If no value specified, uses 'revcdos.bin'. "
+ "Supports brotli passthrough.")
+parser.add_argument("--unpacked", type=str, default=None,
+ help="Unpack archive to local folders and serve from there. Can be a local .bin file or URL. "
+ "Unpacks to unpacked/{md5_hash}/ and sets vcsky_local/vcbr_local automatically. "
+ "If already unpacked, uses existing files without re-unpacking. "
+ "If URL, streams and unpacks during download using downloader_brotli.")
+parser.add_argument("--pack", type=str, default=None,
+ help="Pack a folder to {hash}.bin archive. Can be a folder path or MD5 hash from unpacked/. "
+ "Packs all subfolders (vcsky/, vcbr/, etc.) into a single archive. "
+ "After packing, uses the archive with --packed mode to serve files.")
args = parser.parse_args()
+
+def _md5_hash(text: str) -> str:
+ """Get MD5 hash of text."""
+ return hashlib.md5(text.encode()).hexdigest()
+
+
+def _is_url(path: str) -> bool:
+ """Check if path is a URL."""
+ return path.startswith("http://") or path.startswith("https://")
+
+
+def _is_md5_hash(text: str) -> bool:
+ """Check if text is a valid MD5 hash (32 hex characters)."""
+ if len(text) != 32:
+ return False
+ try:
+ int(text, 16)
+ return True
+ except ValueError:
+ return False
+
+
+def _get_unpacked_dir(source: str) -> str:
+ """
+ Get unpacked directory path for a source.
+
+ If source IS a valid MD5 hash (32 hex chars), uses it directly.
+ Otherwise computes MD5 hash from the source string.
+ """
+ # Check if source itself is a valid MD5 hash
+ if _is_md5_hash(source):
+ return os.path.join("unpacked", source.lower())
+
+ # Compute hash from source
+ source_hash = _md5_hash(source)
+ return os.path.join("unpacked", source_hash)
+
+
+def _check_unpacked_exists(unpacked_dir: str) -> bool:
+ """Check if unpacked directory exists and has content."""
+ if not os.path.isdir(unpacked_dir):
+ return False
+
+ # Check if vcsky or vcbr subdirectory exists with files
+ for subdir in ["vcsky", "vcbr"]:
+ subdir_path = os.path.join(unpacked_dir, subdir)
+ if os.path.isdir(subdir_path):
+ # Check if there are any files in subdirectories
+ for root, dirs, files in os.walk(subdir_path):
+ if files:
+ return True
+
+ return False
+
+
+async def _unpack_from_url(url: str, output_dir: str) -> bool:
+ """
+ Unpack archive directly from URL using streaming download.
+ Uses downloader_brotli for efficient stream unpacking.
+ """
+ try:
+ from utils.downloader_brotli import download_and_unpack_async
+ print(f"Streaming and unpacking from URL: {url}")
+ print(f"Output directory: {output_dir}")
+ await download_and_unpack_async(url, output_dir)
+ return True
+ except Exception as e:
+ print(f"Error unpacking from URL: {e}")
+ return False
+
+
+async def _unpack_from_file(file_path: str, output_dir: str) -> bool:
+ """
+ Unpack archive from local file.
+ Uses packer_brotli.unpack_file for unpacking.
+ """
+ try:
+ from utils.packer_brotli import unpack_file
+ print(f"Unpacking local file: {file_path}")
+ print(f"Output directory: {output_dir}")
+
+ # Run sync unpack in executor
+ loop = asyncio.get_event_loop()
+ await loop.run_in_executor(None, unpack_file, file_path, output_dir)
+ return True
+ except Exception as e:
+ print(f"Error unpacking file: {e}")
+ return False
+
+
+def pack_source(source: str) -> Optional[str]:
+ """
+ Pack folder contents into {hash}.bin archive.
+
+ If source is an MD5 hash, uses unpacked/{hash}/ folder.
+ Otherwise uses the folder path directly.
+
+ Packs all subfolders (vcsky/, vcbr/, etc.) by:
+ 1. Creating archive from first subfolder using pack_folder()
+ 2. Adding remaining subfolders using add_folder()
+
+ Args:
+ source: Folder path or MD5 hash
+
+ Returns:
+ Output filename (e.g., "abc123...def.bin") or None if failed
+ """
+ from utils.packer_brotli import pack_folder, add_folder
+
+ # Resolve source to folder path and output hash
+ if _is_md5_hash(source):
+ folder_path = os.path.join("unpacked", source.lower())
+ output_hash = source.lower()
+ else:
+ folder_path = source.rstrip('/\\')
+ output_hash = _md5_hash(os.path.basename(folder_path))
+
+ if not os.path.isdir(folder_path):
+ print(f"Error: Folder not found: {folder_path}")
+ return None
+
+ output_file = f"{output_hash}.bin"
+
+ # Get immediate subdirectories (vcsky, vcbr, etc.)
+ subdirs = sorted([d for d in os.listdir(folder_path)
+ if os.path.isdir(os.path.join(folder_path, d)) and not d.startswith('.')])
+
+ if not subdirs:
+ print(f"Error: No subdirectories found in {folder_path}")
+ return None
+
+ print(f"Packing {len(subdirs)} subfolders from {folder_path} to {output_file}")
+ print(f"Subfolders: {', '.join(subdirs)}")
+ print()
+
+ # Pack first subfolder (creates new archive)
+ first_subdir = os.path.join(folder_path, subdirs[0])
+ print(f"=== Creating archive from {subdirs[0]} ===")
+ pack_folder(first_subdir, output_file)
+
+ # Add remaining subfolders
+ for subdir_name in subdirs[1:]:
+ subdir_path = os.path.join(folder_path, subdir_name)
+ print(f"\n=== Adding {subdir_name} ===")
+ add_folder(output_file, subdir_path)
+
+ final_size = os.path.getsize(output_file)
+ print(f"\n=== Packing complete ===")
+ print(f"Output: {output_file} ({final_size:,} bytes)")
+
+ return output_file
+
+
+async def setup_unpacked(source: str) -> tuple:
+ """
+ Setup unpacked mode - unpack archive if needed and return local paths.
+
+ Args:
+ source: Local file path, URL to packed archive, or MD5 hash of existing unpacked folder
+
+ Returns:
+ Tuple of (vcsky_local_path, vcbr_local_path) or (None, None) if failed
+ """
+ unpacked_dir = _get_unpacked_dir(source)
+
+ # Check if source is just an MD5 hash (use existing folder only)
+ is_hash_only = _is_md5_hash(source)
+
+ # Check if already unpacked
+ if _check_unpacked_exists(unpacked_dir):
+ print(f"Using existing unpacked directory: {unpacked_dir}")
+ elif is_hash_only:
+ # Source is MD5 hash but folder doesn't exist - error
+ print(f"Error: Unpacked folder not found for hash: {source}")
+ print(f"Expected directory: {unpacked_dir}")
+ return None, None
+ else:
+ # Need to unpack
+ print(f"Unpacking to: {unpacked_dir}")
+ os.makedirs(unpacked_dir, exist_ok=True)
+
+ if _is_url(source):
+ # Stream unpack from URL
+ success = await _unpack_from_url(source, unpacked_dir)
+ else:
+ # Unpack from local file
+ if not os.path.isfile(source):
+ print(f"Error: Archive file not found: {source}")
+ return None, None
+ success = await _unpack_from_file(source, unpacked_dir)
+
+ if not success:
+ print(f"Failed to unpack from: {source}")
+ return None, None
+
+ # Determine vcsky and vcbr paths
+ vcsky_path = None
+ vcbr_path = None
+
+ # Check for vcsky folder
+ vcsky_candidate = os.path.join(unpacked_dir, "vcsky")
+ if os.path.isdir(vcsky_candidate):
+ vcsky_path = vcsky_candidate
+ print(f" vcsky: {vcsky_path}")
+
+ # Check for vcbr folder
+ vcbr_candidate = os.path.join(unpacked_dir, "vcbr")
+ if os.path.isdir(vcbr_candidate):
+ vcbr_path = vcbr_candidate
+ print(f" vcbr: {vcbr_path}")
+
+ if not vcsky_path and not vcbr_path:
+ print(f"Warning: No vcsky or vcbr folders found in {unpacked_dir}")
+ # Maybe the folders are directly in unpacked_dir without vcsky/vcbr prefix
+ # Check if there's a subfolder that looks like the archive name
+ for item in os.listdir(unpacked_dir):
+ item_path = os.path.join(unpacked_dir, item)
+ if os.path.isdir(item_path):
+ vcsky_sub = os.path.join(item_path, "vcsky")
+ vcbr_sub = os.path.join(item_path, "vcbr")
+ if os.path.isdir(vcsky_sub):
+ vcsky_path = vcsky_sub
+ if os.path.isdir(vcbr_sub):
+ vcbr_path = vcbr_sub
+
+ return vcsky_path, vcbr_path
+
+
app = FastAPI()
if args.login and args.password:
@@ -31,6 +282,11 @@ if args.custom_saves:
VCSKY_BASE_URL = args.vcsky_url
VCBR_BASE_URL = args.vcbr_url
+# Local paths (can be overridden by --unpacked)
+VCSKY_LOCAL_PATH = args.vcsky_local # None, 'vcsky', or custom path
+VCBR_LOCAL_PATH = args.vcbr_local # None, 'vcbr', or custom path
+
+
def request_to_url(request: Request, path: str, base_url: str):
query_string = str(request.url.query) if request.url.query else ""
url = f"{base_url}{path}"
@@ -38,32 +294,59 @@ def request_to_url(request: Request, path: str, base_url: str):
url = f"{url}?{query_string}"
return url
-# vcsky routes - either local or proxy
+
+# vcsky routes - packed archive, local, or proxy
@app.api_route("/vcsky/{path:path}", methods=["GET", "POST", "PUT", "DELETE", "PATCH", "HEAD", "OPTIONS"])
async def vc_sky_proxy(request: Request, path: str):
- local_path = os.path.join("vcsky", path)
- if args.vcsky_local:
+ # Try packed archive first if enabled
+ if args.packed and packed_is_initialized():
+ packed_path = f"vcsky/{path}"
+ if response := await get_packed_file(packed_path, request):
+ return response
+
+ # Try local directory
+ if VCSKY_LOCAL_PATH:
+ local_path = os.path.join(VCSKY_LOCAL_PATH, path)
if response := get_local_file(local_path, request):
return response
- raise HTTPException(status_code=404, detail="File not found")
+ # If local mode is explicitly set, don't fall through to proxy
+ if args.vcsky_local is not None or args.unpacked:
+ raise HTTPException(status_code=404, detail="File not found")
+
+ # Proxy mode
url = request_to_url(request, path, VCSKY_BASE_URL)
if args.vcsky_cache:
- return await proxy_and_cache(request, url, local_path)
+ cache_path = os.path.join("vcsky", path)
+ return await proxy_and_cache(request, url, cache_path)
return await proxy_and_cache(request, url, disable_cache=True)
-# vcbr routes - either local or proxy
+
+# vcbr routes - packed archive, local, or proxy
@app.api_route("/vcbr/{path:path}", methods=["GET", "POST", "PUT", "DELETE", "PATCH", "HEAD", "OPTIONS"])
async def vc_br_proxy(request: Request, path: str):
- local_path = os.path.join("vcbr", path)
- if args.vcbr_local:
+ # Try packed archive first if enabled
+ if args.packed and packed_is_initialized():
+ packed_path = f"vcbr/{path}"
+ if response := await get_packed_file(packed_path, request):
+ return response
+
+ # Try local directory
+ if VCBR_LOCAL_PATH:
+ local_path = os.path.join(VCBR_LOCAL_PATH, path)
if response := get_local_file(local_path, request):
return response
- raise HTTPException(status_code=404, detail="File not found")
+ # If local mode is explicitly set, don't fall through to proxy
+ if args.vcbr_local is not None or args.unpacked:
+ raise HTTPException(status_code=404, detail="File not found")
+
+ # Proxy mode
url = request_to_url(request, path, VCBR_BASE_URL)
if args.vcbr_cache:
- return await proxy_and_cache(request, url, local_path)
+ cache_path = os.path.join("vcbr", path)
+ return await proxy_and_cache(request, url, cache_path)
return await proxy_and_cache(request, url, disable_cache=True)
+
@app.get("/")
async def read_index():
if os.path.exists("dist/index.html"):
@@ -85,12 +368,62 @@ async def read_index():
app.mount("/", StaticFiles(directory="dist"), name="root")
+
+async def init_server():
+ """Initialize server components that need async init."""
+ global VCSKY_LOCAL_PATH, VCBR_LOCAL_PATH
+
+ # Handle --unpacked mode first (takes precedence)
+ if args.unpacked:
+ vcsky_path, vcbr_path = await setup_unpacked(args.unpacked)
+ if vcsky_path:
+ VCSKY_LOCAL_PATH = vcsky_path
+ if vcbr_path:
+ VCBR_LOCAL_PATH = vcbr_path
+
+ # Handle --packed mode
+ if args.packed:
+ # init_packed_archive handles both local paths and URLs
+ # If URL is provided, it will download the file if not present locally
+ result = await init_packed_archive(args.packed)
+ if result is None:
+ print(f"Warning: Failed to initialize packed archive from: {args.packed}")
+
+
def start_server(app=app, host="0.0.0.0", port=args.port):
import uvicorn
+
+ # Initialize server components
+ if args.packed or args.unpacked:
+ asyncio.run(init_server())
+
uvicorn.run(app, host=host, port=port)
+
if __name__ == "__main__":
+ # Handle --pack first (pack folder then use packed mode)
+ if args.pack:
+ print(f"Pack mode: {args.pack}")
+ packed_file = pack_source(args.pack)
+ if packed_file:
+ print(f"\nUsing packed archive: {packed_file}")
+ args.packed = packed_file
+ else:
+ print("Packing failed, exiting.")
+ sys.exit(1)
+
print(f"Starting server on http://localhost:{args.port}")
- print(f"vcsky: {'local' if args.vcsky_local else 'proxy'} ({VCSKY_BASE_URL if not args.vcsky_local else 'vcsky/'})")
- print(f"vcbr: {'local' if args.vcbr_local else 'proxy'} ({VCBR_BASE_URL if not args.vcbr_local else 'vcbr/'})")
- start_server()
\ No newline at end of file
+
+ if args.unpacked:
+ print(f"unpacked mode: {args.unpacked}")
+ elif args.packed:
+ print(f"packed: {args.packed}")
+ else:
+ vcsky_mode = 'local' if args.vcsky_local else 'proxy'
+ vcbr_mode = 'local' if args.vcbr_local else 'proxy'
+ vcsky_info = args.vcsky_local or VCSKY_BASE_URL
+ vcbr_info = args.vcbr_local or VCBR_BASE_URL
+ print(f"vcsky: {vcsky_mode} ({vcsky_info})")
+ print(f"vcbr: {vcbr_mode} ({vcbr_info})")
+
+ start_server()
diff --git a/utils/downloader_brotli.py b/utils/downloader_brotli.py
new file mode 100644
index 0000000..fdce38c
--- /dev/null
+++ b/utils/downloader_brotli.py
@@ -0,0 +1,457 @@
+#!/usr/bin/env python3
+"""
+Downloader for packed files with brotli-compressed content.
+Downloads and unpacks directly to disk without saving intermediate .bin file.
+Shows detailed progress and statistics.
+Uses separate coroutines for downloading and unpacking with asyncio.Queue.
+
+This version uses packer_brotli format where individual files are brotli-compressed.
+The stream_unpack_async from packer_brotli automatically decompresses each file.
+"""
+
+import os
+import sys
+import asyncio
+import shutil
+import time
+from dataclasses import dataclass, field
+from typing import Dict, Optional, Tuple
+
+import httpx
+import aiofiles
+
+sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
+from packer_brotli import stream_unpack_async
+
+
+def format_size(size_bytes: int) -> str:
+ """Format bytes to human readable size."""
+ for unit in ['B', 'KB', 'MB', 'GB']:
+ if size_bytes < 1024:
+ return f"{size_bytes:.2f} {unit}"
+ size_bytes /= 1024
+ return f"{size_bytes:.2f} TB"
+
+
+def format_time(seconds: float) -> str:
+ """Format seconds to human readable time."""
+ if seconds < 60:
+ return f"{seconds:.1f}s"
+ elif seconds < 3600:
+ minutes = int(seconds // 60)
+ secs = seconds % 60
+ return f"{minutes}m {secs:.1f}s"
+ else:
+ hours = int(seconds // 3600)
+ minutes = int((seconds % 3600) // 60)
+ return f"{hours}h {minutes}m"
+
+
+def get_terminal_width() -> int:
+ """Get terminal width, default to 80 if not available."""
+ try:
+ return shutil.get_terminal_size().columns
+ except Exception:
+ return 80
+
+
+@dataclass
+class UnpackStats:
+ """Statistics for unpacking progress."""
+ start_time: float = field(default_factory=time.time)
+
+ # Folder tracking
+ current_folder: str = ""
+ files_in_current_folder: int = 0
+ unpacked_in_current_folder: int = 0
+
+ # Global tracking
+ total_folders: int = 0
+ total_files: int = 0
+ total_bytes: int = 0 # Decompressed file content bytes
+ total_compressed_bytes: int = 0 # Compressed file bytes in archive
+ copied_folders: int = 0
+ copied_files: int = 0 # Individual file copies (references)
+
+ # Download tracking
+ downloaded_bytes: int = 0
+ download_complete: bool = False
+
+ # Per-folder stats
+ folder_stats: Dict[str, Dict] = field(default_factory=dict)
+ folder_file_counts: Dict[str, int] = field(default_factory=dict)
+
+ # Track unpacked folders and files for copy references
+ unpacked_folders: Dict[str, str] = field(default_factory=dict)
+ unpacked_files: Dict[Tuple[str, str], str] = field(default_factory=dict)
+
+ # Last printed line length for proper clearing
+ last_line_length: int = 0
+
+ def start_folder(self, folder_name: str, num_files: int):
+ """Start tracking a new folder."""
+ self.current_folder = folder_name
+ self.files_in_current_folder = num_files
+ self.unpacked_in_current_folder = 0
+ self.total_folders += 1
+ self.folder_stats[folder_name] = {
+ 'total_files': num_files,
+ 'unpacked_files': 0,
+ 'total_bytes': 0,
+ 'compressed_bytes': 0
+ }
+ self.folder_file_counts[folder_name] = num_files
+
+ def file_unpacked(self, filename: str, compressed_size: int, decompressed_size: int):
+ """Record a file being unpacked."""
+ self.unpacked_in_current_folder += 1
+ self.total_files += 1
+ self.total_bytes += decompressed_size
+ self.total_compressed_bytes += compressed_size
+
+ if self.current_folder in self.folder_stats:
+ self.folder_stats[self.current_folder]['unpacked_files'] += 1
+ self.folder_stats[self.current_folder]['total_bytes'] += decompressed_size
+ self.folder_stats[self.current_folder]['compressed_bytes'] += compressed_size
+
+ def file_copied(self, filename: str, file_size: int):
+ """Record a file being copied from a reference."""
+ self.unpacked_in_current_folder += 1
+ self.total_files += 1
+ self.total_bytes += file_size
+ self.copied_files += 1
+
+ if self.current_folder in self.folder_stats:
+ self.folder_stats[self.current_folder]['unpacked_files'] += 1
+ self.folder_stats[self.current_folder]['total_bytes'] += file_size
+
+ def add_downloaded(self, size: int):
+ """Add downloaded bytes."""
+ self.downloaded_bytes += size
+
+ def get_elapsed(self) -> float:
+ """Get elapsed time in seconds."""
+ return time.time() - self.start_time
+
+ def clear_line(self):
+ """Clear the current line properly."""
+ term_width = get_terminal_width()
+ print('\r' + ' ' * min(self.last_line_length, term_width - 1) + '\r', end='', flush=True)
+
+ def print_progress(self, filename: str, compressed_size: int, decompressed_size: int, is_copy: bool = False):
+ """Print current progress on a single line."""
+ elapsed = self.get_elapsed()
+ speed = self.total_bytes / elapsed if elapsed > 0 else 0
+
+ # Calculate progress percentage
+ if self.files_in_current_folder > 0:
+ progress_pct = self.unpacked_in_current_folder / self.files_in_current_folder * 100
+ else:
+ progress_pct = 0
+
+ # Progress bar
+ bar_len = 15
+ filled = int(bar_len * self.unpacked_in_current_folder / self.files_in_current_folder) if self.files_in_current_folder > 0 else 0
+ bar = '█' * filled + '░' * (bar_len - filled)
+
+ # Truncate folder name if too long
+ folder_display = self.current_folder
+ if len(folder_display) > 25:
+ folder_display = '...' + folder_display[-22:]
+
+ # Truncate filename if too long
+ file_display = filename
+ if len(file_display) > 15:
+ file_display = file_display[:12] + '...'
+
+ # Download indicator
+ dl_indicator = "⬇️" if not self.download_complete else "✓"
+
+ # Copy indicator
+ copy_marker = "📋" if is_copy else ""
+
+ # Compression info
+ if not is_copy and compressed_size > 0:
+ ratio = decompressed_size / compressed_size if compressed_size > 0 else 1
+ size_info = f"{format_size(compressed_size)}->{format_size(decompressed_size)} ({ratio:.1f}x)"
+ else:
+ size_info = format_size(decompressed_size)
+
+ # Build the line
+ line = (f"[{bar}] {progress_pct:5.1f}% | "
+ f"{folder_display} | "
+ f"{self.unpacked_in_current_folder}/{self.files_in_current_folder}: {copy_marker}{file_display} | "
+ f"{size_info} | "
+ f"{format_size(speed)}/s | "
+ f"{dl_indicator} {format_size(self.downloaded_bytes)}")
+
+ # Get terminal width and truncate if needed
+ term_width = get_terminal_width()
+ if len(line) > term_width - 1:
+ line = line[:term_width - 4] + '...'
+
+ # Clear previous line and print new one
+ self.clear_line()
+ print(line, end='', flush=True)
+ self.last_line_length = len(line)
+
+ def print_folder_complete(self):
+ """Print folder completion message."""
+ self.clear_line()
+ folder_data = self.folder_stats.get(self.current_folder, {})
+ compressed = folder_data.get('compressed_bytes', 0)
+ decompressed = folder_data.get('total_bytes', 0)
+ if compressed > 0:
+ ratio = decompressed / compressed
+ print(f"✓ {self.current_folder}: "
+ f"{folder_data.get('unpacked_files', 0)} files, "
+ f"{format_size(compressed)} -> {format_size(decompressed)} ({ratio:.1f}x)")
+ else:
+ print(f"✓ {self.current_folder}: "
+ f"{folder_data.get('unpacked_files', 0)} files, "
+ f"{format_size(decompressed)}")
+
+ def print_summary(self, output_dir: str):
+ """Print final summary."""
+ elapsed = self.get_elapsed()
+ speed = self.total_bytes / elapsed if elapsed > 0 else 0
+
+ print("\n" + "=" * 60)
+ print(" UNPACKING COMPLETE")
+ print("=" * 60)
+ print(f" Output directory: {output_dir}")
+ print(f" Total time: {format_time(elapsed)}")
+ print(f" Average speed: {format_size(speed)}/s")
+ print("-" * 60)
+ print(f" Folders: {self.total_folders}")
+ if self.copied_folders > 0:
+ print(f" Copied folders: {self.copied_folders}")
+ print(f" Files: {self.total_files}")
+ if self.copied_files > 0:
+ print(f" Copied files: {self.copied_files}")
+ print(f" Total size: {format_size(self.total_bytes)}")
+ print("-" * 60)
+ print(f" Downloaded: {format_size(self.downloaded_bytes)}")
+ if self.total_compressed_bytes > 0:
+ file_ratio = self.total_bytes / self.total_compressed_bytes if self.total_compressed_bytes > 0 else 1
+ print(f" File compression: {format_size(self.total_compressed_bytes)} -> {format_size(self.total_bytes)} ({file_ratio:.1f}x)")
+ print("=" * 60)
+
+ # Top 5 largest folders
+ if self.folder_stats:
+ print("\n Top folders by size:")
+ sorted_folders = sorted(
+ self.folder_stats.items(),
+ key=lambda x: x[1]['total_bytes'],
+ reverse=True
+ )[:5]
+ for folder, data in sorted_folders:
+ compressed = data.get('compressed_bytes', 0)
+ decompressed = data.get('total_bytes', 0)
+ if compressed > 0:
+ ratio = decompressed / compressed
+ print(f" {folder}: {data['unpacked_files']} files, {format_size(decompressed)} ({ratio:.1f}x)")
+ else:
+ print(f" {folder}: {data['unpacked_files']} files, {format_size(decompressed)}")
+
+
+async def download_and_unpack_async(url: str, output_dir: str, chunk_size: int = 65536, queue_maxsize: int = 100) -> None:
+ """
+ Download a packed file (with brotli-compressed files) and unpack directly to disk (async).
+ Uses separate tasks for downloading and unpacking with asyncio.Queue for buffering.
+
+ Individual files in the archive are brotli-compressed and will be decompressed
+ automatically by stream_unpack_async from packer_brotli.
+
+ Args:
+ url: URL of the packed .bin file
+ output_dir: Directory to unpack files into
+ chunk_size: Size of chunks to download
+ queue_maxsize: Max size of the buffer queue (0 for unlimited)
+ """
+ stats = UnpackStats()
+ queue: asyncio.Queue[Optional[bytes]] = asyncio.Queue(maxsize=queue_maxsize)
+
+ async def download_task():
+ """Download data and put into queue."""
+ try:
+ async with httpx.AsyncClient(timeout=httpx.Timeout(300.0), follow_redirects=True) as client:
+ async with client.stream('GET', url) as response:
+ response.raise_for_status()
+
+ content_length = response.headers.get('content-length')
+ if content_length:
+ print(f"📥 Downloading from {url}")
+ print(f" Remote file size: {format_size(int(content_length))}")
+ print(f" Files are brotli-compressed (will decompress)")
+ else:
+ print(f"📥 Downloading from {url}")
+ print(f" Files are brotli-compressed (will decompress)")
+ print()
+
+ async for chunk in response.aiter_bytes(chunk_size):
+ stats.add_downloaded(len(chunk))
+ await queue.put(chunk)
+ finally:
+ # Signal end of download
+ await queue.put(None)
+ stats.download_complete = True
+
+ async def queue_to_async_iter():
+ """Convert queue to async iterator for stream_unpack_async."""
+ while True:
+ chunk = await queue.get()
+ if chunk is None:
+ break
+ yield chunk
+
+ # Start download task
+ download_coro = asyncio.create_task(download_task())
+
+ try:
+ last_folder = None
+
+ async for folder_name, num_files, file_idx, filename, file_size, file_chunks, source_ref in stream_unpack_async(queue_to_async_iter()):
+ # Create folder path
+ folder_path = os.path.join(output_dir, folder_name)
+ os.makedirs(folder_path, exist_ok=True)
+ stats.unpacked_folders[folder_name] = folder_path
+
+ # Check if this is a copy folder marker (file_idx == -1)
+ if file_idx == -1:
+ # This is a copy folder, filename is actually source folder name
+ source_name = filename
+ source_path = stats.unpacked_folders.get(source_name)
+
+ # Print completion for previous folder if needed
+ if last_folder is not None and last_folder != folder_name:
+ stats.print_folder_complete()
+
+ stats.total_folders += 1
+ stats.copied_folders += 1
+
+ if source_path and os.path.exists(source_path):
+ # Copy files from source folder
+ copied_count = 0
+ copied_bytes = 0
+ for fname in os.listdir(source_path):
+ src_file = os.path.join(source_path, fname)
+ dst_file = os.path.join(folder_path, fname)
+ if os.path.isfile(src_file):
+ shutil.copy2(src_file, dst_file)
+ stats.unpacked_files[(folder_name, fname)] = dst_file
+ copied_count += 1
+ copied_bytes += os.path.getsize(src_file)
+
+ stats.total_files += copied_count
+ stats.total_bytes += copied_bytes
+ stats.folder_stats[folder_name] = {
+ 'total_files': copied_count,
+ 'unpacked_files': copied_count,
+ 'total_bytes': copied_bytes,
+ 'compressed_bytes': 0
+ }
+
+ stats.clear_line()
+ print(f"📋 {folder_name} <- {source_name}: {copied_count} files, {format_size(copied_bytes)}")
+ else:
+ stats.clear_line()
+ print(f"⚠️ {folder_name}: Source folder not found: {source_name}")
+
+ last_folder = folder_name
+ elif file_size == -2:
+ # File reference - copy from another file
+ src_folder, src_filename = source_ref
+ src_file_path = stats.unpacked_files.get((src_folder, src_filename))
+
+ # Check if we started a new folder
+ if folder_name != last_folder:
+ if last_folder is not None:
+ stats.print_folder_complete()
+ last_folder = folder_name
+ stats.start_folder(folder_name, num_files)
+
+ file_path = os.path.join(folder_path, filename)
+
+ if src_file_path and os.path.exists(src_file_path):
+ shutil.copy2(src_file_path, file_path)
+ stats.unpacked_files[(folder_name, filename)] = file_path
+ actual_size = os.path.getsize(file_path)
+ stats.file_copied(filename, actual_size)
+ stats.print_progress(filename, 0, actual_size, is_copy=True)
+ else:
+ stats.clear_line()
+ print(f"⚠️ {folder_name}/{filename}: Source file not found: {src_folder}/{src_filename}")
+ else:
+ # Normal file - file_chunks yields already-decompressed data
+ # Check if we started a new folder
+ if folder_name != last_folder:
+ # Print completion for previous folder
+ if last_folder is not None:
+ stats.print_folder_complete()
+
+ last_folder = folder_name
+ # Now we know the exact number of files from the binary header
+ stats.start_folder(folder_name, num_files)
+
+ # Write file - chunks are already decompressed by stream_unpack_async
+ file_path = os.path.join(folder_path, filename)
+ decompressed_size = 0
+ async with aiofiles.open(file_path, 'wb') as f:
+ async for chunk in file_chunks:
+ await f.write(chunk)
+ decompressed_size += len(chunk)
+
+ stats.unpacked_files[(folder_name, filename)] = file_path
+
+ # file_size is the compressed size from the archive
+ # decompressed_size is the actual file size after decompression
+ compressed_size = file_size
+
+ # Update stats
+ stats.file_unpacked(filename, compressed_size, decompressed_size)
+
+ # Print progress
+ stats.print_progress(filename, compressed_size, decompressed_size)
+
+ # Print completion for last folder
+ if last_folder is not None and last_folder in stats.folder_stats:
+ if stats.folder_stats[last_folder].get('unpacked_files', 0) < stats.folder_stats[last_folder].get('total_files', 0):
+ stats.print_folder_complete()
+ elif stats.unpacked_in_current_folder > 0:
+ stats.print_folder_complete()
+
+ # Wait for download to complete (should already be done)
+ await download_coro
+
+ except Exception as e:
+ download_coro.cancel()
+ raise
+
+ # Print final summary
+ stats.print_summary(output_dir)
+
+
+# ============== CLI ==============
+
+def main():
+ if len(sys.argv) < 3:
+ print("Usage: python downloader_brotli.py
")
+ print()
+ print("Downloads a packed file (packer_brotli format) and unpacks directly to disk.")
+ print("Files in the archive are brotli-compressed and will be decompressed automatically.")
+ print("Shows detailed progress and statistics during unpacking.")
+ print("Downloads and unpacks run in parallel using async queue buffering.")
+ print()
+ print("Example:")
+ print(" python downloader_brotli.py https://example.com/files.bin ./unpacked")
+ sys.exit(1)
+
+ url = sys.argv[1]
+ output_dir = sys.argv[2]
+
+ asyncio.run(download_and_unpack_async(url, output_dir))
+
+
+if __name__ == '__main__':
+ main()
diff --git a/utils/packer_brotli.py b/utils/packer_brotli.py
new file mode 100644
index 0000000..f1ec99e
--- /dev/null
+++ b/utils/packer_brotli.py
@@ -0,0 +1,1757 @@
+#!/usr/bin/env python3
+"""
+File packer that packs files from a folder and its subfolders into a single file.
+Uses ULEB128 encoding for string and file lengths.
+Supports folder and file deduplication.
+Uses Brotli compression (quality 11) with parallel processing for maximum compression.
+All strings (folder names, file names) are also Brotli compressed.
+
+Format:
+- For each folder:
+ - Folder type (1 byte): 0 = normal folder, 1 = copy of another folder
+ - Compressed folder name length (ULEB128)
+ - Compressed folder name bytes (Brotli)
+ - If type == 0 (normal folder):
+ - Number of files in folder (ULEB128)
+ - For each file:
+ - Compressed filename length (ULEB128)
+ - Compressed filename bytes (Brotli)
+ - File type (1 byte): 0 = content, 1 = reference to another file
+ - If file type == 0:
+ - File content length (ULEB128) - compressed size
+ - File content bytes (Brotli compressed)
+ - If file type == 1:
+ - Compressed source folder path length (ULEB128)
+ - Compressed source folder path bytes (Brotli)
+ - Compressed source filename length (ULEB128)
+ - Compressed source filename bytes (Brotli)
+ - If type == 1 (copy folder):
+ - Compressed source folder name length (ULEB128)
+ - Compressed source folder name bytes (Brotli)
+
+Supports both sync and async operations with parallel Brotli compression.
+Also provides PackedArchive class for reading files directly from archive.
+"""
+
+import os
+import sys
+import asyncio
+import hashlib
+import shutil
+import io
+import aiofiles
+import brotli
+from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed
+from typing import Iterator, Tuple, Generator, AsyncIterator, AsyncGenerator, Union, Dict, List, Set, Optional, BinaryIO
+from dataclasses import dataclass, field
+from contextlib import asynccontextmanager
+
+# Brotli compression settings
+BROTLI_QUALITY = 11 # Maximum compression
+BROTLI_LGWIN = 24 # Window size (max)
+BROTLI_MODE = brotli.MODE_GENERIC
+
+# Files to ignore during packing (macOS, Windows, etc. junk files)
+IGNORED_FILES = {
+ '.DS_Store',
+ '._.DS_Store',
+ 'Thumbs.db',
+ 'desktop.ini'
+}
+
+# File patterns to ignore (starting with)
+IGNORED_PREFIXES = ('._',)
+
+
+def should_ignore_file(filename: str) -> bool:
+ """Check if a file should be ignored during packing."""
+ if filename in IGNORED_FILES:
+ return True
+ for prefix in IGNORED_PREFIXES:
+ if filename.startswith(prefix):
+ return True
+ return False
+
+
+def is_already_brotli(filename: str) -> bool:
+ """Check if file is already brotli-compressed (.br extension)."""
+ return filename.lower().endswith('.br')
+
+
+def compress_brotli(data: bytes) -> bytes:
+ """Compress data using Brotli with maximum quality."""
+ return brotli.compress(data, quality=BROTLI_QUALITY, lgwin=BROTLI_LGWIN, mode=BROTLI_MODE)
+
+
+def decompress_brotli(data: bytes) -> bytes:
+ """Decompress Brotli-compressed data."""
+ return brotli.decompress(data)
+
+
+def compress_string(s: str) -> bytes:
+ """Compress a string (folder/file name) using Brotli."""
+ return compress_brotli(s.encode('utf-8'))
+
+
+def decompress_string(data: bytes) -> str:
+ """Decompress a Brotli-compressed string."""
+ return decompress_brotli(data).decode('utf-8')
+
+
+def compress_file_task(args: Tuple[str, str, str]) -> Tuple[str, str, str, bytes, int, int, bool]:
+ """
+ Compress a file using Brotli (or keep as-is for .br files). Used for parallel processing.
+ Args: (file_path, rel_path, filename)
+ Returns: (rel_path, filename, file_path, data, original_size, final_size, is_precompressed)
+
+ For .br files: returns data as-is (already brotli-compressed)
+ For other files: returns brotli-compressed data
+ """
+ file_path, rel_path, filename = args
+ with open(file_path, 'rb') as f:
+ content = f.read()
+ original_size = len(content)
+
+ # .br files are already brotli-compressed - store as-is
+ if is_already_brotli(filename):
+ return (rel_path, filename, file_path, content, original_size, original_size, True)
+
+ # Compress other files
+ compressed = compress_brotli(content)
+ compressed_size = len(compressed)
+ return (rel_path, filename, file_path, compressed, original_size, compressed_size, False)
+
+
+def encode_uleb128(value: int) -> bytes:
+ """Encode an unsigned integer as ULEB128 bytes."""
+ result = bytearray()
+ while True:
+ byte = value & 0x7F
+ value >>= 7
+ if value != 0:
+ byte |= 0x80
+ result.append(byte)
+ if value == 0:
+ break
+ return bytes(result)
+
+
+def decode_uleb128(data: bytes, offset: int = 0) -> tuple[int, int]:
+ """Decode ULEB128 bytes to an unsigned integer. Returns (value, bytes_read)."""
+ result = 0
+ shift = 0
+ bytes_read = 0
+ while True:
+ byte = data[offset + bytes_read]
+ bytes_read += 1
+ result |= (byte & 0x7F) << shift
+ if (byte & 0x80) == 0:
+ break
+ shift += 7
+ return result, bytes_read
+
+
+def uleb128_size(value: int) -> int:
+ """Calculate the size of a ULEB128 encoded value."""
+ size = 0
+ while True:
+ value >>= 7
+ size += 1
+ if value == 0:
+ break
+ return size
+
+
+# ============== FOLDER/FILE SIGNATURE ==============
+
+@dataclass
+class FolderSignature:
+ """Signature of a folder for deduplication."""
+ path: str
+ file_count: int
+ files: Dict[str, str] # filename -> content hash
+ total_hash: str # combined hash of all files
+
+ @staticmethod
+ def compute_file_hash(file_path: str) -> str:
+ """Compute MD5 hash of a file."""
+ hasher = hashlib.md5()
+ with open(file_path, 'rb') as f:
+ for chunk in iter(lambda: f.read(65536), b''):
+ hasher.update(chunk)
+ return hasher.hexdigest()
+
+ @classmethod
+ def from_folder(cls, folder_path: str, rel_path: str) -> 'FolderSignature':
+ """Create signature from a folder."""
+ files = {}
+ file_list = sorted(os.listdir(folder_path))
+
+ # Only include regular files, not subdirectories, and skip ignored files
+ for filename in file_list:
+ if should_ignore_file(filename):
+ continue
+ file_path = os.path.join(folder_path, filename)
+ if os.path.isfile(file_path):
+ files[filename] = cls.compute_file_hash(file_path)
+
+ # Compute total hash from sorted file hashes
+ total_hasher = hashlib.md5()
+ for filename in sorted(files.keys()):
+ total_hasher.update(filename.encode('utf-8'))
+ total_hasher.update(files[filename].encode('utf-8'))
+
+ return cls(
+ path=rel_path,
+ file_count=len(files),
+ files=files,
+ total_hash=total_hasher.hexdigest()
+ )
+
+ def matches(self, other: 'FolderSignature') -> bool:
+ """Check if this folder has identical content to another."""
+ if self.file_count != other.file_count:
+ return False
+ if self.total_hash != other.total_hash:
+ return False
+ return self.files == other.files
+
+
+@dataclass
+class FileInfo:
+ """Information about a file for deduplication."""
+ folder_path: str
+ filename: str
+ full_path: str
+ size: int
+ hash: str
+
+
+def find_duplicates(folder_path: str, parent_dir: str) -> Tuple[Dict[str, str], Dict[str, Tuple[str, str]]]:
+ """
+ Scan folder structure and find duplicates.
+ Returns:
+ - folder_duplicates: dict mapping duplicate folder path -> source folder path
+ - file_duplicates: dict mapping (folder_path, filename) -> (source_folder, source_filename)
+ """
+ folder_path = folder_path.rstrip('/\\')
+
+ # First pass: collect all folder and file signatures
+ folder_signatures: Dict[str, FolderSignature] = {}
+ all_files: Dict[str, List[FileInfo]] = {} # hash -> list of files with that hash
+
+ for root, dirs, files in os.walk(folder_path):
+ if not files:
+ continue
+
+ rel_path = os.path.relpath(root, parent_dir)
+ sig = FolderSignature.from_folder(root, rel_path)
+ folder_signatures[rel_path] = sig
+
+ # Collect individual file info (skip ignored files)
+ for filename in files:
+ if should_ignore_file(filename):
+ continue
+ file_path = os.path.join(root, filename)
+ if os.path.isfile(file_path):
+ file_size = os.path.getsize(file_path)
+ file_hash = sig.files.get(filename) or FolderSignature.compute_file_hash(file_path)
+
+ file_info = FileInfo(
+ folder_path=rel_path,
+ filename=filename,
+ full_path=file_path,
+ size=file_size,
+ hash=file_hash
+ )
+
+ if file_hash not in all_files:
+ all_files[file_hash] = []
+ all_files[file_hash].append(file_info)
+
+ # Find folder duplicates
+ folder_duplicates: Dict[str, str] = {}
+ seen_folder_hashes: Dict[str, str] = {}
+
+ for rel_path in sorted(folder_signatures.keys()):
+ sig = folder_signatures[rel_path]
+
+ if sig.total_hash in seen_folder_hashes:
+ source_path = seen_folder_hashes[sig.total_hash]
+ source_sig = folder_signatures[source_path]
+
+ if sig.matches(source_sig):
+ folder_duplicates[rel_path] = source_path
+ print(f" Duplicate folder: {rel_path} -> {source_path}")
+ else:
+ seen_folder_hashes[sig.total_hash] = rel_path
+
+ # Find file duplicates (only for files not in duplicate folders)
+ file_duplicates: Dict[Tuple[str, str], Tuple[str, str]] = {}
+
+ for file_hash, file_list in all_files.items():
+ if len(file_list) <= 1:
+ continue
+
+ # Sort by path to ensure consistent ordering
+ file_list.sort(key=lambda f: (f.folder_path, f.filename))
+
+ # First file is the source
+ source = file_list[0]
+
+ # Skip if source is in a duplicate folder
+ if source.folder_path in folder_duplicates:
+ continue
+
+ for dup in file_list[1:]:
+ # Skip if this file is in a duplicate folder (will be copied with folder)
+ if dup.folder_path in folder_duplicates:
+ continue
+
+ # Check if reference would save space
+ # Reference format: 1 byte type + source_folder_len + source_folder + source_filename_len + source_filename
+ ref_size = (1 +
+ uleb128_size(len(source.folder_path.encode('utf-8'))) +
+ len(source.folder_path.encode('utf-8')) +
+ uleb128_size(len(source.filename.encode('utf-8'))) +
+ len(source.filename.encode('utf-8')))
+
+ # Content format: 1 byte type + content_len + content
+ content_size = 1 + uleb128_size(dup.size) + dup.size
+
+ if ref_size < content_size:
+ file_duplicates[(dup.folder_path, dup.filename)] = (source.folder_path, source.filename)
+ print(f" Duplicate file: {dup.folder_path}/{dup.filename} -> {source.folder_path}/{source.filename} (saves {content_size - ref_size} bytes)")
+
+ return folder_duplicates, file_duplicates
+
+
+# ============== SYNC FUNCTIONS ==============
+
+# Type constants
+FOLDER_TYPE_NORMAL = 0
+FOLDER_TYPE_COPY = 1
+FILE_TYPE_CONTENT = 0
+FILE_TYPE_REFERENCE = 1
+
+
+def pack_folder(folder_path: str, output_file: str, deduplicate: bool = True, max_workers: int = None) -> None:
+ """
+ Pack all files from folder and subfolders into a single file (sync).
+ Uses parallel Brotli compression for maximum speed with quality 11.
+
+ Args:
+ folder_path: Path to folder to pack
+ output_file: Output file path
+ deduplicate: If True, detect and deduplicate identical folders and files
+ max_workers: Maximum number of parallel compression workers (default: CPU count)
+ """
+ folder_path = folder_path.rstrip('/\\')
+ parent_dir = os.path.dirname(folder_path) or '.'
+
+ if max_workers is None:
+ max_workers = os.cpu_count() or 4
+
+ # Find duplicates if deduplication is enabled
+ folder_duplicates: Dict[str, str] = {}
+ file_duplicates: Dict[Tuple[str, str], Tuple[str, str]] = {}
+
+ if deduplicate:
+ print("Scanning for duplicates...")
+ folder_duplicates, file_duplicates = find_duplicates(folder_path, parent_dir)
+ if folder_duplicates or file_duplicates:
+ print(f"Found {len(folder_duplicates)} duplicate folder(s), {len(file_duplicates)} duplicate file(s)")
+ else:
+ print("No duplicates found")
+ print()
+
+ folder_bytes_saved = 0
+ file_bytes_saved = 0
+ total_original_size = 0
+ total_compressed_size = 0
+
+ # First pass: collect all files that need compression
+ print("Collecting files for compression...")
+ files_to_compress: List[Tuple[str, str, str]] = [] # (file_path, rel_path, filename)
+ folder_structure: List[Tuple[str, List[str], bool, str]] = [] # (rel_path, files, is_duplicate, source_path)
+
+ for root, dirs, files in os.walk(folder_path):
+ # Filter out ignored files
+ files = [f for f in files if not should_ignore_file(f)]
+ if not files:
+ continue
+
+ rel_path = os.path.relpath(root, parent_dir)
+
+ if rel_path in folder_duplicates:
+ source_path = folder_duplicates[rel_path]
+ folder_structure.append((rel_path, list(files), True, source_path))
+ for filename in files:
+ file_path = os.path.join(root, filename)
+ folder_bytes_saved += os.path.getsize(file_path)
+ else:
+ folder_structure.append((rel_path, sorted(files), False, None))
+ for filename in sorted(files):
+ file_key = (rel_path, filename)
+ if file_key not in file_duplicates:
+ file_path = os.path.join(root, filename)
+ files_to_compress.append((file_path, rel_path, filename))
+ else:
+ file_path = os.path.join(root, filename)
+ file_bytes_saved += os.path.getsize(file_path)
+
+ print(f"Compressing {len(files_to_compress)} files using {max_workers} workers (Brotli quality {BROTLI_QUALITY})...")
+
+ # Parallel compression of all files
+ compressed_files: Dict[Tuple[str, str], bytes] = {} # (rel_path, filename) -> compressed_data
+ precompressed_files: Set[Tuple[str, str]] = set() # Track which files were already .br
+
+ with ProcessPoolExecutor(max_workers=max_workers) as executor:
+ futures = {executor.submit(compress_file_task, args): args for args in files_to_compress}
+ completed = 0
+
+ for future in as_completed(futures):
+ rel_path, filename, file_path, data, original_size, final_size, is_precompressed = future.result()
+ compressed_files[(rel_path, filename)] = data
+ if is_precompressed:
+ precompressed_files.add((rel_path, filename))
+ print(f" [{completed + 1}/{len(files_to_compress)}] Stored as-is (.br): {rel_path}/{filename} ({original_size} bytes)")
+ else:
+ total_original_size += original_size
+ total_compressed_size += final_size
+ ratio = (final_size / original_size * 100) if original_size > 0 else 0
+ print(f" [{completed + 1}/{len(files_to_compress)}] Compressed: {rel_path}/{filename} ({original_size} -> {final_size} bytes, {ratio:.1f}%)")
+ completed += 1
+
+ print(f"\nWriting packed file...")
+
+ # Write the packed file
+ with open(output_file, 'wb') as out:
+ for rel_path, files, is_duplicate, source_path in folder_structure:
+ folder_name_bytes = rel_path.encode('utf-8')
+
+ if is_duplicate:
+ source_path_bytes = source_path.encode('utf-8')
+
+ # Write copy folder entry
+ out.write(bytes([FOLDER_TYPE_COPY]))
+ out.write(encode_uleb128(len(folder_name_bytes)))
+ out.write(folder_name_bytes)
+ out.write(encode_uleb128(len(source_path_bytes)))
+ out.write(source_path_bytes)
+
+ print(f" Copy folder: {rel_path} -> {source_path}")
+ else:
+ # Write normal folder entry
+ out.write(bytes([FOLDER_TYPE_NORMAL]))
+ out.write(encode_uleb128(len(folder_name_bytes)))
+ out.write(folder_name_bytes)
+ out.write(encode_uleb128(len(files)))
+
+ for filename in files:
+ filename_bytes = filename.encode('utf-8')
+
+ out.write(encode_uleb128(len(filename_bytes)))
+ out.write(filename_bytes)
+
+ # Check if this file is a duplicate
+ file_key = (rel_path, filename)
+ if file_key in file_duplicates:
+ source_folder, source_filename = file_duplicates[file_key]
+ source_folder_bytes = source_folder.encode('utf-8')
+ source_filename_bytes = source_filename.encode('utf-8')
+
+ # Write file reference
+ out.write(bytes([FILE_TYPE_REFERENCE]))
+ out.write(encode_uleb128(len(source_folder_bytes)))
+ out.write(source_folder_bytes)
+ out.write(encode_uleb128(len(source_filename_bytes)))
+ out.write(source_filename_bytes)
+
+ print(f" Ref: {rel_path}/{filename} -> {source_folder}/{source_filename}")
+ else:
+ # Write compressed file content
+ compressed_content = compressed_files[(rel_path, filename)]
+
+ out.write(bytes([FILE_TYPE_CONTENT]))
+ out.write(encode_uleb128(len(compressed_content)))
+ out.write(compressed_content)
+
+ total_size = os.path.getsize(output_file)
+ print(f"\nPacked to {output_file} ({total_size} bytes)")
+ if total_original_size > 0:
+ overall_ratio = total_compressed_size / total_original_size * 100
+ print(f"Compression: {total_original_size} -> {total_compressed_size} bytes ({overall_ratio:.1f}%)")
+ if folder_bytes_saved > 0 or file_bytes_saved > 0:
+ print(f"Deduplication saved: {folder_bytes_saved + file_bytes_saved} bytes (folders: {folder_bytes_saved}, files: {file_bytes_saved})")
+
+
+def unpack_file(input_file: str, output_dir: str) -> None:
+ """Unpack a packed file back to folder structure (sync). Decompresses Brotli-compressed content."""
+ with open(input_file, 'rb') as f:
+ data = f.read()
+
+ # Track unpacked folders and files for copy references
+ unpacked_folders: Dict[str, str] = {} # rel_path -> absolute path
+ unpacked_files: Dict[Tuple[str, str], str] = {} # (folder, filename) -> absolute path
+
+ offset = 0
+ while offset < len(data):
+ # Read folder type
+ folder_type = data[offset]
+ offset += 1
+
+ # Read folder name
+ folder_name_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ folder_name = data[offset:offset + folder_name_len].decode('utf-8')
+ offset += folder_name_len
+
+ folder_path = os.path.join(output_dir, folder_name)
+ os.makedirs(folder_path, exist_ok=True)
+ unpacked_folders[folder_name] = folder_path
+
+ if folder_type == FOLDER_TYPE_COPY:
+ # Read source folder name
+ source_name_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ source_name = data[offset:offset + source_name_len].decode('utf-8')
+ offset += source_name_len
+
+ # Copy files from source folder
+ source_path = unpacked_folders.get(source_name)
+ if source_path and os.path.exists(source_path):
+ for filename in os.listdir(source_path):
+ src_file = os.path.join(source_path, filename)
+ dst_file = os.path.join(folder_path, filename)
+ if os.path.isfile(src_file):
+ shutil.copy2(src_file, dst_file)
+ unpacked_files[(folder_name, filename)] = dst_file
+ print(f"Copied folder: {folder_name} <- {source_name}")
+ else:
+ print(f"Warning: Source folder not found: {source_name}")
+ else:
+ # Normal folder - read files
+ num_files, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+
+ print(f"Folder: {folder_name} ({num_files} files)")
+
+ for _ in range(num_files):
+ filename_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ filename = data[offset:offset + filename_len].decode('utf-8')
+ offset += filename_len
+
+ file_path = os.path.join(folder_path, filename)
+
+ # Read file type
+ file_type = data[offset]
+ offset += 1
+
+ if file_type == FILE_TYPE_REFERENCE:
+ # Read source reference
+ src_folder_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ src_folder = data[offset:offset + src_folder_len].decode('utf-8')
+ offset += src_folder_len
+
+ src_filename_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ src_filename = data[offset:offset + src_filename_len].decode('utf-8')
+ offset += src_filename_len
+
+ # Copy from source file
+ src_file_path = unpacked_files.get((src_folder, src_filename))
+ if src_file_path and os.path.exists(src_file_path):
+ shutil.copy2(src_file_path, file_path)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f" Copied: {filename} <- {src_folder}/{src_filename}")
+ else:
+ print(f" Warning: Source file not found: {src_folder}/{src_filename}")
+ else:
+ # Read content
+ content_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ content = data[offset:offset + content_len]
+ offset += content_len
+
+ # .br files are stored as-is (not brotli-compressed), write directly
+ if is_already_brotli(filename):
+ with open(file_path, 'wb') as f:
+ f.write(content)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f" Unpacked: {filename} ({content_len} bytes, stored as-is)")
+ else:
+ # Decompress with Brotli
+ decompressed = decompress_brotli(content)
+ with open(file_path, 'wb') as f:
+ f.write(decompressed)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f" Unpacked: {filename} ({content_len} -> {len(decompressed)} bytes)")
+
+ print(f"\nUnpacked to {output_dir}")
+
+
+def stream_unpack(chunks: Iterator[bytes]) -> Generator[Tuple[str, str, int, Generator[bytes, None, None], Tuple[str, str]], None, None]:
+ """
+ Stream unpack a packed file from an iterable of byte chunks (sync).
+ Decompresses Brotli-compressed content.
+
+ Yields tuples of: (folder_name, file_name, decompressed_size, file_chunks_generator, source_ref)
+ - For normal files: (folder_name, filename, size, chunks_gen, None)
+ - For file references: (folder_name, filename, -2, None, (src_folder, src_filename))
+ - For folder copies: (folder_name, source_folder, -1, None, None)
+ """
+ buffer = bytearray()
+ chunk_iter = iter(chunks)
+
+ def read_bytes(n: int) -> bytes:
+ nonlocal buffer
+ while len(buffer) < n:
+ try:
+ chunk = next(chunk_iter)
+ buffer.extend(chunk)
+ except StopIteration:
+ if len(buffer) < n:
+ raise EOFError(f"Expected {n} bytes, got {len(buffer)}")
+ result = bytes(buffer[:n])
+ del buffer[:n]
+ return result
+
+ def read_uleb128() -> int:
+ result = 0
+ shift = 0
+ while True:
+ byte_data = read_bytes(1)
+ byte = byte_data[0]
+ result |= (byte & 0x7F) << shift
+ if (byte & 0x80) == 0:
+ break
+ shift += 7
+ return result
+
+ def file_chunk_generator_decompressed(compressed_size: int) -> Generator[bytes, None, None]:
+ """Read compressed data, decompress, and yield as single chunk."""
+ compressed_data = read_bytes(compressed_size)
+ decompressed = decompress_brotli(compressed_data)
+ yield decompressed
+
+ try:
+ while True:
+ try:
+ folder_type = read_bytes(1)[0]
+ except EOFError:
+ break
+
+ folder_name_len = read_uleb128()
+ folder_name = read_bytes(folder_name_len).decode('utf-8')
+
+ if folder_type == FOLDER_TYPE_COPY:
+ source_name_len = read_uleb128()
+ source_name = read_bytes(source_name_len).decode('utf-8')
+ yield (folder_name, source_name, -1, None, None)
+ else:
+ num_files = read_uleb128()
+
+ for _ in range(num_files):
+ filename_len = read_uleb128()
+ filename = read_bytes(filename_len).decode('utf-8')
+
+ file_type = read_bytes(1)[0]
+
+ if file_type == FILE_TYPE_REFERENCE:
+ src_folder_len = read_uleb128()
+ src_folder = read_bytes(src_folder_len).decode('utf-8')
+ src_filename_len = read_uleb128()
+ src_filename = read_bytes(src_filename_len).decode('utf-8')
+ yield (folder_name, filename, -2, None, (src_folder, src_filename))
+ else:
+ compressed_len = read_uleb128()
+ # We can't know decompressed size without decompressing,
+ # so we pass compressed_len and decompress in the generator
+ yield (folder_name, filename, compressed_len, file_chunk_generator_decompressed(compressed_len), None)
+ except EOFError:
+ pass
+
+
+def stream_unpack_to_disk(chunks: Iterator[bytes], output_dir: str) -> None:
+ """Stream unpack directly to disk (sync)."""
+ unpacked_folders: Dict[str, str] = {}
+ unpacked_files: Dict[Tuple[str, str], str] = {}
+
+ for folder_name, filename, file_size, file_chunks, source_ref in stream_unpack(chunks):
+ folder_path = os.path.join(output_dir, folder_name)
+ os.makedirs(folder_path, exist_ok=True)
+ unpacked_folders[folder_name] = folder_path
+
+ if file_size == -1:
+ # Copy folder
+ source_name = filename
+ source_path = unpacked_folders.get(source_name)
+ if source_path and os.path.exists(source_path):
+ for fname in os.listdir(source_path):
+ src_file = os.path.join(source_path, fname)
+ dst_file = os.path.join(folder_path, fname)
+ if os.path.isfile(src_file):
+ shutil.copy2(src_file, dst_file)
+ unpacked_files[(folder_name, fname)] = dst_file
+ print(f"Copied folder: {folder_name} <- {source_name}")
+ elif file_size == -2:
+ # File reference
+ src_folder, src_filename = source_ref
+ src_file_path = unpacked_files.get((src_folder, src_filename))
+ file_path = os.path.join(folder_path, filename)
+ if src_file_path and os.path.exists(src_file_path):
+ shutil.copy2(src_file_path, file_path)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f"Copied: {folder_name}/{filename} <- {src_folder}/{src_filename}")
+ else:
+ file_path = os.path.join(folder_path, filename)
+ with open(file_path, 'wb') as f:
+ for chunk in file_chunks:
+ f.write(chunk)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f"Unpacked: {folder_name}/{filename} ({file_size} bytes)")
+
+ print(f"\nStream unpacked to {output_dir}")
+
+
+# ============== ASYNC FUNCTIONS ==============
+
+async def pack_folder_async(folder_path: str, output_file: str, deduplicate: bool = True, max_workers: int = None) -> None:
+ """
+ Pack all files from folder and subfolders into a single file (async).
+ Uses parallel Brotli compression for maximum speed with quality 11.
+
+ Args:
+ folder_path: Path to folder to pack
+ output_file: Output file path
+ deduplicate: If True, detect and deduplicate identical folders and files
+ max_workers: Maximum number of parallel compression workers (default: CPU count)
+ """
+ folder_path = folder_path.rstrip('/\\')
+ parent_dir = os.path.dirname(folder_path) or '.'
+
+ if max_workers is None:
+ max_workers = os.cpu_count() or 4
+
+ # Find duplicates if deduplication is enabled
+ folder_duplicates: Dict[str, str] = {}
+ file_duplicates: Dict[Tuple[str, str], Tuple[str, str]] = {}
+
+ if deduplicate:
+ print("Scanning for duplicates...")
+ folder_duplicates, file_duplicates = await asyncio.get_event_loop().run_in_executor(
+ None, find_duplicates, folder_path, parent_dir
+ )
+ if folder_duplicates or file_duplicates:
+ print(f"Found {len(folder_duplicates)} duplicate folder(s), {len(file_duplicates)} duplicate file(s)")
+ else:
+ print("No duplicates found")
+ print()
+
+ folder_bytes_saved = 0
+ file_bytes_saved = 0
+ total_original_size = 0
+ total_compressed_size = 0
+
+ # First pass: collect all files that need compression
+ print("Collecting files for compression...")
+ files_to_compress: List[Tuple[str, str, str]] = [] # (file_path, rel_path, filename)
+ folder_structure: List[Tuple[str, List[str], bool, str]] = [] # (rel_path, files, is_duplicate, source_path)
+
+ for root, dirs, files in os.walk(folder_path):
+ # Filter out ignored files
+ files = [f for f in files if not should_ignore_file(f)]
+ if not files:
+ continue
+
+ rel_path = os.path.relpath(root, parent_dir)
+
+ if rel_path in folder_duplicates:
+ source_path = folder_duplicates[rel_path]
+ folder_structure.append((rel_path, list(files), True, source_path))
+ for filename in files:
+ file_path = os.path.join(root, filename)
+ folder_bytes_saved += os.path.getsize(file_path)
+ else:
+ folder_structure.append((rel_path, sorted(files), False, None))
+ for filename in sorted(files):
+ file_key = (rel_path, filename)
+ if file_key not in file_duplicates:
+ file_path = os.path.join(root, filename)
+ files_to_compress.append((file_path, rel_path, filename))
+ else:
+ file_path = os.path.join(root, filename)
+ file_bytes_saved += os.path.getsize(file_path)
+
+ print(f"Compressing {len(files_to_compress)} files using {max_workers} workers (Brotli quality {BROTLI_QUALITY})...")
+
+ # Parallel compression using ProcessPoolExecutor with asyncio
+ compressed_files: Dict[Tuple[str, str], bytes] = {} # (rel_path, filename) -> compressed_data
+ precompressed_files: Set[Tuple[str, str]] = set() # Track which files were already .br
+ loop = asyncio.get_event_loop()
+
+ with ProcessPoolExecutor(max_workers=max_workers) as executor:
+ # Submit all tasks
+ future_to_args = {
+ loop.run_in_executor(executor, compress_file_task, args): args
+ for args in files_to_compress
+ }
+
+ completed = 0
+ for coro in asyncio.as_completed(future_to_args.keys()):
+ result = await coro
+ rel_path, filename, file_path, data, original_size, final_size, is_precompressed = result
+ compressed_files[(rel_path, filename)] = data
+ if is_precompressed:
+ precompressed_files.add((rel_path, filename))
+ print(f" [{completed + 1}/{len(files_to_compress)}] Stored as-is (.br): {rel_path}/{filename} ({original_size} bytes)")
+ else:
+ total_original_size += original_size
+ total_compressed_size += final_size
+ ratio = (final_size / original_size * 100) if original_size > 0 else 0
+ print(f" [{completed + 1}/{len(files_to_compress)}] Compressed: {rel_path}/{filename} ({original_size} -> {final_size} bytes, {ratio:.1f}%)")
+ completed += 1
+
+ print(f"\nWriting packed file...")
+
+ # Write the packed file
+ async with aiofiles.open(output_file, 'wb') as out:
+ for rel_path, files, is_duplicate, source_path in folder_structure:
+ folder_name_bytes = rel_path.encode('utf-8')
+
+ if is_duplicate:
+ source_path_bytes = source_path.encode('utf-8')
+
+ await out.write(bytes([FOLDER_TYPE_COPY]))
+ await out.write(encode_uleb128(len(folder_name_bytes)))
+ await out.write(folder_name_bytes)
+ await out.write(encode_uleb128(len(source_path_bytes)))
+ await out.write(source_path_bytes)
+
+ print(f" Copy folder: {rel_path} -> {source_path}")
+ else:
+ await out.write(bytes([FOLDER_TYPE_NORMAL]))
+ await out.write(encode_uleb128(len(folder_name_bytes)))
+ await out.write(folder_name_bytes)
+ await out.write(encode_uleb128(len(files)))
+
+ for filename in files:
+ filename_bytes = filename.encode('utf-8')
+
+ await out.write(encode_uleb128(len(filename_bytes)))
+ await out.write(filename_bytes)
+
+ file_key = (rel_path, filename)
+ if file_key in file_duplicates:
+ source_folder, source_filename = file_duplicates[file_key]
+ source_folder_bytes = source_folder.encode('utf-8')
+ source_filename_bytes = source_filename.encode('utf-8')
+
+ await out.write(bytes([FILE_TYPE_REFERENCE]))
+ await out.write(encode_uleb128(len(source_folder_bytes)))
+ await out.write(source_folder_bytes)
+ await out.write(encode_uleb128(len(source_filename_bytes)))
+ await out.write(source_filename_bytes)
+
+ print(f" Ref: {rel_path}/{filename} -> {source_folder}/{source_filename}")
+ else:
+ # Write compressed file content
+ compressed_content = compressed_files[(rel_path, filename)]
+
+ await out.write(bytes([FILE_TYPE_CONTENT]))
+ await out.write(encode_uleb128(len(compressed_content)))
+ await out.write(compressed_content)
+
+ total_size = os.path.getsize(output_file)
+ print(f"\nPacked to {output_file} ({total_size} bytes)")
+ if total_original_size > 0:
+ overall_ratio = total_compressed_size / total_original_size * 100
+ print(f"Compression: {total_original_size} -> {total_compressed_size} bytes ({overall_ratio:.1f}%)")
+ if folder_bytes_saved > 0 or file_bytes_saved > 0:
+ print(f"Deduplication saved: {folder_bytes_saved + file_bytes_saved} bytes (folders: {folder_bytes_saved}, files: {file_bytes_saved})")
+
+
+async def unpack_file_async(input_file: str, output_dir: str) -> None:
+ """Unpack a packed file back to folder structure (async). Decompresses Brotli-compressed content."""
+ async with aiofiles.open(input_file, 'rb') as f:
+ data = await f.read()
+
+ unpacked_folders: Dict[str, str] = {}
+ unpacked_files: Dict[Tuple[str, str], str] = {}
+
+ offset = 0
+ while offset < len(data):
+ folder_type = data[offset]
+ offset += 1
+
+ folder_name_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ folder_name = data[offset:offset + folder_name_len].decode('utf-8')
+ offset += folder_name_len
+
+ folder_path = os.path.join(output_dir, folder_name)
+ os.makedirs(folder_path, exist_ok=True)
+ unpacked_folders[folder_name] = folder_path
+
+ if folder_type == FOLDER_TYPE_COPY:
+ source_name_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ source_name = data[offset:offset + source_name_len].decode('utf-8')
+ offset += source_name_len
+
+ source_path = unpacked_folders.get(source_name)
+ if source_path and os.path.exists(source_path):
+ for filename in os.listdir(source_path):
+ src_file = os.path.join(source_path, filename)
+ dst_file = os.path.join(folder_path, filename)
+ if os.path.isfile(src_file):
+ shutil.copy2(src_file, dst_file)
+ unpacked_files[(folder_name, filename)] = dst_file
+ print(f"Copied folder: {folder_name} <- {source_name}")
+ else:
+ num_files, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+
+ print(f"Folder: {folder_name} ({num_files} files)")
+
+ for _ in range(num_files):
+ filename_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ filename = data[offset:offset + filename_len].decode('utf-8')
+ offset += filename_len
+
+ file_path = os.path.join(folder_path, filename)
+
+ file_type = data[offset]
+ offset += 1
+
+ if file_type == FILE_TYPE_REFERENCE:
+ src_folder_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ src_folder = data[offset:offset + src_folder_len].decode('utf-8')
+ offset += src_folder_len
+
+ src_filename_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ src_filename = data[offset:offset + src_filename_len].decode('utf-8')
+ offset += src_filename_len
+
+ src_file_path = unpacked_files.get((src_folder, src_filename))
+ if src_file_path and os.path.exists(src_file_path):
+ shutil.copy2(src_file_path, file_path)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f" Copied: {filename} <- {src_folder}/{src_filename}")
+ else:
+ # Read content
+ content_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ content = data[offset:offset + content_len]
+ offset += content_len
+
+ # .br files are stored as-is (not brotli-compressed), write directly
+ if is_already_brotli(filename):
+ async with aiofiles.open(file_path, 'wb') as f:
+ await f.write(content)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f" Unpacked: {filename} ({content_len} bytes, stored as-is)")
+ else:
+ # Decompress with Brotli
+ decompressed = decompress_brotli(content)
+ async with aiofiles.open(file_path, 'wb') as f:
+ await f.write(decompressed)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f" Unpacked: {filename} ({content_len} -> {len(decompressed)} bytes)")
+
+ print(f"\nUnpacked to {output_dir}")
+
+
+async def stream_unpack_async(
+ chunks: AsyncIterator[bytes]
+) -> AsyncGenerator[Tuple[str, int, int, str, int, AsyncGenerator[bytes, None], Tuple[str, str]], None]:
+ """
+ Stream unpack a packed file from an async iterable of byte chunks.
+ Decompresses Brotli-compressed content.
+
+ Yields tuples of:
+ - For normal files: (folder_name, num_files, file_idx, filename, decompressed_size, chunks_gen, None)
+ - For file references: (folder_name, num_files, file_idx, filename, -2, None, (src_folder, src_filename))
+ - For folder copies: (folder_name, 0, -1, source_folder, -1, None, None)
+ """
+ buffer = bytearray()
+ chunk_aiter = chunks.__aiter__()
+
+ async def read_bytes(n: int) -> bytes:
+ nonlocal buffer
+ while len(buffer) < n:
+ try:
+ chunk = await chunk_aiter.__anext__()
+ buffer.extend(chunk)
+ except StopAsyncIteration:
+ if len(buffer) < n:
+ raise EOFError(f"Expected {n} bytes, got {len(buffer)}")
+ result = bytes(buffer[:n])
+ del buffer[:n]
+ return result
+
+ async def read_uleb128() -> int:
+ result = 0
+ shift = 0
+ while True:
+ byte_data = await read_bytes(1)
+ byte = byte_data[0]
+ result |= (byte & 0x7F) << shift
+ if (byte & 0x80) == 0:
+ break
+ shift += 7
+ return result
+
+ async def file_chunk_generator_decompressed(compressed_size: int) -> AsyncGenerator[bytes, None]:
+ """Read compressed data, decompress with Brotli, and yield as single chunk."""
+ compressed_data = await read_bytes(compressed_size)
+ decompressed = decompress_brotli(compressed_data)
+ yield decompressed
+
+ try:
+ while True:
+ try:
+ folder_type = (await read_bytes(1))[0]
+ except EOFError:
+ break
+
+ folder_name_len = await read_uleb128()
+ folder_name_bytes = await read_bytes(folder_name_len)
+ folder_name = folder_name_bytes.decode('utf-8')
+
+ if folder_type == FOLDER_TYPE_COPY:
+ source_name_len = await read_uleb128()
+ source_name_bytes = await read_bytes(source_name_len)
+ source_name = source_name_bytes.decode('utf-8')
+ yield (folder_name, 0, -1, source_name, -1, None, None)
+ else:
+ num_files = await read_uleb128()
+
+ for file_idx in range(num_files):
+ filename_len = await read_uleb128()
+ filename_bytes = await read_bytes(filename_len)
+ filename = filename_bytes.decode('utf-8')
+
+ file_type = (await read_bytes(1))[0]
+
+ if file_type == FILE_TYPE_REFERENCE:
+ src_folder_len = await read_uleb128()
+ src_folder_bytes = await read_bytes(src_folder_len)
+ src_folder = src_folder_bytes.decode('utf-8')
+ src_filename_len = await read_uleb128()
+ src_filename_bytes = await read_bytes(src_filename_len)
+ src_filename = src_filename_bytes.decode('utf-8')
+ yield (folder_name, num_files, file_idx, filename, -2, None, (src_folder, src_filename))
+ else:
+ compressed_len = await read_uleb128()
+ # We compress and decompress in the generator
+ yield (folder_name, num_files, file_idx, filename, compressed_len, file_chunk_generator_decompressed(compressed_len), None)
+ except EOFError:
+ pass
+
+
+async def stream_unpack_to_disk_async(chunks: AsyncIterator[bytes], output_dir: str) -> None:
+ """Stream unpack directly to disk (async)."""
+ unpacked_folders: Dict[str, str] = {}
+ unpacked_files: Dict[Tuple[str, str], str] = {}
+
+ async for folder_name, num_files, file_idx, filename, file_size, file_chunks, source_ref in stream_unpack_async(chunks):
+ folder_path = os.path.join(output_dir, folder_name)
+ os.makedirs(folder_path, exist_ok=True)
+ unpacked_folders[folder_name] = folder_path
+
+ if file_idx == -1:
+ # Copy folder
+ source_name = filename
+ source_path = unpacked_folders.get(source_name)
+ if source_path and os.path.exists(source_path):
+ for fname in os.listdir(source_path):
+ src_file = os.path.join(source_path, fname)
+ dst_file = os.path.join(folder_path, fname)
+ if os.path.isfile(src_file):
+ shutil.copy2(src_file, dst_file)
+ unpacked_files[(folder_name, fname)] = dst_file
+ print(f"Copied folder: {folder_name} <- {source_name}")
+ elif file_size == -2:
+ # File reference
+ src_folder, src_filename = source_ref
+ src_file_path = unpacked_files.get((src_folder, src_filename))
+ file_path = os.path.join(folder_path, filename)
+ if src_file_path and os.path.exists(src_file_path):
+ shutil.copy2(src_file_path, file_path)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f"Copied: {folder_name}/{filename} <- {src_folder}/{src_filename}")
+ else:
+ file_path = os.path.join(folder_path, filename)
+ async with aiofiles.open(file_path, 'wb') as f:
+ async for chunk in file_chunks:
+ await f.write(chunk)
+ unpacked_files[(folder_name, filename)] = file_path
+ print(f"Unpacked: {folder_name}/{filename} ({file_idx+1}/{num_files}, {file_size} bytes)")
+
+ print(f"\nStream unpacked to {output_dir}")
+
+
+# ============== PACKED ARCHIVE CLASS ==============
+
+@dataclass
+class FileEntry:
+ """Information about a file in the archive."""
+ folder: str
+ filename: str
+ file_type: int # FILE_TYPE_CONTENT or FILE_TYPE_REFERENCE
+ data_offset: int # Position of file content/reference data in archive
+ compressed_size: int # Size of compressed data (0 for references)
+ # For references:
+ ref_folder: Optional[str] = None
+ ref_filename: Optional[str] = None
+
+
+class PackedArchiveFile:
+ """
+ A file-like object for reading a single file from a PackedArchive.
+ Supports read(), readline(), and async iteration.
+ """
+
+ def __init__(self, data: bytes, keep_brotli: bool = False):
+ """
+ Initialize with the file data.
+
+ Args:
+ data: The file data (compressed or decompressed based on keep_brotli)
+ keep_brotli: If True, data is still brotli-compressed
+ """
+ self._data = data
+ self._keep_brotli = keep_brotli
+ self._position = 0
+
+ @property
+ def data(self) -> bytes:
+ """Get all file data."""
+ return self._data
+
+ def read(self, size: int = -1) -> bytes:
+ """Read up to size bytes. If size is -1, read all remaining data."""
+ if size == -1:
+ result = self._data[self._position:]
+ self._position = len(self._data)
+ else:
+ result = self._data[self._position:self._position + size]
+ self._position += len(result)
+ return result
+
+ def readline(self, size: int = -1) -> bytes:
+ """Read a line (up to newline or size bytes)."""
+ if self._position >= len(self._data):
+ return b''
+
+ # Find newline
+ newline_pos = self._data.find(b'\n', self._position)
+ if newline_pos == -1:
+ # No newline, read to end
+ end = len(self._data)
+ else:
+ end = newline_pos + 1
+
+ if size != -1:
+ end = min(end, self._position + size)
+
+ result = self._data[self._position:end]
+ self._position = end
+ return result
+
+ def readlines(self) -> List[bytes]:
+ """Read all remaining lines."""
+ lines = []
+ while True:
+ line = self.readline()
+ if not line:
+ break
+ lines.append(line)
+ return lines
+
+ def seek(self, offset: int, whence: int = 0) -> int:
+ """Seek to position. whence: 0=start, 1=current, 2=end."""
+ if whence == 0:
+ self._position = offset
+ elif whence == 1:
+ self._position += offset
+ elif whence == 2:
+ self._position = len(self._data) + offset
+ self._position = max(0, min(self._position, len(self._data)))
+ return self._position
+
+ def tell(self) -> int:
+ """Return current position."""
+ return self._position
+
+ def __len__(self) -> int:
+ """Return total size."""
+ return len(self._data)
+
+ def __iter__(self):
+ """Iterate over lines."""
+ return self
+
+ def __next__(self) -> bytes:
+ line = self.readline()
+ if not line:
+ raise StopIteration
+ return line
+
+
+class PackedArchive:
+ """
+ Async class to read files from a packed archive as if it were a folder.
+
+ Usage:
+ archive = PackedArchive('packed.bin')
+ await archive.init()
+
+ async with archive.open('vcsky/fetched/model.txd') as f:
+ data = f.read() # Read all
+ # or
+ chunk = f.read(1024) # Read 1024 bytes
+ # or
+ for line in f:
+ print(line)
+
+ # With keep_brotli=True to get compressed data
+ async with archive.open('vcsky/fetched/model.txd', keep_brotli=True) as f:
+ compressed_data = f.read()
+
+ # List files
+ files = archive.list_files()
+ folders = archive.list_folders()
+ """
+
+ def __init__(self, archive_path: str):
+ """
+ Initialize the archive reader.
+
+ Args:
+ archive_path: Path to the .bin archive file
+ """
+ self._path = archive_path
+ self._file: Optional[BinaryIO] = None
+ self._entries: Dict[str, FileEntry] = {} # full_path -> FileEntry
+ self._folders: Dict[str, List[str]] = {} # folder_path -> list of filenames
+ self._folder_copies: Dict[str, str] = {} # copy_folder -> source_folder
+ self._initialized = False
+
+ async def init(self) -> None:
+ """
+ Initialize the archive by reading the index.
+ Must be called before using open().
+ """
+ if self._initialized:
+ return
+
+ async with aiofiles.open(self._path, 'rb') as f:
+ data = await f.read()
+
+ self._parse_index(data)
+ self._initialized = True
+
+ def _parse_index(self, data: bytes) -> None:
+ """Parse the archive to build the file index."""
+ offset = 0
+
+ while offset < len(data):
+ # Read folder type
+ folder_type = data[offset]
+ offset += 1
+
+ # Read folder name
+ folder_name_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ folder_name = data[offset:offset + folder_name_len].decode('utf-8')
+ offset += folder_name_len
+
+ if folder_type == FOLDER_TYPE_COPY:
+ # Read source folder name
+ source_name_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ source_name = data[offset:offset + source_name_len].decode('utf-8')
+ offset += source_name_len
+
+ self._folder_copies[folder_name] = source_name
+ # Copy entries from source folder
+ if source_name in self._folders:
+ self._folders[folder_name] = list(self._folders[source_name])
+ for filename in self._folders[source_name]:
+ src_path = f"{source_name}/{filename}"
+ dst_path = f"{folder_name}/{filename}"
+ if src_path in self._entries:
+ src_entry = self._entries[src_path]
+ self._entries[dst_path] = FileEntry(
+ folder=folder_name,
+ filename=filename,
+ file_type=src_entry.file_type,
+ data_offset=src_entry.data_offset,
+ compressed_size=src_entry.compressed_size,
+ ref_folder=src_entry.ref_folder,
+ ref_filename=src_entry.ref_filename
+ )
+ else:
+ # Normal folder
+ num_files, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+
+ self._folders[folder_name] = []
+
+ for _ in range(num_files):
+ filename_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ filename = data[offset:offset + filename_len].decode('utf-8')
+ offset += filename_len
+
+ self._folders[folder_name].append(filename)
+
+ file_type = data[offset]
+ offset += 1
+
+ full_path = f"{folder_name}/{filename}"
+
+ if file_type == FILE_TYPE_REFERENCE:
+ # Read source reference
+ src_folder_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ src_folder = data[offset:offset + src_folder_len].decode('utf-8')
+ offset += src_folder_len
+
+ src_filename_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+ src_filename = data[offset:offset + src_filename_len].decode('utf-8')
+ offset += src_filename_len
+
+ self._entries[full_path] = FileEntry(
+ folder=folder_name,
+ filename=filename,
+ file_type=FILE_TYPE_REFERENCE,
+ data_offset=0,
+ compressed_size=0,
+ ref_folder=src_folder,
+ ref_filename=src_filename
+ )
+ else:
+ # Read content length and record position
+ compressed_len, bytes_read = decode_uleb128(data, offset)
+ offset += bytes_read
+
+ self._entries[full_path] = FileEntry(
+ folder=folder_name,
+ filename=filename,
+ file_type=FILE_TYPE_CONTENT,
+ data_offset=offset,
+ compressed_size=compressed_len
+ )
+
+ # Skip content
+ offset += compressed_len
+
+ def list_folders(self) -> List[str]:
+ """List all folders in the archive."""
+ if not self._initialized:
+ raise RuntimeError("Archive not initialized. Call init() first.")
+ return list(self._folders.keys())
+
+ def list_files(self, folder: Optional[str] = None) -> List[str]:
+ """
+ List files in the archive.
+
+ Args:
+ folder: If provided, list files only in this folder.
+ If None, list all files with full paths.
+ """
+ if not self._initialized:
+ raise RuntimeError("Archive not initialized. Call init() first.")
+
+ if folder is not None:
+ return list(self._folders.get(folder, []))
+ else:
+ return list(self._entries.keys())
+
+ def exists(self, path: str) -> bool:
+ """Check if a file exists in the archive."""
+ if not self._initialized:
+ raise RuntimeError("Archive not initialized. Call init() first.")
+ return path in self._entries
+
+ @asynccontextmanager
+ async def open(self, path: str, keep_brotli: bool = False):
+ """
+ Open a file from the archive.
+
+ Args:
+ path: Path to the file, e.g., 'vcsky/fetched/model.txd'
+ keep_brotli: If False (default), decompress the data.
+ If True, return the raw data (for brotli passthrough).
+
+ Yields:
+ PackedArchiveFile object for reading the file data.
+
+ Raises:
+ FileNotFoundError: If the file doesn't exist in the archive.
+
+ Note:
+ Files with .br extension are stored without compression in the archive,
+ so they are returned as-is regardless of keep_brotli setting.
+ """
+ if not self._initialized:
+ raise RuntimeError("Archive not initialized. Call init() first.")
+
+ if path not in self._entries:
+ raise FileNotFoundError(f"File not found in archive: {path}")
+
+ entry = self._entries[path]
+ original_filename = entry.filename
+
+ # Resolve references
+ while entry.file_type == FILE_TYPE_REFERENCE:
+ ref_path = f"{entry.ref_folder}/{entry.ref_filename}"
+ if ref_path not in self._entries:
+ raise FileNotFoundError(f"Reference target not found: {ref_path}")
+ entry = self._entries[ref_path]
+
+ # Read the data
+ async with aiofiles.open(self._path, 'rb') as f:
+ await f.seek(entry.data_offset)
+ data = await f.read(entry.compressed_size)
+
+ # .br files are stored as-is (not brotli-compressed in archive)
+ # So we return them directly without decompression
+ if is_already_brotli(original_filename):
+ yield PackedArchiveFile(data, keep_brotli=False)
+ elif keep_brotli:
+ # Return raw brotli-compressed data from archive
+ yield PackedArchiveFile(data, keep_brotli=True)
+ else:
+ # Decompress brotli data
+ decompressed_data = decompress_brotli(data)
+ yield PackedArchiveFile(decompressed_data, keep_brotli=False)
+
+ async def read_file(self, path: str, keep_brotli: bool = False) -> bytes:
+ """
+ Read and return the entire file content.
+
+ Args:
+ path: Path to the file
+ keep_brotli: If False, decompress. If True, return compressed.
+
+ Returns:
+ File content as bytes.
+ """
+ async with self.open(path, keep_brotli=keep_brotli) as f:
+ return f.read()
+
+
+# ============== ADD FOLDER FUNCTION ==============
+
+def add_folder(archive_path: str, folder_path: str, max_workers: int = None) -> None:
+ """
+ Add a folder to an existing archive by appending to the end.
+
+ Note: This appends to the archive without deduplication against existing content.
+ The new folder will be added as a top-level folder in the archive.
+
+ Args:
+ archive_path: Path to existing .bin archive
+ folder_path: Path to folder to add
+ max_workers: Number of parallel compression workers
+ """
+ folder_path = folder_path.rstrip('/\\')
+ parent_dir = os.path.dirname(folder_path) or '.'
+
+ if max_workers is None:
+ max_workers = os.cpu_count() or 4
+
+ if not os.path.isfile(archive_path):
+ raise FileNotFoundError(f"Archive not found: {archive_path}")
+
+ if not os.path.isdir(folder_path):
+ raise NotADirectoryError(f"Not a directory: {folder_path}")
+
+ # Collect files to compress
+ print(f"Adding {folder_path} to {archive_path}")
+ print("Collecting files for compression...")
+
+ files_to_compress: List[Tuple[str, str, str]] = []
+ folder_structure: List[Tuple[str, List[str]]] = []
+
+ for root, dirs, files in os.walk(folder_path):
+ # Filter out ignored files
+ files = [f for f in files if not should_ignore_file(f)]
+ if not files:
+ continue
+
+ rel_path = os.path.relpath(root, parent_dir)
+ folder_structure.append((rel_path, sorted(files)))
+
+ for filename in sorted(files):
+ file_path = os.path.join(root, filename)
+ files_to_compress.append((file_path, rel_path, filename))
+
+ print(f"Compressing {len(files_to_compress)} files using {max_workers} workers...")
+
+ # Parallel compression
+ compressed_files: Dict[Tuple[str, str], bytes] = {}
+ total_original = 0
+ total_compressed = 0
+
+ with ProcessPoolExecutor(max_workers=max_workers) as executor:
+ futures = {executor.submit(compress_file_task, args): args for args in files_to_compress}
+ completed = 0
+
+ for future in as_completed(futures):
+ rel_path, filename, file_path, data, original_size, final_size, is_precompressed = future.result()
+ compressed_files[(rel_path, filename)] = data
+ if is_precompressed:
+ print(f" [{completed + 1}/{len(files_to_compress)}] Stored as-is (.br): {rel_path}/{filename} ({original_size} bytes)")
+ else:
+ total_original += original_size
+ total_compressed += final_size
+ ratio = (final_size / original_size * 100) if original_size > 0 else 0
+ print(f" [{completed + 1}/{len(files_to_compress)}] Compressed: {rel_path}/{filename} ({original_size} -> {final_size} bytes, {ratio:.1f}%)")
+ completed += 1
+
+ print(f"\nAppending to archive...")
+
+ # Append to archive
+ with open(archive_path, 'ab') as out:
+ for rel_path, files in folder_structure:
+ folder_name_bytes = rel_path.encode('utf-8')
+
+ out.write(bytes([FOLDER_TYPE_NORMAL]))
+ out.write(encode_uleb128(len(folder_name_bytes)))
+ out.write(folder_name_bytes)
+ out.write(encode_uleb128(len(files)))
+
+ for filename in files:
+ filename_bytes = filename.encode('utf-8')
+
+ out.write(encode_uleb128(len(filename_bytes)))
+ out.write(filename_bytes)
+
+ compressed_content = compressed_files[(rel_path, filename)]
+
+ out.write(bytes([FILE_TYPE_CONTENT]))
+ out.write(encode_uleb128(len(compressed_content)))
+ out.write(compressed_content)
+
+ new_size = os.path.getsize(archive_path)
+ print(f"\nAdded to {archive_path} (total size: {new_size} bytes)")
+ if total_original > 0:
+ ratio = total_compressed / total_original * 100
+ print(f"Compression: {total_original} -> {total_compressed} bytes ({ratio:.1f}%)")
+
+
+async def add_folder_async(archive_path: str, folder_path: str, max_workers: int = None) -> None:
+ """
+ Add a folder to an existing archive (async version).
+ """
+ folder_path = folder_path.rstrip('/\\')
+ parent_dir = os.path.dirname(folder_path) or '.'
+
+ if max_workers is None:
+ max_workers = os.cpu_count() or 4
+
+ if not os.path.isfile(archive_path):
+ raise FileNotFoundError(f"Archive not found: {archive_path}")
+
+ if not os.path.isdir(folder_path):
+ raise NotADirectoryError(f"Not a directory: {folder_path}")
+
+ print(f"Adding {folder_path} to {archive_path}")
+ print("Collecting files for compression...")
+
+ files_to_compress: List[Tuple[str, str, str]] = []
+ folder_structure: List[Tuple[str, List[str]]] = []
+
+ for root, dirs, files in os.walk(folder_path):
+ # Filter out ignored files
+ files = [f for f in files if not should_ignore_file(f)]
+ if not files:
+ continue
+
+ rel_path = os.path.relpath(root, parent_dir)
+ folder_structure.append((rel_path, sorted(files)))
+
+ for filename in sorted(files):
+ file_path = os.path.join(root, filename)
+ files_to_compress.append((file_path, rel_path, filename))
+
+ print(f"Compressing {len(files_to_compress)} files using {max_workers} workers...")
+
+ compressed_files: Dict[Tuple[str, str], bytes] = {}
+ total_original = 0
+ total_compressed = 0
+ loop = asyncio.get_event_loop()
+
+ with ProcessPoolExecutor(max_workers=max_workers) as executor:
+ future_to_args = {
+ loop.run_in_executor(executor, compress_file_task, args): args
+ for args in files_to_compress
+ }
+
+ completed = 0
+ for coro in asyncio.as_completed(future_to_args.keys()):
+ result = await coro
+ rel_path, filename, file_path, data, original_size, final_size, is_precompressed = result
+ compressed_files[(rel_path, filename)] = data
+ if is_precompressed:
+ print(f" [{completed + 1}/{len(files_to_compress)}] Stored as-is (.br): {rel_path}/{filename} ({original_size} bytes)")
+ else:
+ total_original += original_size
+ total_compressed += final_size
+ ratio = (final_size / original_size * 100) if original_size > 0 else 0
+ print(f" [{completed + 1}/{len(files_to_compress)}] Compressed: {rel_path}/{filename} ({original_size} -> {final_size} bytes, {ratio:.1f}%)")
+ completed += 1
+
+ print(f"\nAppending to archive...")
+
+ async with aiofiles.open(archive_path, 'ab') as out:
+ for rel_path, files in folder_structure:
+ folder_name_bytes = rel_path.encode('utf-8')
+
+ await out.write(bytes([FOLDER_TYPE_NORMAL]))
+ await out.write(encode_uleb128(len(folder_name_bytes)))
+ await out.write(folder_name_bytes)
+ await out.write(encode_uleb128(len(files)))
+
+ for filename in files:
+ filename_bytes = filename.encode('utf-8')
+
+ await out.write(encode_uleb128(len(filename_bytes)))
+ await out.write(filename_bytes)
+
+ compressed_content = compressed_files[(rel_path, filename)]
+
+ await out.write(bytes([FILE_TYPE_CONTENT]))
+ await out.write(encode_uleb128(len(compressed_content)))
+ await out.write(compressed_content)
+
+ new_size = os.path.getsize(archive_path)
+ print(f"\nAdded to {archive_path} (total size: {new_size} bytes)")
+ if total_original > 0:
+ ratio = total_compressed / total_original * 100
+ print(f"Compression: {total_original} -> {total_compressed} bytes ({ratio:.1f}%)")
+
+
+# ============== CLI ==============
+
+def main():
+ if len(sys.argv) < 3:
+ print("Usage:")
+ print(" Pack: python packer_brotli.py pack [--no-dedup] [--workers N]")
+ print(" Unpack: python packer_brotli.py unpack ")
+ print(" Add: python packer_brotli.py add [--workers N]")
+ print()
+ print("Options:")
+ print(" --no-dedup Disable folder and file deduplication during packing")
+ print(" --workers N Number of parallel compression workers (default: CPU count)")
+ print()
+ print("Example:")
+ print(" python packer_brotli.py pack vcsky packed.bin")
+ print(" python packer_brotli.py pack vcsky packed.bin --workers 8")
+ print(" python packer_brotli.py unpack packed.bin unpacked/")
+ print(" python packer_brotli.py add packed.bin vcbr # Add vcbr folder to existing archive")
+ print()
+ print("Features:")
+ print(" - Brotli compression with quality 11 (maximum compression)")
+ print(" - Parallel file compression for maximum speed")
+ print(" - Folder and file deduplication to reduce archive size")
+ print(" - PackedArchive class for reading files directly from archive")
+ print()
+ print("Deduplication: Identical folders and files are detected by comparing")
+ print("content hashes. Duplicates reference the original instead of storing")
+ print("content twice, reducing archive size. File references are only created")
+ print("when the reference path is shorter than storing the file content.")
+ print()
+ print("PackedArchive Usage:")
+ print(" archive = PackedArchive('packed.bin')")
+ print(" await archive.init()")
+ print(" async with archive.open('vcsky/file.txd') as f:")
+ print(" data = f.read()")
+ sys.exit(1)
+
+ command = sys.argv[1]
+
+ if command == 'pack':
+ if len(sys.argv) < 4:
+ print("Usage: python packer_brotli.py pack [--no-dedup] [--workers N]")
+ sys.exit(1)
+ folder_path = sys.argv[2]
+ output_file = sys.argv[3]
+ deduplicate = '--no-dedup' not in sys.argv
+
+ # Parse --workers option
+ max_workers = None
+ if '--workers' in sys.argv:
+ try:
+ workers_idx = sys.argv.index('--workers')
+ max_workers = int(sys.argv[workers_idx + 1])
+ except (IndexError, ValueError):
+ print("Error: --workers requires a numeric argument")
+ sys.exit(1)
+
+ if not os.path.isdir(folder_path):
+ print(f"Error: {folder_path} is not a directory")
+ sys.exit(1)
+
+ pack_folder(folder_path, output_file, deduplicate=deduplicate, max_workers=max_workers)
+
+ elif command == 'unpack':
+ if len(sys.argv) < 4:
+ print("Usage: python packer_brotli.py unpack ")
+ sys.exit(1)
+ input_file = sys.argv[2]
+ output_dir = sys.argv[3]
+
+ if not os.path.isfile(input_file):
+ print(f"Error: {input_file} is not a file")
+ sys.exit(1)
+
+ unpack_file(input_file, output_dir)
+
+ elif command == 'add':
+ if len(sys.argv) < 4:
+ print("Usage: python packer_brotli.py add [--workers N]")
+ sys.exit(1)
+ archive_path = sys.argv[2]
+ folder_path = sys.argv[3]
+
+ # Parse --workers option
+ max_workers = None
+ if '--workers' in sys.argv:
+ try:
+ workers_idx = sys.argv.index('--workers')
+ max_workers = int(sys.argv[workers_idx + 1])
+ except (IndexError, ValueError):
+ print("Error: --workers requires a numeric argument")
+ sys.exit(1)
+
+ if not os.path.isfile(archive_path):
+ print(f"Error: {archive_path} is not a file")
+ sys.exit(1)
+
+ if not os.path.isdir(folder_path):
+ print(f"Error: {folder_path} is not a directory")
+ sys.exit(1)
+
+ add_folder(archive_path, folder_path, max_workers=max_workers)
+
+ else:
+ print(f"Unknown command: {command}")
+ print("Use 'pack', 'unpack', or 'add'")
+ sys.exit(1)
+
+
+if __name__ == '__main__':
+ main()