"""
PIR (Package Install/Restore) Fuzz Test
Long-running 7x24 fuzz test for epkg install/remove/restore cycles.
Designed to run in a dedicated user account with optimized filesystem layout.
Usage:
CACHE_DIR=/path/to/large/disk python3 pir.py setup
CACHE_DIR=/path/to/large/disk python3 pir.py run --os=openeuler
CACHE_DIR=/path/to/large/disk python3 pir.py --os=openeuler # setup + run
"""
import argparse
import os
import platform
import random
import re
import signal
import subprocess
import sys
import time
from datetime import datetime
from pathlib import Path
from typing import Optional, Tuple
_interrupted = False
DEFAULT_BATCH_SIZE = 5
DEFAULT_MAX_ERRORS = 10
TMPFS_SIZE_GB = 6
USAGE_THRESHOLD_RECREATE = 40
USAGE_THRESHOLD_SKIP = 80
USAGE_THRESHOLD_GC = 60
ESTIMATION_ERROR_PERCENT = 10
EPKG_BIN = os.environ.get('EPKG_BIN', None)
CACHE_DIR = os.environ.get('CACHE_DIR', None)
BAD_CASES_DIR = None
IS_MACOS = platform.system() == 'Darwin'
IS_LINUX = platform.system() == 'Linux'
HOME = Path.home()
USER = os.environ.get('USER', 'unknown')
def log(msg: str):
"""Log message with timestamp."""
timestamp = datetime.now().strftime('%Y-%m-%d %H:%M:%S')
print(f"[{timestamp}] {msg}", flush=True)
def signal_handler(signum, frame):
"""Handle SIGINT (Ctrl-C) to gracefully stop the fuzz loop."""
global _interrupted
_interrupted = True
log(f"Received signal {signum}, stopping after current iteration...")
def setup_signal_handlers():
"""Set up signal handlers for graceful shutdown."""
signal.signal(signal.SIGINT, signal_handler)
signal.signal(signal.SIGTERM, signal_handler)
def get_system_memory_gb() -> float:
"""Get system total memory in GB."""
if IS_MACOS:
result = subprocess.run(['sysctl', '-n', 'hw.memsize'], capture_output=True)
return int(result.stdout.strip()) / (1024 ** 3)
elif IS_LINUX:
with open('/proc/meminfo') as f:
for line in f:
if line.startswith('MemTotal'):
return int(line.split()[1]) / (1024 ** 2)
return 16
def get_tmpfs_target_size_gb() -> int:
"""Calculate tmpfs target size: 4GB or < half system memory."""
mem_gb = get_system_memory_gb()
half_mem = int(mem_gb * 0.45)
return min(TMPFS_SIZE_GB, half_mem)
def get_tmpfs_path() -> Path:
"""Get tmpfs base path for epkg."""
return Path(f"/tmp/epkg-{USER}")
def get_epkg_symlink_path() -> Path:
"""Get .epkg symlink path."""
return HOME / ".epkg"
def get_cache_symlink_path() -> Path:
"""Get cache symlink path based on platform."""
if IS_MACOS:
return HOME / "Library" / "Caches" / "epkg"
else:
return HOME / ".cache" / "epkg"
def is_path_mounted(path: Path) -> bool:
"""Check if a path is a mount point."""
result = subprocess.run(['mount'], capture_output=True, text=True)
for line in result.stdout.splitlines():
if str(path) in line:
return True
return False
def setup_macos_ramdisk(tmpfs_path: Path):
"""Create and mount RamDisk on macOS."""
if tmpfs_path.exists() and is_path_mounted(tmpfs_path):
log(f"RamDisk already mounted at {tmpfs_path}")
return True
size_gb = get_tmpfs_target_size_gb()
num_blocks = size_gb * 2097152
log(f"Creating macOS RamDisk: {size_gb}GB ({num_blocks} blocks)")
result = subprocess.run(
['hdid', '-nomount', f'ram://{num_blocks}'],
capture_output=True, text=True
)
if result.returncode != 0:
log(f"ERROR: hdid failed: {result.stderr}")
return False
device = result.stdout.strip().split()[0]
log(f"RAM device: {device}")
result = subprocess.run(
['newfs_hfs', '-v', f'epkg-{USER}', device],
capture_output=True, text=True
)
if result.returncode != 0:
log(f"ERROR: newfs_hfs failed: {result.stderr}")
subprocess.run(['hdiutil', 'detach', device], capture_output=True)
return False
tmpfs_path.mkdir(parents=True, exist_ok=True)
result = subprocess.run(
['mount', '-t', 'hfs', device, str(tmpfs_path)],
capture_output=True, text=True
)
if result.returncode != 0:
log(f"ERROR: mount failed: {result.stderr}")
subprocess.run(['hdiutil', 'detach', device], capture_output=True)
return False
log(f"RamDisk mounted at {tmpfs_path}")
return True
def cmd_setup(dry_run: bool = False) -> bool:
"""
Setup filesystem layout for fuzz testing.
Layout:
- $HOME/.epkg -> /tmp/epkg-$USER (tmpfs)
- $HOME/.cache/epkg or Library/Caches/epkg -> $CACHE_DIR
"""
tmpfs_path = get_tmpfs_path()
epkg_link = get_epkg_symlink_path()
cache_link = get_cache_symlink_path()
if not CACHE_DIR:
log("ERROR: CACHE_DIR environment variable is required")
return False
cache_dir = Path(CACHE_DIR)
if not cache_dir.exists():
log(f"Creating CACHE_DIR: {cache_dir}")
if not dry_run:
cache_dir.mkdir(parents=True, exist_ok=True)
global BAD_CASES_DIR
BAD_CASES_DIR = cache_dir / "bad-cases"
if not BAD_CASES_DIR.exists() and not dry_run:
BAD_CASES_DIR.mkdir(parents=True, exist_ok=True)
log(f"Layout configuration:")
log(f" TMPFS path: {tmpfs_path}")
log(f" .epkg link: {epkg_link} -> {tmpfs_path}")
log(f" cache link: {cache_link} -> {cache_dir}")
log(f" bad-cases: {BAD_CASES_DIR}")
if dry_run:
return True
if IS_MACOS:
setup_macos_ramdisk(tmpfs_path)
elif IS_LINUX:
tmpfs_path.mkdir(parents=True, exist_ok=True)
if epkg_link.exists():
if epkg_link.is_symlink():
current_target = epkg_link.resolve()
expected_target = tmpfs_path.resolve()
if current_target != expected_target:
log(f"WARNING: .epkg links to {current_target}, expected {expected_target}")
elif IS_LINUX and is_path_mounted(epkg_link):
log(f"INFO: .epkg is a tmpfs mount (valid alternative to symlink)")
else:
log(f"ERROR: .epkg exists but is not a symlink or tmpfs mount")
return False
else:
log(f"Creating symlink: {epkg_link} -> {tmpfs_path}")
epkg_link.symlink_to(tmpfs_path)
if cache_link.exists():
if cache_link.is_symlink():
current_target = cache_link.resolve()
expected_target = cache_dir.resolve()
if current_target != expected_target:
log(f"WARNING: cache links to {current_target}, expected {expected_target}")
elif cache_link.resolve() == cache_dir.resolve():
log(f"INFO: cache is the actual directory (valid alternative to symlink)")
else:
log(f"ERROR: cache exists but is not a symlink or the specified CACHE_DIR")
return False
else:
log(f"Creating symlink: {cache_link} -> {cache_dir}")
cache_link.parent.mkdir(parents=True, exist_ok=True)
cache_link.symlink_to(cache_dir)
try:
epkg_bin = find_epkg_binary()
log(f"epkg binary: {epkg_bin}")
except RuntimeError as e:
log(f"ERROR: {e}")
return False
return True
def find_epkg_binary() -> Path:
"""Find epkg binary path, run self install if needed."""
candidates = []
if EPKG_BIN:
candidates.append(Path(EPKG_BIN))
script_dir = Path(__file__).parent
project_root = script_dir.parent.parent
self_epkg = HOME / ".epkg" / "envs" / "self" / "usr" / "bin" / "epkg"
self_assets = HOME / ".epkg" / "envs" / "self" / "usr" / "src" / "epkg" / "assets" / "repos"
debug_epkg = project_root / "target" / "debug" / "epkg"
release_epkg = project_root / "target" / "release" / "epkg"
if self_epkg.exists() and self_assets.exists():
return self_epkg.resolve()
candidates.extend([debug_epkg, release_epkg])
built_epkg = None
for path in candidates:
if path.exists() and path.is_file():
built_epkg = path
break
if built_epkg:
log(f"Running epkg self install (self env not found or incomplete)")
result = subprocess.run(
[str(built_epkg), 'self', 'install'],
capture_output=True, text=True
)
if result.returncode != 0:
log(f"ERROR: epkg self install failed: {result.stderr}")
raise RuntimeError("epkg self install failed")
if self_epkg.exists():
return self_epkg.resolve()
return built_epkg.resolve()
raise RuntimeError("epkg binary not found. Set EPKG_BIN or build the project.")
def run_epkg(args: list, env_name: str, capture_output: bool = True,
timeout: Optional[int] = None) -> subprocess.CompletedProcess:
"""
Run epkg command with logging.
Environment variables:
- RUST_LOG=warn
- RUST_BACKTRACE=1
Args:
args: epkg command arguments (e.g. ['install', 'htop'])
env_name: environment name
capture_output: if True, capture stdout/stderr
timeout: timeout in seconds (None = no timeout, 0 = use default)
Returns:
subprocess.CompletedProcess result
"""
epkg_bin = find_epkg_binary()
cmd = [str(epkg_bin), '-e', env_name] + args
env = os.environ.copy()
env['RUST_LOG'] = 'warn'
env['RUST_BACKTRACE'] = '1'
log(f"Running: {' '.join(cmd)}")
if timeout == 0:
timeout = 600
try:
if capture_output:
return subprocess.run(cmd, capture_output=True, text=True, env=env, timeout=timeout)
else:
return subprocess.run(cmd, env=env, timeout=timeout)
except subprocess.TimeoutExpired:
log(f"Command timed out after {timeout} seconds: {' '.join(cmd)}")
return subprocess.CompletedProcess(cmd, returncode=-1, stdout="", stderr=f"Timeout after {timeout}s")
def get_available_packages(os_name: str, env_name: str) -> list:
"""Get list of available packages for the OS."""
log(f"Getting available packages for {os_name}")
result = run_epkg(['list', '--available'], env_name)
if result.returncode != 0:
log(f"ERROR: Failed to list packages: {result.stderr}")
return []
packages = []
for line in result.stdout.splitlines():
parts = line.split()
if len(parts) >= 5 and parts[0].startswith(('A', '_')):
packages.append(parts[4])
log(f"Found {len(packages)} available packages")
return packages
def get_installed_executables(env_name: str) -> list:
"""Get list of installed executables in the environment."""
env_path = get_epkg_symlink_path() / "envs" / env_name
bin_path = env_path / "usr" / "bin"
if not bin_path.exists():
return []
executables = []
for exe in bin_path.iterdir():
if exe.is_file() and not exe.is_symlink():
executables.append(f"/usr/bin/{exe.name}")
return executables
def detect_supported_flags(real_exe_path: str) -> list:
"""Detect supported flags by scanning binary for printable strings."""
try:
with open(real_exe_path, 'rb') as f:
data = f.read()
except (FileNotFoundError, PermissionError):
return []
strings = re.findall(rb'[\x20-\x7e]{4,}', data)
text = b'\n'.join(strings).decode('ascii', errors='ignore')
flags = []
if '--help' in text:
flags.append('--help')
if '--version' in text:
flags.append('--version')
return flags
def test_executable_help(env_name: str, exe_path: str) -> Tuple[bool, str]:
"""Test executable with detected flags only.
First scans binary for --help/--version strings to avoid calling
unsupported flags that may cause programs to block or misbehave.
"""
env_path = get_epkg_symlink_path() / "envs" / env_name
exe_name = exe_path.split('/')[-1]
real_exe = env_path / "usr" / "bin" / exe_name
supported = detect_supported_flags(str(real_exe))
if not supported:
return True, f"Skipped (no --help/--version): {exe_path}"
last_output = ""
for flag in supported:
result = run_epkg(['run', '--', exe_path, flag], env_name, timeout=5)
if result.returncode == -1 and 'Timeout' in result.stderr:
continue
if result.returncode == 0:
return True, result.stdout + result.stderr
output = result.stdout + result.stderr
last_output = output
if any(kw.lower() in output.lower() for kw in ['Usage', 'Options', 'version', 'Copyright']):
return True, output
if any(kw in output for kw in [
'error while loading shared libraries',
'cannot open shared object file',
'No such file or directory'
]):
return False, output
return False, f"Flags failed: {exe_path}\nOutput:\n{last_output}"
def check_log_for_errors(log_content: str) -> list:
"""Check log content for error patterns."""
errors = []
for line in log_content.splitlines():
if 'ERROR' in line:
if 'read_chunk_from_stream: Read error' in line:
continue
if 'Peer disconnected' in line:
continue
if 'File size mismatch after write' in line:
continue
errors.append(f"ERROR: {line}")
if 'WARN' in line and any(keyword in line for keyword in [
'failed', 'error', 'cannot', 'missing', 'broken'
]):
if 'Transaction file conflict' in line:
continue
errors.append(f"WARN: {line}")
if 'panic' in line.lower() or 'thread panicked' in line.lower():
errors.append(f"PANIC: {line}")
return errors
def get_tmpfs_usage_bytes() -> Tuple[int, int]:
"""Get current tmpfs usage and total in bytes.
Returns:
Tuple: (used_bytes, total_bytes)
"""
tmpfs_path = get_tmpfs_path()
result = subprocess.run(['df', '-k', str(tmpfs_path)], capture_output=True, text=True)
for line in result.stdout.splitlines():
if str(tmpfs_path) in line or 'epkg' in line:
parts = line.split()
if len(parts) >= 5:
total_kb = int(parts[1])
used_kb = int(parts[2])
return used_kb * 1024, total_kb * 1024
return 0, 0
def parse_epkg_estimated_space(output: str) -> Optional[int]:
"""Parse epkg output to extract estimated disk space usage.
Parses lines like:
- "After this operation, 2.5 GB of additional disk space will be used."
- "After this operation, 1.2 MB of additional disk space will be used (50 MB available)."
Note: "After block alignment" output has been removed because concurrent download+unpack
means packages already occupy most disk space when info/filelist.txt is obtained,
making file-count-based adjustment meaningless.
Returns:
Estimated space in bytes, or None if not found
"""
pattern = r"After this operation, ([\d.]+)\s*(KB|MB|GB|TB)? of additional disk space will be used"
match = re.search(pattern, output, re.IGNORECASE)
if not match:
return None
size_str = match.group(1)
unit = match.group(2) or 'B'
try:
size = float(size_str)
except ValueError:
return None
unit_multipliers = {
'B': 1,
'KB': 1024,
'MB': 1024 * 1024,
'GB': 1024 * 1024 * 1024,
'TB': 1024 * 1024 * 1024 * 1024,
}
multiplier = unit_multipliers.get(unit.upper(), 1)
return int(size * multiplier)
def format_size(bytes_val: int) -> str:
"""Format bytes to human readable size string."""
if bytes_val >= 1024 * 1024 * 1024:
return f"{bytes_val / (1024 * 1024 * 1024):.2f} GB"
elif bytes_val >= 1024 * 1024:
return f"{bytes_val / (1024 * 1024):.2f} MB"
elif bytes_val >= 1024:
return f"{bytes_val / 1024:.2f} KB"
else:
return f"{bytes_val} B"
def compare_disk_space_estimate(
before_bytes: int,
after_bytes: int,
estimated_bytes: Optional[int],
batch: list
) -> None:
"""Compare actual disk space change with epkg estimate and log results.
Args:
before_bytes: Disk usage before install (bytes)
after_bytes: Disk usage after install (bytes)
estimated_bytes: Epkg estimated space usage (bytes), or None
batch: List of package names installed
"""
actual_delta = after_bytes - before_bytes
_, total_bytes = get_tmpfs_usage_bytes()
free_bytes = total_bytes - after_bytes
error_threshold_bytes = free_bytes // 8
log(f"Disk space comparison for batch: {' '.join(batch)}")
log(f" Before: {format_size(before_bytes)}")
log(f" After: {format_size(after_bytes)}")
log(f" Actual Δ: {format_size(actual_delta)} ({actual_delta} bytes)")
if estimated_bytes is not None:
if estimated_bytes == 0:
error_pct = 0.0 if actual_delta == 0 else float('inf')
else:
error_pct = ((actual_delta - estimated_bytes) / estimated_bytes) * 100
error_bytes = actual_delta - estimated_bytes
log(f" Estimated: {format_size(estimated_bytes)} ({estimated_bytes} bytes)")
log(f" Error: {format_size(error_bytes)} ({error_pct:+.1f}%)")
log(f" Free space: {format_size(free_bytes)}, threshold: {format_size(error_threshold_bytes)}")
if abs(error_pct) > ESTIMATION_ERROR_PERCENT:
log(f" WARNING: Estimation error too large!")
else:
log(f" Estimated: (not found in epkg output)")
def get_tmpfs_usage_percent() -> float:
"""Get current tmpfs usage percentage."""
used_bytes, total_bytes = get_tmpfs_usage_bytes()
if total_bytes > 0:
return (used_bytes / total_bytes) * 100
return 0.0
def load_whitelist() -> list:
"""Load dependency resolution whitelist from tests/fuzz/whitelist.txt."""
script_dir = Path(__file__).parent
whitelist_file = script_dir / "whitelist.txt"
patterns = []
if whitelist_file.exists():
with open(whitelist_file) as f:
for line in f:
line = line.strip()
if line and not line.startswith('#'):
patterns.append(line)
log(f"Loaded {len(patterns)} dependency whitelist patterns")
return patterns
def load_exe_whitelist() -> list:
"""Load executable test whitelist from tests/fuzz/exe_whitelist.txt.
File format: pattern # reason (inline comment after #)
"""
script_dir = Path(__file__).parent
whitelist_file = script_dir / "exe_whitelist.txt"
patterns = []
if whitelist_file.exists():
with open(whitelist_file) as f:
for line in f:
line = line.strip()
if not line or line.startswith('#'):
continue
if '#' in line:
pattern = line.split('#')[0].strip()
else:
pattern = line
if pattern:
patterns.append(pattern)
log(f"Loaded {len(patterns)} executable whitelist patterns")
return patterns
def error_matches_whitelist(error_msg: str, whitelist: list) -> bool:
"""Check if error message matches any whitelist pattern using regex.
Uses case-insensitive matching since error messages from different
programs may have varying capitalization (e.g., "Cannot open display"
vs "cannot open display").
"""
for pattern in whitelist:
if re.search(pattern, error_msg, re.IGNORECASE):
return True
return False
def save_bad_case(os_name: str, commands: list, log_content: str, error_type: str,
error_msg: str = "", is_depends_error: bool = False):
"""Save bad case artifacts for later analysis."""
if not BAD_CASES_DIR:
log("ERROR: BAD_CASES_DIR not set")
return
timestamp = datetime.now().strftime('%Y%m%d_%H%M%S')
case_dir = BAD_CASES_DIR / f"{timestamp}_{os_name}_{error_type}"
case_dir.mkdir(parents=True, exist_ok=True)
env_name = f"fuzz-{os_name}"
epkg_bin = find_epkg_binary()
reproduce_sh = case_dir / "reproduce.sh"
with open(reproduce_sh, 'w') as f:
f.write("#!/bin/sh\n")
f.write(f"# Generated by {__file__}\n")
f.write("#\n")
f.write("# ============================================================\n")
f.write("# Problem Analysis Principles (问题分析原则):\n")
f.write("# 1. 分析问题的性质 (nature) - 是代码bug、配置问题、还是repo问题?\n")
f.write("# 2. 分析问题的本质 (essence) - 根本原因是什么?\n")
f.write("# 3. 确定妥善处理方法 (proper solution):\n")
f.write("# - 代码bug: 修复代码,不要workaround\n")
f.write("# - repo问题: 添加whitelist并记录原因\n")
f.write("# - 配置问题: 调整配置或文档\n")
f.write("#\n")
f.write("# Key principles (关键原则):\n")
f.write("# - \"do proper fix\" - 正确修复,不要临时方案\n")
f.write("# - \"no fix for sake of fix\" - 不要为了修而修\n")
f.write("# - 反虚假声明: 不确定的地方说不确定,没验证就不要暗示通过了\n")
f.write("# - 消除干扰项: 例如,日志中的正常行为warn(各用户场景下均无实际风险), 应降级为debug/info\n")
f.write("# (就像需要消除所有build warnings一样)\n")
f.write("# - 先看输出,再决定: 不基于猜测添加 skip pattern 或 workaround\n")
f.write("# - 保留证据: 错误输出要完整记录,便于后续分析\n")
f.write("# - No hack fix: 不要因为测试失败就跳过,要找到真正原因\n")
f.write("# ============================================================\n")
f.write("#\n")
f.write("# AI troubleshooting guide:\n")
f.write("# 1. Turn on MORE debug log: RUST_LOG=trace\n")
f.write("# 2. Refer to below commands to reproduce the error\n")
f.write("# 3. Redirect log output to file, then analyze relevant lines like grep -B3 -A3 ...\n")
if is_depends_error:
f.write("#\n")
f.write("# Whitelist handling:\n")
f.write("# If this is a repo dependency issue (not epkg bug), add pattern to\n")
f.write("# tests/fuzz/whitelist.txt with a comment explaining:\n")
f.write("# - Package name and version\n")
f.write("# - Why it's unresolvable (e.g., broken deps in upstream repo)\n")
f.write("#\n")
f.write(f"# OS: {os_name}\n")
f.write(f"# Error type: {error_type}\n")
if error_msg:
error_preview = error_msg[:500] if len(error_msg) > 500 else error_msg
f.write(f"# Error preview: {error_preview}\n")
f.write(f"# Generated: {timestamp}\n\n")
f.write("set -x\n\n")
f.write(f"EPKG_BIN='{epkg_bin}'\n")
f.write(f"ENV_NAME='{env_name}'\n\n")
f.write(f"$EPKG_BIN env remove $ENV_NAME 2>/dev/null # ignore error: may fail on first run\n")
f.write(f"$EPKG_BIN env create $ENV_NAME -c {os_name}\n\n")
for cmd in commands:
cmd_replaced = re.sub(r'^epkg', '$EPKG_BIN', cmd)
f.write(f"{cmd_replaced}\n")
reproduce_sh.chmod(0o755)
epkg_log = case_dir / "epkg.log"
with open(epkg_log, 'w') as f:
f.write(log_content)
log(f"Saved bad case to: {case_dir}")
log(f" reproduce.sh: {len(commands)} commands")
log(f" epkg.log: {len(log_content)} bytes")
def create_environment(os_name: str) -> str:
"""Create fuzz test environment for the OS."""
env_name = f"fuzz-{os_name}"
epkg_bin = find_epkg_binary()
env_path = get_epkg_symlink_path() / "envs" / env_name
env_vars = os.environ.copy()
env_vars['RUST_LOG'] = 'warn'
env_vars['RUST_BACKTRACE'] = '1'
if env_path.exists():
log(f"Removing existing environment: {env_name}")
result = subprocess.run(
[str(epkg_bin), '-e', 'self', 'env', 'remove', env_name],
capture_output=True, text=True, env=env_vars
)
if result.returncode != 0:
log(f"env remove failed: {result.stderr[:200]}")
log(f"Creating environment: {env_name}")
result = subprocess.run(
[str(epkg_bin), '-e', 'self', 'env', 'create', env_name, '-c', os_name],
capture_output=True, text=True, env=env_vars
)
if result.returncode != 0:
raise RuntimeError(f"Failed to create environment: {result.stderr}")
return env_name
def recreate_environment(os_name: str) -> str:
"""Recreate fuzz test environment (more thorough cleanup than restore)."""
log(f"Recreating environment for {os_name}")
return create_environment(os_name)
def get_current_generation(env_name: str) -> int:
"""Get current generation number for an environment."""
env_path = get_epkg_symlink_path() / "envs" / env_name
current_link = env_path / "generations" / "current"
if not current_link.exists():
return 0
target = current_link.resolve()
if not target.exists():
return 0
gen_name = target.name
try:
return int(gen_name)
except ValueError:
return 0
def select_random_packages(packages: list, batch_size: int) -> list:
"""Select random batch of packages for testing."""
batch = random.sample(packages, min(batch_size, len(packages)))
batch_str = ' '.join(batch)
log(f"Selected packages: {batch_str}")
return batch
def install_packages_batch(env_name: str, batch: list) -> Tuple[subprocess.CompletedProcess, str, str, int]:
"""Install a batch of packages and return results.
Returns:
Tuple: (result, cmd_str, log_output, before_bytes)
"""
batch_str = ' '.join(batch)
cmd_str = f"epkg -e {env_name} install --assume-yes --ignore-file-conflicts {batch_str}"
before_bytes, _ = get_tmpfs_usage_bytes()
result = run_epkg(['install', '--assume-yes', '--ignore-file-conflicts'] + batch, env_name, timeout=0)
log_output = f"=== INSTALL ===\n{result.stdout}\n{result.stderr}\n"
return result, cmd_str, log_output, before_bytes
def check_install_errors(result: subprocess.CompletedProcess, whitelist: list) -> Tuple[str, bool, bool]:
"""Check install result for errors.
Returns:
Tuple: (error_type, has_error, is_whitelisted)
error_type can be: None, "install_fail", "install_warn", "install_timeout",
"disk_space" (special: triggers gc+recreate, not counted as error)
"""
install_error = result.returncode != 0
log_errors = check_log_for_errors(result.stdout + result.stderr)
combined_output = result.stdout + result.stderr
if "Insufficient disk space" in combined_output:
log(f"Insufficient disk space detected - will trigger gc and recreate env")
return "disk_space", True, False
if result.returncode == -1 and 'Timeout' in result.stderr:
return "install_timeout", True, False
if "Dependency resolution failed" in combined_output:
if error_matches_whitelist(combined_output, whitelist):
log(f"Dependency resolution error matches whitelist - skipping (repo issue)")
return None, False, True
for pattern in whitelist:
if pattern in combined_output:
log(f"Dependency error matches whitelist pattern '{pattern}' - skipping (repo issue)")
return None, False, True
if install_error or log_errors:
error_type = "install_fail" if install_error else "install_warn"
return error_type, True, False
return None, False, False
def test_executables_batch(env_name: str, exe_whitelist: list = None) -> Tuple[list, list, str]:
"""Test installed executables.
Returns:
Tuple: (exe_errors, commands, log_output)
"""
if exe_whitelist is None:
exe_whitelist = []
executables = get_installed_executables(env_name)
log(f"Testing {len(executables)} executables")
exe_errors = []
commands = []
failed_outputs = []
whitelisted_count = 0
for exe in executables:
env_path = get_epkg_symlink_path() / "envs" / env_name
exe_name = exe.split('/')[-1]
real_exe = env_path / "usr" / "bin" / exe_name
supported = detect_supported_flags(str(real_exe))
for flag in supported:
cmd_str = f"epkg -e {env_name} run -- {exe} {flag}"
commands.append(cmd_str)
success, output = test_executable_help(env_name, exe)
if not success:
match_text = f"{exe}\n{output}"
if error_matches_whitelist(match_text, exe_whitelist):
log(f"Executable test whitelisted: {exe}")
whitelisted_count += 1
continue
exe_errors.append(exe)
failed_outputs.append(f"=== RUN {exe} (FAILED) ===\n{output}\n")
log(f"Executable test failed: {exe}")
if whitelisted_count > 0:
log(f"Whitelisted {whitelisted_count} executable failures (VM/upstream issues)")
return exe_errors, commands, "".join(failed_outputs)
def restore_to_generation_zero(env_name: str) -> Tuple[subprocess.CompletedProcess, str, str]:
"""Restore environment to generation 0.
Returns:
Tuple: (result, cmd_str, log_output)
"""
log("Restoring to generation 0")
cmd_str = f"epkg -e {env_name} restore 0"
result = run_epkg(['--assume-yes', 'restore', '0'], env_name, timeout=0)
log_output = f"=== RESTORE ===\n{result.stdout}\n{result.stderr}\n"
return result, cmd_str, log_output
def check_restore_errors(result: subprocess.CompletedProcess) -> Tuple[str, bool]:
"""Check restore result for errors.
Returns:
Tuple: (error_type, has_error)
"""
if result.returncode == -1 and 'Timeout' in result.stderr:
return "restore_timeout", True
if result.returncode != 0:
return "restore_fail", True
return None, False
def run_gc_if_needed(usage: float, force_gc: bool = False) -> None:
"""Run gc if usage exceeds threshold or force_gc is True."""
if force_gc or usage > USAGE_THRESHOLD_GC:
log(f"Running epkg gc (usage: {usage:.1f}%)")
result = run_epkg(['--assume-yes', 'gc'], env_name='self')
log(f"GC output: {result.stdout[:100]}")
class FuzzIterationContext:
"""Context for a single fuzz iteration."""
def __init__(self, os_name: str, env_name: str, packages: list,
batch_size: int, whitelist: list, exe_whitelist: list, usage: float):
self.os_name = os_name
self.env_name = env_name
self.packages = packages
self.batch_size = batch_size
self.whitelist = whitelist
self.exe_whitelist = exe_whitelist
self.usage = usage
self.loop_commands = []
self.loop_log = ""
self.batch = []
self.need_recreate = False
self.need_gc = False
def should_skip_iteration(self) -> bool:
"""Check if iteration should be skipped due to high disk usage."""
if self.usage > USAGE_THRESHOLD_SKIP:
log(f"Disk usage {self.usage:.1f}% > {USAGE_THRESHOLD_SKIP}%, skipping iteration")
self.need_gc = True
return True
return False
def should_recreate_env(self) -> bool:
"""Check if env should be recreated due to moderate disk usage."""
if self.usage > USAGE_THRESHOLD_RECREATE:
log(f"Disk usage {self.usage:.1f}% > {USAGE_THRESHOLD_RECREATE}%, will recreate env after iteration")
self.need_recreate = True
return True
return False
def run_fuzz_iteration(ctx: FuzzIterationContext) -> Tuple[str, bool]:
"""
Run single fuzz iteration: install, test executables, restore.
Returns:
Tuple: (error_type or None, has_error)
"""
log(f"TMPFS usage: {ctx.usage:.1f}%")
if ctx.should_skip_iteration():
return None, False
ctx.should_recreate_env()
ctx.batch = select_random_packages(ctx.packages, ctx.batch_size)
result, cmd_str, log_output, before_bytes = install_packages_batch(ctx.env_name, ctx.batch)
ctx.loop_commands.append(cmd_str)
ctx.loop_log += log_output
after_bytes, _ = get_tmpfs_usage_bytes()
post_install_usage = get_tmpfs_usage_percent()
if post_install_usage > USAGE_THRESHOLD_GC:
log(f"Post-install disk usage {post_install_usage:.1f}% > {USAGE_THRESHOLD_GC}%, will run gc after iteration")
ctx.need_gc = True
if post_install_usage > USAGE_THRESHOLD_RECREATE and not ctx.need_recreate:
log(f"Post-install disk usage {post_install_usage:.1f}% > {USAGE_THRESHOLD_RECREATE}%, will recreate env")
ctx.need_recreate = True
error_type, has_error, is_whitelisted = check_install_errors(result, ctx.whitelist)
if is_whitelisted:
return None, False
if error_type == "disk_space":
ctx.need_gc = True
ctx.need_recreate = True
return None, False
if has_error:
is_depends_error = "Dependency resolution failed" in (result.stdout + result.stderr)
error_msg = result.stderr if error_type == "install_fail" else str(check_log_for_errors(result.stdout + result.stderr)[:3])
save_bad_case(ctx.os_name, ctx.loop_commands, ctx.loop_log, error_type, error_msg, is_depends_error)
log(f"Install error detected: {error_type}")
ctx.need_recreate = True
return error_type, True
combined_output = result.stdout + result.stderr
estimated_bytes = parse_epkg_estimated_space(combined_output)
compare_disk_space_estimate(before_bytes, after_bytes, estimated_bytes, ctx.batch)
exe_errors, exe_commands, exe_log = test_executables_batch(ctx.env_name, ctx.exe_whitelist)
ctx.loop_commands.extend(exe_commands)
ctx.loop_log += exe_log
if exe_errors:
save_bad_case(ctx.os_name, ctx.loop_commands, ctx.loop_log, "exe_fail", f"Failed executables: {exe_errors}")
log(f"Executable errors detected: {exe_errors}")
ctx.need_recreate = True
return "exe_fail", True
if ctx.need_recreate:
log("Skipping restore since env will be recreated")
return None, False
current_gen = get_current_generation(ctx.env_name)
assert current_gen > 0, f"BUG: generation is 0 after successful install of {ctx.batch}"
result, cmd_str, log_output = restore_to_generation_zero(ctx.env_name)
ctx.loop_commands.append(cmd_str)
ctx.loop_log += log_output
error_type, has_error = check_restore_errors(result)
if has_error:
save_bad_case(ctx.os_name, ctx.loop_commands, ctx.loop_log, error_type, result.stderr)
log(f"Restore error detected: {error_type}")
ctx.need_recreate = True
return error_type, True
final_usage = get_tmpfs_usage_percent()
if final_usage > USAGE_THRESHOLD_GC:
log(f"Post-restore disk usage {final_usage:.1f}% > {USAGE_THRESHOLD_GC}%, will run gc")
ctx.need_gc = True
return None, False
def get_cache_dir_from_symlink() -> Optional[Path]:
"""Get CACHE_DIR from existing cache symlink."""
cache_link = get_cache_symlink_path()
if cache_link.exists() and cache_link.is_symlink():
return cache_link.resolve()
return None
def cmd_run(os_name: str, batch_size: int, max_errors: int):
"""
Main fuzz test loop.
Loop:
- Check tmpfs usage, decide cleanup strategy
- Install random packages
- Test executables
- Restore to generation 0 or recreate env
- Run gc at end of loop if needed
- Save bad cases on errors
"""
setup_signal_handlers()
cache_dir = None
if CACHE_DIR:
cache_dir = Path(CACHE_DIR)
else:
cache_dir = get_cache_dir_from_symlink()
if not cache_dir:
log("ERROR: CACHE_DIR not set and cache symlink not found")
log(" Run 'pir.py setup' first or set CACHE_DIR environment variable")
return
global BAD_CASES_DIR
BAD_CASES_DIR = cache_dir / "bad-cases"
if not BAD_CASES_DIR.exists():
BAD_CASES_DIR.mkdir(parents=True, exist_ok=True)
old_dir = BAD_CASES_DIR / ".old"
if not old_dir.exists():
old_dir.mkdir(parents=True, exist_ok=True)
for item in BAD_CASES_DIR.iterdir():
if item.name == ".old":
continue
timestamp = time.strftime('%Y%m%d_%H%M%S')
dest = old_dir / f"{timestamp}_{item.name}"
try:
item.rename(dest)
log(f"Archived old bad case: {item.name} -> .old/{dest.name}")
except Exception as e:
log(f"Warning: Could not archive {item.name}: {e}")
whitelist = load_whitelist()
exe_whitelist = load_exe_whitelist()
env_name = create_environment(os_name)
packages = get_available_packages(os_name, env_name)
if not packages:
log("ERROR: No packages available")
return
error_count = 0
loop_count = 0
log(f"Starting fuzz loop: batch_size={batch_size}, max_errors={max_errors}")
log(f"Thresholds: recreate_env={USAGE_THRESHOLD_RECREATE}%, skip_iter={USAGE_THRESHOLD_SKIP}%, run_gc={USAGE_THRESHOLD_GC}%")
while error_count < max_errors:
if _interrupted:
log("Interrupted by user, stopping...")
break
loop_count += 1
log(f"=== Loop {loop_count} ===")
usage = get_tmpfs_usage_percent()
ctx = FuzzIterationContext(
os_name=os_name,
env_name=env_name,
packages=packages,
batch_size=batch_size,
whitelist=whitelist,
exe_whitelist=exe_whitelist,
usage=usage
)
error_type, has_error = run_fuzz_iteration(ctx)
if has_error:
error_count += 1
log(f"Error count: {error_count}/{max_errors}")
if ctx.need_recreate:
log("Recreating environment for thorough cleanup")
env_name = recreate_environment(os_name)
packages = get_available_packages(os_name, env_name)
end_usage = get_tmpfs_usage_percent()
run_gc_if_needed(end_usage, ctx.need_gc)
time.sleep(1)
log(f"Fuzz test completed: {loop_count} loops, {error_count} errors")
if error_count >= max_errors and BAD_CASES_DIR:
prompt = f"""
================================================================================
AI Agent Prompt for Error Analysis
================================================================================
请分类合并 {BAD_CASES_DIR}/* 里暴露的问题(ignore .old/):
1. 先做简单分类(不做分析!):
- 列出各组的问题清单作为 todos
2. 然后逐个处理:
- 深入分析:问题的性质、本质、以及妥善处理方法
- do proper fix(正确修复,不要临时方案)
- commit
- 处完一个再下一个
关键原则:
- 一个个分析,深入分析,别一起看
- 别根据走马观花的浅层分析来指导后续 fix
- no fix for sake of fix(不要为了修而修)
- 反虚假声明:不要根据猜测来决定fix,而要步步为营,根据事实和证据链一步步推导,没验证就不要假想搞定了
================================================================================
"""
print(prompt)
def main():
parser = argparse.ArgumentParser(
description='PIR (Package Install/Restore) Fuzz Test for epkg',
usage='python3 pir.py [setup|run] [--os OS] [--batch N] [--max-errors N]'
)
subparsers = parser.add_subparsers(dest='command', help='Commands')
setup_parser = subparsers.add_parser('setup', help='Setup filesystem layout only')
setup_parser.add_argument('--dry-run', action='store_true', help='Show layout without creating')
run_parser = subparsers.add_parser('run', help='Run fuzz test loop (assumes layout already setup)')
run_parser.add_argument('--os', required=True, help='Target OS')
run_parser.add_argument('--batch', type=int, default=DEFAULT_BATCH_SIZE, help='Batch size')
run_parser.add_argument('--max-errors', type=int, default=DEFAULT_MAX_ERRORS, help='Max errors')
parser.add_argument('--os', help='Target OS (for default mode: setup + run)')
parser.add_argument('--batch', type=int, default=DEFAULT_BATCH_SIZE, help='Batch size')
parser.add_argument('--max-errors', type=int, default=DEFAULT_MAX_ERRORS, help='Max errors')
parser.add_argument('--dry-run', action='store_true', help='Dry run (setup only)')
args = parser.parse_args()
log("PIR (Package Install/Restore) Fuzz Test")
log(f"Platform: {platform.system()}")
if args.command == 'setup':
log("Mode: setup only")
if not cmd_setup(args.dry_run):
sys.exit(1)
log("Setup complete")
elif args.command == 'run':
log(f"Mode: run only (assumes layout already setup)")
log(f"OS target: {args.os}")
log(f"Batch size: {args.batch}")
log(f"Max errors: {args.max_errors}")
try:
cmd_run(args.os, args.batch, args.max_errors)
except KeyboardInterrupt:
log("Interrupted by user")
except Exception as e:
log(f"ERROR: {e}")
sys.exit(1)
else:
if not args.os:
parser.error("--os is required for default mode")
log(f"Mode: setup + run")
log(f"OS target: {args.os}")
log(f"Batch size: {args.batch}")
log(f"Max errors: {args.max_errors}")
if not cmd_setup(args.dry_run):
sys.exit(1)
if args.dry_run:
log("Dry run complete")
sys.exit(0)
try:
cmd_run(args.os, args.batch, args.max_errors)
except KeyboardInterrupt:
log("Interrupted by user")
except Exception as e:
log(f"ERROR: {e}")
sys.exit(1)
if __name__ == '__main__':
main()