diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..a73762b --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,65 @@ +# CLAUDE.md + +This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. + +## Project Overview + +**imgeda** is a CLI tool for exploratory data analysis (EDA) of image datasets. It scans image directories, generates JSONL manifests with metadata/pixel statistics, detects quality issues, finds duplicates, and produces visualizations. Built for both local use and AWS Lambda deployment. + +## Commands + +```bash +# Install dependencies +uv sync --all-extras + +# Run all tests +uv run pytest + +# Run single test file +uv run pytest tests/test_analyzer.py + +# Run single test +uv run pytest tests/test_analyzer.py::TestAnalyzeImage::test_normal_image + +# Run with coverage +uv run pytest --cov=src/imgeda --cov-report=html + +# Lint +uv run ruff check src/ tests/ + +# Lint with autofix +uv run ruff check --fix src/ tests/ + +# Format +uv run ruff format src/ tests/ + +# Type check (strict mode) +uv run mypy src/imgeda/ +``` + +## Architecture + +The codebase follows a layered architecture: **CLI → Pipeline → Core (pure functions)**. + +- **`src/imgeda/cli/`** — Typer-based CLI commands (`scan`, `check`, `plot`, `report`, `info`, interactive wizard). Entry point: `cli/app.py`. +- **`src/imgeda/core/`** — Pure analysis functions with zero CLI dependencies, designed to be Lambda-compatible. Includes `analyzer.py` (single-image analysis), `detector.py` (exposure/artifact detection), `hasher.py` (perceptual hashing), `duplicates.py` (hash-based clustering with sub-hash bucketing to avoid O(n²)), and `aggregator.py` (dataset summary). +- **`src/imgeda/pipeline/`** — Orchestration layer: `ProcessPoolExecutor` parallelism with Rich progress bars, crash-tolerant resume via checkpoint logic, and graceful Ctrl+C signal handling. Batched processing with memory-bounded futures (batch size 5000). +- **`src/imgeda/io/`** — JSONL manifest I/O with atomic writes (temp file + rename) and corruption-tolerant parsing (skips malformed lines). +- **`src/imgeda/models/`** — Dataclasses with `__slots__`: `ImageRecord`, `PixelStats`, `CornerStats`, `ManifestMeta`, `ScanConfig`, `PlotConfig`. +- **`src/imgeda/plotting/`** — Eight plot types (dimensions, file_size, aspect_ratio, brightness, channels, artifacts, duplicates), each in its own module. +- **`src/imgeda/lambda_handler/`** — AWS Lambda entry point wrapping core functions. + +## Key Design Patterns + +- **Core functions never raise exceptions** — corrupt/unreadable files are flagged in the `ImageRecord` rather than throwing. +- **Resume is keyed on `(path, file_size, mtime)`** — modified files are automatically re-analyzed on resume. +- **JSONL manifest format** — first line is metadata (`__manifest_meta__: true`), remaining lines are `ImageRecord` entries. Append-only with atomic metadata updates. +- **Serialization uses orjson** for performance. + +## Code Style + +- Line length: 100 characters +- Target Python: 3.10+ (uses `from __future__ import annotations`) +- Type annotations throughout; mypy strict mode +- `__slots__` on all dataclasses +- Tests are class-based with pytest fixtures; test images are generated programmatically in `conftest.py` diff --git a/src/imgeda/pipeline/runner.py b/src/imgeda/pipeline/runner.py index 580ad0b..c569535 100644 --- a/src/imgeda/pipeline/runner.py +++ b/src/imgeda/pipeline/runner.py @@ -20,7 +20,7 @@ from imgeda.core.analyzer import analyze_image from imgeda.io.image_reader import discover_images -from imgeda.io.manifest_io import append_records, create_manifest +from imgeda.io.manifest_io import append_records, create_manifest, write_meta from imgeda.models.config import ScanConfig from imgeda.models.manifest import ImageRecord, ManifestMeta from imgeda.pipeline.checkpoint import filter_pending, load_processed_set @@ -67,10 +67,26 @@ def _run_scan_inner( # Resume logic already_processed = 0 + meta = ManifestMeta( + input_dir=os.path.abspath(input_dir), + total_files=total_discovered, + created_at=datetime.now(timezone.utc).isoformat(), + settings={ + "workers": config.workers, + "include_hashes": config.include_hashes, + "skip_pixel_stats": config.skip_pixel_stats, + "artifact_threshold": config.artifact_threshold, + "dark_threshold": config.dark_threshold, + "overexposed_threshold": config.overexposed_threshold, + }, + ) + if config.resume and not config.force and output.exists(): processed_set, existing_records = load_processed_set(output_path) already_processed = len(existing_records) pending = filter_pending(all_images, processed_set) + # Update metadata header without truncating existing records + write_meta(output_path, meta) if already_processed > 0: console.print( f" Resuming: [green]{already_processed:,}[/green] already processed, " @@ -78,20 +94,6 @@ def _run_scan_inner( ) else: pending = all_images - # Truncate and write fresh manifest header - meta = ManifestMeta( - input_dir=os.path.abspath(input_dir), - total_files=total_discovered, - created_at=datetime.now(timezone.utc).isoformat(), - settings={ - "workers": config.workers, - "include_hashes": config.include_hashes, - "skip_pixel_stats": config.skip_pixel_stats, - "artifact_threshold": config.artifact_threshold, - "dark_threshold": config.dark_threshold, - "overexposed_threshold": config.overexposed_threshold, - }, - ) create_manifest(output_path, meta) if not pending: diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index 7dd862d..536423b 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -2,12 +2,14 @@ from __future__ import annotations +import os from pathlib import Path import pytest -from imgeda.io.manifest_io import read_manifest +from imgeda.io.manifest_io import append_records, create_manifest, read_manifest from imgeda.models.config import ScanConfig +from imgeda.models.manifest import ManifestMeta from imgeda.pipeline.checkpoint import filter_pending from imgeda.pipeline.runner import run_scan @@ -28,8 +30,8 @@ def test_scan_basic(self, tmp_image_dir: Path, tmp_path: Path) -> None: assert corrupt >= 1 # we have one corrupt file @pytest.mark.timeout(60) - def test_scan_resume(self, tmp_image_dir: Path, tmp_path: Path) -> None: - """Test that resume skips already-processed images.""" + def test_scan_resume_no_duplicates(self, tmp_image_dir: Path, tmp_path: Path) -> None: + """Resume on a fully-scanned manifest must not duplicate records.""" output = str(tmp_path / "manifest.jsonl") config = ScanConfig(workers=2, checkpoint_every=5) @@ -42,22 +44,119 @@ def test_scan_resume(self, tmp_image_dir: Path, tmp_path: Path) -> None: # Should not add duplicates _, records = read_manifest(output) assert len(records) == total1 + assert total2 == total1 @pytest.mark.timeout(60) - def test_scan_force(self, tmp_image_dir: Path, tmp_path: Path) -> None: - """Test force rescan.""" + def test_scan_resume_preserves_records(self, tmp_image_dir: Path, tmp_path: Path) -> None: + """Resume must preserve all previously-written records (no truncation).""" output = str(tmp_path / "manifest.jsonl") - config = ScanConfig(workers=2, checkpoint_every=5, force=True, resume=False) + config = ScanConfig(workers=2, checkpoint_every=5) + # Full scan run_scan(str(tmp_image_dir), output, config) - _, records1 = read_manifest(output) + _, records_before = read_manifest(output) + paths_before = {r.path for r in records_before} + # Resume run run_scan(str(tmp_image_dir), output, config) - _, records2 = read_manifest(output) + _, records_after = read_manifest(output) + paths_after = {r.path for r in records_after} + + # Every record from the first run must still be present + assert paths_before == paths_after + + @pytest.mark.timeout(60) + def test_scan_resume_partial(self, tmp_image_dir: Path, tmp_path: Path) -> None: + """Simulate a partial run: pre-populate manifest with some records, then resume.""" + from imgeda.io.image_reader import discover_images + + output = str(tmp_path / "manifest.jsonl") + config = ScanConfig(workers=2, checkpoint_every=5) + all_images = discover_images(str(tmp_image_dir), config.extensions) + total_images = len(all_images) + assert total_images > 3, "Need enough test images for a meaningful partial test" + + # Create a manifest with only the first 3 images processed + meta = ManifestMeta( + input_dir=os.path.abspath(str(tmp_image_dir)), + total_files=total_images, + created_at="2025-01-01T00:00:00+00:00", + ) + create_manifest(output, meta) + + # Analyze first 3 images and write their records + from imgeda.core.analyzer import analyze_image + + partial_records = [analyze_image(p, config) for p in all_images[:3]] + append_records(output, partial_records) + partial_paths = {r.path for r in partial_records} + + # Resume should process the remaining images + total, _ = run_scan(str(tmp_image_dir), output, config) + _, records = read_manifest(output) + + assert total == total_images + assert len(records) == total_images - # Force should have rescanned but result in same count + # Verify previously-processed records are still present + final_paths = {r.path for r in records} + assert partial_paths.issubset(final_paths) + + @pytest.mark.timeout(60) + def test_scan_force_truncates(self, tmp_image_dir: Path, tmp_path: Path) -> None: + """--force must truncate and rescan from scratch.""" + output = str(tmp_path / "manifest.jsonl") + + # Normal scan first + config = ScanConfig(workers=2, checkpoint_every=5) + run_scan(str(tmp_image_dir), output, config) + meta1, records1 = read_manifest(output) + + # Force rescan + config_force = ScanConfig(workers=2, checkpoint_every=5, force=True, resume=False) + run_scan(str(tmp_image_dir), output, config_force) + meta2, records2 = read_manifest(output) + + # Force should have rescanned; same count but fresh metadata timestamp assert len(records1) > 0 - assert len(records2) > 0 + assert len(records2) == len(records1) + assert meta1 is not None + assert meta2 is not None + assert meta2.created_at != meta1.created_at + + @pytest.mark.timeout(60) + def test_scan_resume_nonexistent_manifest(self, tmp_image_dir: Path, tmp_path: Path) -> None: + """Resume with no existing manifest should create a new one and scan all images.""" + output = str(tmp_path / "manifest.jsonl") + assert not Path(output).exists() + + config = ScanConfig(workers=2, checkpoint_every=5, resume=True) + total, _ = run_scan(str(tmp_image_dir), output, config) + + meta, records = read_manifest(output) + assert meta is not None + assert len(records) == total + assert total > 0 + + @pytest.mark.timeout(60) + def test_scan_resume_updates_metadata(self, tmp_image_dir: Path, tmp_path: Path) -> None: + """Resume must update the metadata header (e.g., total_files) without truncating.""" + output = str(tmp_path / "manifest.jsonl") + config = ScanConfig(workers=2, checkpoint_every=5) + + # First scan + run_scan(str(tmp_image_dir), output, config) + meta1, records1 = read_manifest(output) + assert meta1 is not None + + # Resume scan + run_scan(str(tmp_image_dir), output, config) + meta2, records2 = read_manifest(output) + assert meta2 is not None + + # Records preserved, metadata refreshed + assert len(records2) == len(records1) + assert meta2.total_files == meta1.total_files @pytest.mark.timeout(60) def test_scan_metadata_only(self, tmp_image_dir: Path, tmp_path: Path) -> None: