PythonMastery
intermediate 25 min read · lesson 10 of 15 in Projects

Project: File Organizer (Tidy a Messy Folder)

1 · The lesson

read

You'll build a real automation tool — point it at a messy Downloads folder, and it sorts every file into category subfolders (Documents, Images, Code, Archives, Music, Videos, Other). With a dry-run mode so it never moves anything until you say go.

This is the kind of script that earns Python a permanent spot on your machine.

What you'll practice: pathlib, shutil, dict-of-categories, dry-run pattern, collision handling, recursive walking.


Step 1 — Categorize by Extension

The core idea: a dict mapping extensions → category names.

python
CATEGORIES = {
    "Images":    {".jpg", ".jpeg", ".png", ".gif", ".webp", ".svg", ".heic", ".bmp"},
    "Documents": {".pdf", ".docx", ".doc", ".txt", ".md", ".rtf", ".odt"},
    "Spreadsheets": {".xlsx", ".xls", ".csv", ".ods"},
    "Code":      {".py", ".js", ".ts", ".html", ".css", ".json", ".yml", ".yaml", ".sh"},
    "Archives":  {".zip", ".tar", ".gz", ".7z", ".rar"},
    "Audio":     {".mp3", ".wav", ".flac", ".m4a", ".ogg"},
    "Video":     {".mp4", ".mov", ".mkv", ".avi", ".webm"},
}

def category_for(extension):
    """Return the category name for a file extension, or 'Other' if unknown."""
    ext = extension.lower()
    for category, exts in CATEGORIES.items():
        if ext in exts:
            return category
    return "Other"

# Test
print(category_for(".PDF"))         # Documents
print(category_for(".py"))          # Code
print(category_for(".xyz"))         # Other

The set-of-extensions per category is faster to look up than a list (set membership is O(1)) and reads cleanly.


Step 2 — Walk a Folder

pathlib makes folder iteration pleasant:

python
from pathlib import Path

# In a real script:
#   folder = Path("/Users/me/Downloads")
# For the demo we'll fabricate a structure in memory:
fake_files = [
    "Downloads/photo.jpg",
    "Downloads/resume.pdf",
    "Downloads/notes.txt",
    "Downloads/script.py",
    "Downloads/data.csv",
    "Downloads/movie.mp4",
    "Downloads/archive.zip",
    "Downloads/something.weird",
]

# Pretend each file exists in `folder`
folder = Path("Downloads")
files = [Path(p) for p in fake_files]

print(f"Found {len(files)} files in {folder}:")
for f in files:
    cat = category_for(f.suffix)
    print(f"  {f.name:<25} → {cat}")
+ setup added so this can run · defines category_for
# Lightweight mock for objects whose attributes/methods aren't critical
class _AutoMock:
    def __init__(self, name='mock'): self._name = name
    def __getattr__(self, k): return _AutoMock(self._name + '.' + k)
    def __call__(self, *a, **kw):
        print('-> ' + self._name + '() called')
        return _AutoMock(self._name + '()')
    def __repr__(self): return '<mock ' + self._name + '>'
    def __str__(self): return '<mock ' + self._name + '>'
    def __bool__(self): return True
    def __iter__(self): return iter([])
    def __len__(self): return 0
    def __getitem__(self, k): return _AutoMock(self._name + '[...]')
    def __setitem__(self, k, v): pass
    def __enter__(self): return self
    def __exit__(self, *a): return False
    async def __aenter__(self): return self
    async def __aexit__(self, *a): return False
    def __add__(self, o): return self
    def __radd__(self, o): return self
    def __sub__(self, o): return self
    def __mul__(self, o): return self
    def __rmul__(self, o): return self
    def __truediv__(self, o): return self
    def __eq__(self, o): return isinstance(o, _AutoMock)
    def __hash__(self): return hash(self._name)
    def __lt__(self, o): return True
    def __le__(self, o): return True
    def __gt__(self, o): return False
    def __ge__(self, o): return False
    def __mro_entries__(self, bases): return (object,)

def category_for(*_a, **_kw):
    print('-> category_for() called')
    return _AutoMock('category_for()')

In a real script:

python
# from pathlib import Path
# folder = Path.home() / "Downloads"
# files = [p for p in folder.iterdir() if p.is_file()]

Path.iterdir() yields each item in a directory. .is_file() filters out subdirectories. .suffix gives the extension (including the leading dot).


Step 3 — Plan the Moves (Dry-Run)

Never move files until you've planned exactly what's going to happen and let the user confirm. Build the plan as a list:

python
from pathlib import Path

def plan_moves(files, source):
    """Return a list of (source_path, destination_path) tuples — no actual moves yet."""
    plan = []
    for f in files:
        if not f.name or f.name.startswith("."):
            continue            # skip hidden files
        category = category_for(f.suffix)
        dest_dir = source / category
        dest_path = dest_dir / f.name
        plan.append((f, dest_path))
    return plan

# Demo
source = Path("Downloads")
files = [source / "photo.jpg", source / "resume.pdf", source / "script.py"]
plan = plan_moves(files, source)

print("Planned moves (dry-run):")
for src, dst in plan:
    print(f"  {src.name:<20} → {dst.relative_to(source)}")
+ setup added so this can run · defines category_for
# Lightweight mock for objects whose attributes/methods aren't critical
class _AutoMock:
    def __init__(self, name='mock'): self._name = name
    def __getattr__(self, k): return _AutoMock(self._name + '.' + k)
    def __call__(self, *a, **kw):
        print('-> ' + self._name + '() called')
        return _AutoMock(self._name + '()')
    def __repr__(self): return '<mock ' + self._name + '>'
    def __str__(self): return '<mock ' + self._name + '>'
    def __bool__(self): return True
    def __iter__(self): return iter([])
    def __len__(self): return 0
    def __getitem__(self, k): return _AutoMock(self._name + '[...]')
    def __setitem__(self, k, v): pass
    def __enter__(self): return self
    def __exit__(self, *a): return False
    async def __aenter__(self): return self
    async def __aexit__(self, *a): return False
    def __add__(self, o): return self
    def __radd__(self, o): return self
    def __sub__(self, o): return self
    def __mul__(self, o): return self
    def __rmul__(self, o): return self
    def __truediv__(self, o): return self
    def __eq__(self, o): return isinstance(o, _AutoMock)
    def __hash__(self): return hash(self._name)
    def __lt__(self, o): return True
    def __le__(self, o): return True
    def __gt__(self, o): return False
    def __ge__(self, o): return False
    def __mro_entries__(self, bases): return (object,)

def category_for(*_a, **_kw):
    print('-> category_for() called')
    return _AutoMock('category_for()')

The .relative_to() is a nice touch — shows the destination as Code/script.py instead of the full path.

Dry-run output:

python
Planned moves (dry-run):
  photo.jpg            → Images/photo.jpg
  resume.pdf           → Documents/resume.pdf
  script.py            → Code/script.py

The user reviews this, says "yes go", and only then we move.


Step 4 — Handle Collisions

What if Documents/resume.pdf already exists in the destination? Overwriting silently is dangerous. The standard fix: append a number until we find a unique name.

python
def unique_destination(dest):
    """If dest exists, return dest with `(2)`, `(3)`, ... appended until it doesn't."""
    if not dest.exists():
        return dest
    stem = dest.stem               # filename without extension
    suffix = dest.suffix           # extension including dot
    parent = dest.parent
    n = 2
    while True:
        candidate = parent / f"{stem} ({n}){suffix}"
        if not candidate.exists():
            return candidate
        n += 1

# Demo (simulated):
class FakePath:
    def __init__(self, p): self.p = p
    def exists(self): return self.p in {"Documents/resume.pdf", "Documents/resume (2).pdf"}
    @property
    def stem(self): return "resume"
    @property
    def suffix(self): return ".pdf"
    @property
    def parent(self): return FakePath("Documents")
    def __truediv__(self, other): return FakePath(f"{self.p}/{other}")
    def __str__(self): return self.p

# In real code with real Path objects:
# next_free = unique_destination(Path("Documents/resume.pdf"))
# print(next_free)              # → Documents/resume (3).pdf

print("(See unique_destination's docstring — handles 'resume.pdf' → 'resume (2).pdf' → ... when collisions occur)")

Look up pathlib.Path.stem, .suffix, .parent — they give you the building blocks for any path operation without string-munging.


Step 5 — Execute (with shutil.move)

For actually moving files, shutil.move() handles the cross-filesystem cases that os.rename() doesn't.

python
import shutil
from pathlib import Path

def execute_plan(plan, *, dry_run=True):
    """Carry out the planned moves. Set dry_run=False to actually move."""
    moved = 0
    skipped = 0

    for src, dst in plan:
        if not src.exists():
            print(f"  ⚠️  source vanished: {src.name}")
            skipped += 1
            continue

        # Ensure destination directory exists
        if not dry_run:
            dst.parent.mkdir(parents=True, exist_ok=True)
            dst = unique_destination(dst)
            shutil.move(str(src), str(dst))

        rel = dst.relative_to(dst.parent.parent) if dry_run else dst.relative_to(dst.parent.parent)
        marker = "DRY" if dry_run else "✓"
        print(f"  [{marker}] {src.name:<25} → {rel}")
        moved += 1

    print(f"\n{moved} moved, {skipped} skipped " + ("(dry-run)" if dry_run else ""))
+ setup added so this can run · defines unique_destination
# Lightweight mock for objects whose attributes/methods aren't critical
class _AutoMock:
    def __init__(self, name='mock'): self._name = name
    def __getattr__(self, k): return _AutoMock(self._name + '.' + k)
    def __call__(self, *a, **kw):
        print('-> ' + self._name + '() called')
        return _AutoMock(self._name + '()')
    def __repr__(self): return '<mock ' + self._name + '>'
    def __str__(self): return '<mock ' + self._name + '>'
    def __bool__(self): return True
    def __iter__(self): return iter([])
    def __len__(self): return 0
    def __getitem__(self, k): return _AutoMock(self._name + '[...]')
    def __setitem__(self, k, v): pass
    def __enter__(self): return self
    def __exit__(self, *a): return False
    async def __aenter__(self): return self
    async def __aexit__(self, *a): return False
    def __add__(self, o): return self
    def __radd__(self, o): return self
    def __sub__(self, o): return self
    def __mul__(self, o): return self
    def __rmul__(self, o): return self
    def __truediv__(self, o): return self
    def __eq__(self, o): return isinstance(o, _AutoMock)
    def __hash__(self): return hash(self._name)
    def __lt__(self, o): return True
    def __le__(self, o): return True
    def __gt__(self, o): return False
    def __ge__(self, o): return False
    def __mro_entries__(self, bases): return (object,)

def unique_destination(*_a, **_kw):
    print('-> unique_destination() called')
    return _AutoMock('unique_destination()')

The dry_run flag is the safety pattern. Default it to True. The user has to explicitly say dry_run=False to actually move things.


Step 6 — The Polished Tool

python
from pathlib import Path

CATEGORIES = {
    "Images":      {".jpg", ".jpeg", ".png", ".gif", ".webp", ".svg", ".heic", ".bmp"},
    "Documents":   {".pdf", ".docx", ".doc", ".txt", ".md", ".rtf", ".odt"},
    "Spreadsheets":{".xlsx", ".xls", ".csv", ".ods"},
    "Code":        {".py", ".js", ".ts", ".html", ".css", ".json", ".yml", ".yaml", ".sh"},
    "Archives":    {".zip", ".tar", ".gz", ".7z", ".rar"},
    "Audio":       {".mp3", ".wav", ".flac", ".m4a", ".ogg"},
    "Video":       {".mp4", ".mov", ".mkv", ".avi", ".webm"},
}

def category_for(suffix):
    s = suffix.lower()
    for cat, exts in CATEGORIES.items():
        if s in exts:
            return cat
    return "Other"

def make_plan(folder):
    plan = []
    # In real code: items = folder.iterdir()
    # For demo, fake a list of files:
    items = [
        Path(folder, name) for name in
        ["photo.jpg", "report.pdf", "data.csv", "script.py",
         "song.mp3", "movie.mp4", "archive.zip", "mystery.bin"]
    ]
    for f in items:
        # Skip directories (we can't .is_file() on fake paths — assume all files here)
        cat = category_for(f.suffix)
        dest = folder / cat / f.name
        plan.append((f, dest))
    return plan

def summarize(plan):
    """Print 'Documents: 3 files' style summary."""
    from collections import Counter
    counts = Counter(dst.parent.name for _, dst in plan)
    for cat, n in counts.most_common():
        print(f"  {cat:<15} {n} file{'s' if n != 1 else ''}")

# Run it
source = Path("Downloads")
plan = make_plan(source)

print(f"Plan for {source}:")
summarize(plan)
print()
print("Files:")
for src, dst in plan:
    print(f"  {src.name:<20} → {dst.relative_to(source)}")

print("\nDry-run only — no files moved. In real usage, call execute_plan(plan, dry_run=False).")

Stretch Goals

1. Recursive mode: descend into subfolders (use folder.rglob("*") and .is_file()).
2. Age-based cleanup: move files older than 90 days to an Archive folder (use path.stat().st_mtime).
3. Size-based: separate "large" files (> 100 MB) so you can decide manually.
4. Custom rules file: load categories from rules.yaml so non-coders can extend.
5. Undo log: write each move to organizer.log so you can build an unorganize command.
6. Watcher mode: use the watchdog package to auto-organize as new files arrive in Downloads.


🎯 Your Turn — A summarize_folder Function

Write a function that takes a list of file paths and returns a summary dict: total size per category, file counts, largest file in each category. No actual moving — just inspection.

python
from pathlib import Path
from collections import defaultdict

# Reuse from earlier
CATEGORIES = {
    "Images":    {".jpg", ".jpeg", ".png", ".gif", ".webp"},
    "Documents": {".pdf", ".docx", ".txt", ".md"},
    "Code":      {".py", ".js", ".ts", ".html", ".css"},
}

def category_for(suffix):
    s = suffix.lower()
    for cat, exts in CATEGORIES.items():
        if s in exts: return cat
    return "Other"

def summarize_files(files_with_sizes):
    """Take a list of (path, size_in_bytes) tuples.
    Return a dict: category → {'count': int, 'total_bytes': int, 'largest': (name, size)}.
    """
    # TODO 1: initialize a result dict, perhaps using defaultdict
    # TODO 2: for each (path, size), find its category
    # TODO 3: increment count, add to total_bytes
    # TODO 4: track the largest file per category
    pass

# Test
files = [
    ("photo.jpg", 2_500_000),
    ("vacation.png", 4_100_000),
    ("resume.pdf", 320_000),
    ("notes.txt", 4_000),
    ("script.py", 12_500),
    ("mystery.xyz", 99_000),
]
summary = summarize_files(files)
for cat, info in summary.items():
    print(f"  {cat}: {info['count']} files, {info['total_bytes']:,} bytes, largest: {info['largest']}")
Hint 1 — defaultdict makes this clean from collections import defaultdict result = defaultdict(lambda: {"count": 0, "total_bytes": 0, "largest": ("", 0)}) Then you don't need to check if a category key exists before using it.
Hint 2 — Tracking the largest After updating count and total_bytes, check: if size > result[cat]["largest"][1]: result[cat]["largest"] = (name, size).
Show full solution
python
from pathlib import Path
from collections import defaultdict

def summarize_files(files_with_sizes):
    result = defaultdict(lambda: {"count": 0, "total_bytes": 0, "largest": ("", 0)})
    for path, size in files_with_sizes:
        cat = category_for(Path(path).suffix)
        result[cat]["count"] += 1
        result[cat]["total_bytes"] += size
        if size > result[cat]["largest"][1]:
            result[cat]["largest"] = (path, size)
    return dict(result)

files = [
    ("photo.jpg", 2_500_000),
    ("vacation.png", 4_100_000),
    ("resume.pdf", 320_000),
    ("notes.txt", 4_000),
    ("script.py", 12_500),
    ("mystery.xyz", 99_000),
]

summary = summarize_files(files)
for cat, info in sorted(summary.items(), key=lambda kv: -kv[1]["total_bytes"]):
    name, size = info["largest"]
    mb = info["total_bytes"] / 1_000_000
    print(f"  {cat:<12} {info['count']:>2} files  {mb:>6.2f} MB total  (largest: {name})")
+ setup added so this can run · defines category_for
# Lightweight mock for objects whose attributes/methods aren't critical
class _AutoMock:
    def __init__(self, name='mock'): self._name = name
    def __getattr__(self, k): return _AutoMock(self._name + '.' + k)
    def __call__(self, *a, **kw):
        print('-> ' + self._name + '() called')
        return _AutoMock(self._name + '()')
    def __repr__(self): return '<mock ' + self._name + '>'
    def __str__(self): return '<mock ' + self._name + '>'
    def __bool__(self): return True
    def __iter__(self): return iter([])
    def __len__(self): return 0
    def __getitem__(self, k): return _AutoMock(self._name + '[...]')
    def __setitem__(self, k, v): pass
    def __enter__(self): return self
    def __exit__(self, *a): return False
    async def __aenter__(self): return self
    async def __aexit__(self, *a): return False
    def __add__(self, o): return self
    def __radd__(self, o): return self
    def __sub__(self, o): return self
    def __mul__(self, o): return self
    def __rmul__(self, o): return self
    def __truediv__(self, o): return self
    def __eq__(self, o): return isinstance(o, _AutoMock)
    def __hash__(self): return hash(self._name)
    def __lt__(self, o): return True
    def __le__(self, o): return True
    def __gt__(self, o): return False
    def __ge__(self, o): return False
    def __mro_entries__(self, bases): return (object,)

def category_for(*_a, **_kw):
    print('-> category_for() called')
    return _AutoMock('category_for()')

In a real script, get sizes via path.stat().st_size. This summary is the FIRST thing a "before you move things" tool should produce — so the user knows what's about to happen.


What You Learned

  • pathlib as the modern way to handle paths: .suffix, .stem, .parent, .iterdir(), Path / "subdir".
  • shutil.move() for cross-filesystem-safe moves.
  • The dry-run pattern — every destructive script should have one.
  • Set-of-extensions for fast category lookup.
  • Collision handling via a numbered-suffix loop.

You can now write scripts that automate the boring stuff (Al Sweigart would approve). The dry-run discipline alone separates a useful script from a dangerous one.

You've completed 10 projects across the beginner-to-intermediate range. From here, the Projects path will grow with weather/API tools, web scrapers, mini-web-apps, and data analyses — once the Practical and Web tracks are filled out.

← Back to Academy

Practice this

on practicepython.in

Short exercises that run in your browser and tell you what your code actually did, not just whether a test passed.