Project: File Organizer (Tidy a Messy Folder)
1 · The lesson
readYou'll build a real automation tool — point it at a messy Downloads folder, and it sorts every file into category subfolders (Documents, Images, Code, Archives, Music, Videos, Other). With a dry-run mode so it never moves anything until you say go.
This is the kind of script that earns Python a permanent spot on your machine.
What you'll practice: pathlib, shutil, dict-of-categories, dry-run pattern, collision handling, recursive walking.
Step 1 — Categorize by Extension
The core idea: a dict mapping extensions → category names.
CATEGORIES = {
"Images": {".jpg", ".jpeg", ".png", ".gif", ".webp", ".svg", ".heic", ".bmp"},
"Documents": {".pdf", ".docx", ".doc", ".txt", ".md", ".rtf", ".odt"},
"Spreadsheets": {".xlsx", ".xls", ".csv", ".ods"},
"Code": {".py", ".js", ".ts", ".html", ".css", ".json", ".yml", ".yaml", ".sh"},
"Archives": {".zip", ".tar", ".gz", ".7z", ".rar"},
"Audio": {".mp3", ".wav", ".flac", ".m4a", ".ogg"},
"Video": {".mp4", ".mov", ".mkv", ".avi", ".webm"},
}
def category_for(extension):
"""Return the category name for a file extension, or 'Other' if unknown."""
ext = extension.lower()
for category, exts in CATEGORIES.items():
if ext in exts:
return category
return "Other"
# Test
print(category_for(".PDF")) # Documents
print(category_for(".py")) # Code
print(category_for(".xyz")) # OtherThe set-of-extensions per category is faster to look up than a list (set membership is O(1)) and reads cleanly.
Step 2 — Walk a Folder
pathlib makes folder iteration pleasant:
from pathlib import Path # In a real script: # folder = Path("/Users/me/Downloads") # For the demo we'll fabricate a structure in memory: fake_files = [ "Downloads/photo.jpg", "Downloads/resume.pdf", "Downloads/notes.txt", "Downloads/script.py", "Downloads/data.csv", "Downloads/movie.mp4", "Downloads/archive.zip", "Downloads/something.weird", ] # Pretend each file exists in `folder` folder = Path("Downloads") files = [Path(p) for p in fake_files] print(f"Found {len(files)} files in {folder}:") for f in files: cat = category_for(f.suffix) print(f" {f.name:<25} → {cat}")
setup added so this can run · defines category_for
# Lightweight mock for objects whose attributes/methods aren't critical class _AutoMock: def __init__(self, name='mock'): self._name = name def __getattr__(self, k): return _AutoMock(self._name + '.' + k) def __call__(self, *a, **kw): print('-> ' + self._name + '() called') return _AutoMock(self._name + '()') def __repr__(self): return '<mock ' + self._name + '>' def __str__(self): return '<mock ' + self._name + '>' def __bool__(self): return True def __iter__(self): return iter([]) def __len__(self): return 0 def __getitem__(self, k): return _AutoMock(self._name + '[...]') def __setitem__(self, k, v): pass def __enter__(self): return self def __exit__(self, *a): return False async def __aenter__(self): return self async def __aexit__(self, *a): return False def __add__(self, o): return self def __radd__(self, o): return self def __sub__(self, o): return self def __mul__(self, o): return self def __rmul__(self, o): return self def __truediv__(self, o): return self def __eq__(self, o): return isinstance(o, _AutoMock) def __hash__(self): return hash(self._name) def __lt__(self, o): return True def __le__(self, o): return True def __gt__(self, o): return False def __ge__(self, o): return False def __mro_entries__(self, bases): return (object,) def category_for(*_a, **_kw): print('-> category_for() called') return _AutoMock('category_for()')
In a real script:
# from pathlib import Path # folder = Path.home() / "Downloads" # files = [p for p in folder.iterdir() if p.is_file()]
Path.iterdir() yields each item in a directory. .is_file() filters out subdirectories. .suffix gives the extension (including the leading dot).
Step 3 — Plan the Moves (Dry-Run)
Never move files until you've planned exactly what's going to happen and let the user confirm. Build the plan as a list:
from pathlib import Path def plan_moves(files, source): """Return a list of (source_path, destination_path) tuples — no actual moves yet.""" plan = [] for f in files: if not f.name or f.name.startswith("."): continue # skip hidden files category = category_for(f.suffix) dest_dir = source / category dest_path = dest_dir / f.name plan.append((f, dest_path)) return plan # Demo source = Path("Downloads") files = [source / "photo.jpg", source / "resume.pdf", source / "script.py"] plan = plan_moves(files, source) print("Planned moves (dry-run):") for src, dst in plan: print(f" {src.name:<20} → {dst.relative_to(source)}")
setup added so this can run · defines category_for
# Lightweight mock for objects whose attributes/methods aren't critical class _AutoMock: def __init__(self, name='mock'): self._name = name def __getattr__(self, k): return _AutoMock(self._name + '.' + k) def __call__(self, *a, **kw): print('-> ' + self._name + '() called') return _AutoMock(self._name + '()') def __repr__(self): return '<mock ' + self._name + '>' def __str__(self): return '<mock ' + self._name + '>' def __bool__(self): return True def __iter__(self): return iter([]) def __len__(self): return 0 def __getitem__(self, k): return _AutoMock(self._name + '[...]') def __setitem__(self, k, v): pass def __enter__(self): return self def __exit__(self, *a): return False async def __aenter__(self): return self async def __aexit__(self, *a): return False def __add__(self, o): return self def __radd__(self, o): return self def __sub__(self, o): return self def __mul__(self, o): return self def __rmul__(self, o): return self def __truediv__(self, o): return self def __eq__(self, o): return isinstance(o, _AutoMock) def __hash__(self): return hash(self._name) def __lt__(self, o): return True def __le__(self, o): return True def __gt__(self, o): return False def __ge__(self, o): return False def __mro_entries__(self, bases): return (object,) def category_for(*_a, **_kw): print('-> category_for() called') return _AutoMock('category_for()')
The .relative_to() is a nice touch — shows the destination as Code/script.py instead of the full path.
Dry-run output:
Planned moves (dry-run): photo.jpg → Images/photo.jpg resume.pdf → Documents/resume.pdf script.py → Code/script.py
The user reviews this, says "yes go", and only then we move.
Step 4 — Handle Collisions
What if Documents/resume.pdf already exists in the destination? Overwriting silently is dangerous. The standard fix: append a number until we find a unique name.
def unique_destination(dest): """If dest exists, return dest with `(2)`, `(3)`, ... appended until it doesn't.""" if not dest.exists(): return dest stem = dest.stem # filename without extension suffix = dest.suffix # extension including dot parent = dest.parent n = 2 while True: candidate = parent / f"{stem} ({n}){suffix}" if not candidate.exists(): return candidate n += 1 # Demo (simulated): class FakePath: def __init__(self, p): self.p = p def exists(self): return self.p in {"Documents/resume.pdf", "Documents/resume (2).pdf"} @property def stem(self): return "resume" @property def suffix(self): return ".pdf" @property def parent(self): return FakePath("Documents") def __truediv__(self, other): return FakePath(f"{self.p}/{other}") def __str__(self): return self.p # In real code with real Path objects: # next_free = unique_destination(Path("Documents/resume.pdf")) # print(next_free) # → Documents/resume (3).pdf print("(See unique_destination's docstring — handles 'resume.pdf' → 'resume (2).pdf' → ... when collisions occur)")
Look up pathlib.Path.stem, .suffix, .parent — they give you the building blocks for any path operation without string-munging.
Step 5 — Execute (with shutil.move)
For actually moving files, shutil.move() handles the cross-filesystem cases that os.rename() doesn't.
import shutil from pathlib import Path def execute_plan(plan, *, dry_run=True): """Carry out the planned moves. Set dry_run=False to actually move.""" moved = 0 skipped = 0 for src, dst in plan: if not src.exists(): print(f" ⚠️ source vanished: {src.name}") skipped += 1 continue # Ensure destination directory exists if not dry_run: dst.parent.mkdir(parents=True, exist_ok=True) dst = unique_destination(dst) shutil.move(str(src), str(dst)) rel = dst.relative_to(dst.parent.parent) if dry_run else dst.relative_to(dst.parent.parent) marker = "DRY" if dry_run else "✓" print(f" [{marker}] {src.name:<25} → {rel}") moved += 1 print(f"\n{moved} moved, {skipped} skipped " + ("(dry-run)" if dry_run else ""))
setup added so this can run · defines unique_destination
# Lightweight mock for objects whose attributes/methods aren't critical class _AutoMock: def __init__(self, name='mock'): self._name = name def __getattr__(self, k): return _AutoMock(self._name + '.' + k) def __call__(self, *a, **kw): print('-> ' + self._name + '() called') return _AutoMock(self._name + '()') def __repr__(self): return '<mock ' + self._name + '>' def __str__(self): return '<mock ' + self._name + '>' def __bool__(self): return True def __iter__(self): return iter([]) def __len__(self): return 0 def __getitem__(self, k): return _AutoMock(self._name + '[...]') def __setitem__(self, k, v): pass def __enter__(self): return self def __exit__(self, *a): return False async def __aenter__(self): return self async def __aexit__(self, *a): return False def __add__(self, o): return self def __radd__(self, o): return self def __sub__(self, o): return self def __mul__(self, o): return self def __rmul__(self, o): return self def __truediv__(self, o): return self def __eq__(self, o): return isinstance(o, _AutoMock) def __hash__(self): return hash(self._name) def __lt__(self, o): return True def __le__(self, o): return True def __gt__(self, o): return False def __ge__(self, o): return False def __mro_entries__(self, bases): return (object,) def unique_destination(*_a, **_kw): print('-> unique_destination() called') return _AutoMock('unique_destination()')
The dry_run flag is the safety pattern. Default it to True. The user has to explicitly say dry_run=False to actually move things.
Step 6 — The Polished Tool
from pathlib import Path CATEGORIES = { "Images": {".jpg", ".jpeg", ".png", ".gif", ".webp", ".svg", ".heic", ".bmp"}, "Documents": {".pdf", ".docx", ".doc", ".txt", ".md", ".rtf", ".odt"}, "Spreadsheets":{".xlsx", ".xls", ".csv", ".ods"}, "Code": {".py", ".js", ".ts", ".html", ".css", ".json", ".yml", ".yaml", ".sh"}, "Archives": {".zip", ".tar", ".gz", ".7z", ".rar"}, "Audio": {".mp3", ".wav", ".flac", ".m4a", ".ogg"}, "Video": {".mp4", ".mov", ".mkv", ".avi", ".webm"}, } def category_for(suffix): s = suffix.lower() for cat, exts in CATEGORIES.items(): if s in exts: return cat return "Other" def make_plan(folder): plan = [] # In real code: items = folder.iterdir() # For demo, fake a list of files: items = [ Path(folder, name) for name in ["photo.jpg", "report.pdf", "data.csv", "script.py", "song.mp3", "movie.mp4", "archive.zip", "mystery.bin"] ] for f in items: # Skip directories (we can't .is_file() on fake paths — assume all files here) cat = category_for(f.suffix) dest = folder / cat / f.name plan.append((f, dest)) return plan def summarize(plan): """Print 'Documents: 3 files' style summary.""" from collections import Counter counts = Counter(dst.parent.name for _, dst in plan) for cat, n in counts.most_common(): print(f" {cat:<15} {n} file{'s' if n != 1 else ''}") # Run it source = Path("Downloads") plan = make_plan(source) print(f"Plan for {source}:") summarize(plan) print() print("Files:") for src, dst in plan: print(f" {src.name:<20} → {dst.relative_to(source)}") print("\nDry-run only — no files moved. In real usage, call execute_plan(plan, dry_run=False).")
Stretch Goals
1. Recursive mode: descend into subfolders (use folder.rglob("*") and .is_file()).
2. Age-based cleanup: move files older than 90 days to an Archive folder (use path.stat().st_mtime).
3. Size-based: separate "large" files (> 100 MB) so you can decide manually.
4. Custom rules file: load categories from rules.yaml so non-coders can extend.
5. Undo log: write each move to organizer.log so you can build an unorganize command.
6. Watcher mode: use the watchdog package to auto-organize as new files arrive in Downloads.
🎯 Your Turn — A summarize_folder Function
Write a function that takes a list of file paths and returns a summary dict: total size per category, file counts, largest file in each category. No actual moving — just inspection.
from pathlib import Path from collections import defaultdict # Reuse from earlier CATEGORIES = { "Images": {".jpg", ".jpeg", ".png", ".gif", ".webp"}, "Documents": {".pdf", ".docx", ".txt", ".md"}, "Code": {".py", ".js", ".ts", ".html", ".css"}, } def category_for(suffix): s = suffix.lower() for cat, exts in CATEGORIES.items(): if s in exts: return cat return "Other" def summarize_files(files_with_sizes): """Take a list of (path, size_in_bytes) tuples. Return a dict: category → {'count': int, 'total_bytes': int, 'largest': (name, size)}. """ # TODO 1: initialize a result dict, perhaps using defaultdict # TODO 2: for each (path, size), find its category # TODO 3: increment count, add to total_bytes # TODO 4: track the largest file per category pass # Test files = [ ("photo.jpg", 2_500_000), ("vacation.png", 4_100_000), ("resume.pdf", 320_000), ("notes.txt", 4_000), ("script.py", 12_500), ("mystery.xyz", 99_000), ] summary = summarize_files(files) for cat, info in summary.items(): print(f" {cat}: {info['count']} files, {info['total_bytes']:,} bytes, largest: {info['largest']}")
Hint 1 — defaultdict makes this clean
from collections import defaultdict
result = defaultdict(lambda: {"count": 0, "total_bytes": 0, "largest": ("", 0)})
Then you don't need to check if a category key exists before using it.
Hint 2 — Tracking the largest
After updating count and total_bytes, check:if size > result[cat]["largest"][1]: result[cat]["largest"] = (name, size).
Show full solution
from pathlib import Path from collections import defaultdict def summarize_files(files_with_sizes): result = defaultdict(lambda: {"count": 0, "total_bytes": 0, "largest": ("", 0)}) for path, size in files_with_sizes: cat = category_for(Path(path).suffix) result[cat]["count"] += 1 result[cat]["total_bytes"] += size if size > result[cat]["largest"][1]: result[cat]["largest"] = (path, size) return dict(result) files = [ ("photo.jpg", 2_500_000), ("vacation.png", 4_100_000), ("resume.pdf", 320_000), ("notes.txt", 4_000), ("script.py", 12_500), ("mystery.xyz", 99_000), ] summary = summarize_files(files) for cat, info in sorted(summary.items(), key=lambda kv: -kv[1]["total_bytes"]): name, size = info["largest"] mb = info["total_bytes"] / 1_000_000 print(f" {cat:<12} {info['count']:>2} files {mb:>6.2f} MB total (largest: {name})")
setup added so this can run · defines category_for
# Lightweight mock for objects whose attributes/methods aren't critical class _AutoMock: def __init__(self, name='mock'): self._name = name def __getattr__(self, k): return _AutoMock(self._name + '.' + k) def __call__(self, *a, **kw): print('-> ' + self._name + '() called') return _AutoMock(self._name + '()') def __repr__(self): return '<mock ' + self._name + '>' def __str__(self): return '<mock ' + self._name + '>' def __bool__(self): return True def __iter__(self): return iter([]) def __len__(self): return 0 def __getitem__(self, k): return _AutoMock(self._name + '[...]') def __setitem__(self, k, v): pass def __enter__(self): return self def __exit__(self, *a): return False async def __aenter__(self): return self async def __aexit__(self, *a): return False def __add__(self, o): return self def __radd__(self, o): return self def __sub__(self, o): return self def __mul__(self, o): return self def __rmul__(self, o): return self def __truediv__(self, o): return self def __eq__(self, o): return isinstance(o, _AutoMock) def __hash__(self): return hash(self._name) def __lt__(self, o): return True def __le__(self, o): return True def __gt__(self, o): return False def __ge__(self, o): return False def __mro_entries__(self, bases): return (object,) def category_for(*_a, **_kw): print('-> category_for() called') return _AutoMock('category_for()')
In a real script, get sizes via path.stat().st_size. This summary is the FIRST thing a "before you move things" tool should produce — so the user knows what's about to happen.
What You Learned
pathlibas the modern way to handle paths:.suffix,.stem,.parent,.iterdir(),Path / "subdir".shutil.move()for cross-filesystem-safe moves.- The dry-run pattern — every destructive script should have one.
- Set-of-extensions for fast category lookup.
- Collision handling via a numbered-suffix loop.
You can now write scripts that automate the boring stuff (Al Sweigart would approve). The dry-run discipline alone separates a useful script from a dangerous one.
You've completed 10 projects across the beginner-to-intermediate range. From here, the Projects path will grow with weather/API tools, web scrapers, mini-web-apps, and data analyses — once the Practical and Web tracks are filled out.
Practice this
on practicepython.inShort exercises that run in your browser and tell you what your code actually did, not just whether a test passed.