Lesson 25 · Python standard library deep dive
Build a Python CLI Tool from Scratch | Standard Library Capstone #25
Video twenty-five, the final video of the twenty-five-part series: a genuine command-line duplicate-file finder. Combining argparse, pathlib, glob, hashlib,…
- CoursePython standard library deep dive
- Lesson25 of 24
- Video18 min
- FormatJupyter notebook · 10 code cells
- Data5 datasets
What you'll learn
Datasets used in this lesson
Save these next to the notebook. In Google Colab, upload them with the 📁 icon on the left first.
- report.json70 B
- demo_report.json299 B
- demo_report.csv87 B
- demo_main_report.json339 B
- demo_main_report.csv125 B
📓 Full notebook
Download .ipynbPython Standard Library Deep-Dive, Video 25: Capstone - a Real CLI Tool#
- Video twenty-five, the final video of the twenty-five-part series: a genuine command-line duplicate-file finder.
- Combining argparse, pathlib, glob, hashlib, collections, logging, dataclasses, json, csv, and datetime.
- Let's build something real.
Part 1: What We're Building#
MODULES_USED = ['argparse', 'pathlib', 'glob', 'hashlib', 'collections', 'logging', 'dataclasses', 'json', 'csv', 'datetime']
print(f'This capstone combines {len(MODULES_USED)} stdlib modules:')
for m in MODULES_USED:
print(f' - {m}')
Part 2: FileRecord (dataclasses) and build_argparser() (argparse)#
import argparse
from dataclasses import dataclass
@dataclass
class FileRecord:
path: str
size: int
digest: str
def build_argparser():
parser = argparse.ArgumentParser(description='Find duplicate files by content.')
parser.add_argument('directory', help='directory to scan')
parser.add_argument('--pattern', default='*', help='glob pattern to match files')
parser.add_argument('--json-out', default='report.json', help='path for the JSON report')
parser.add_argument('--csv-out', default='report.csv', help='path for the CSV report')
return parser
parser = build_argparser()
args = parser.parse_args(['demo_scan_dir', '--pattern', '*.txt'])
print(args)
Part 3: discover_files() Using pathlib and glob#
import os
from pathlib import Path
os.makedirs('demo_scan_dir/sub', exist_ok=True)
(Path('demo_scan_dir') / 'a.txt').write_text('shared content')
(Path('demo_scan_dir') / 'b.txt').write_text('shared content')
(Path('demo_scan_dir') / 'c.txt').write_text('unique content')
(Path('demo_scan_dir/sub') / 'd.txt').write_text('shared content')
(Path('demo_scan_dir') / 'notes.md').write_text('not matched by pattern')
def discover_files(directory, pattern):
return list(Path(directory).rglob(pattern))
found = discover_files('demo_scan_dir', '*.txt')
print(sorted(str(p) for p in found))
Part 4: hash_file() Using hashlib#
import hashlib
def hash_file(path, chunk_size=8192):
hasher = hashlib.sha256()
with open(path, 'rb') as f:
while chunk := f.read(chunk_size):
hasher.update(chunk)
return hasher.hexdigest()
hash_a = hash_file('demo_scan_dir/a.txt')
hash_b = hash_file('demo_scan_dir/b.txt')
hash_c = hash_file('demo_scan_dir/c.txt')
print(hash_a == hash_b)
print(hash_a == hash_c)
Part 5: find_duplicates() Using collections#
from collections import defaultdict
def find_duplicates(records):
by_hash = defaultdict(list)
for record in records:
by_hash[record.digest].append(record.path)
return {digest: paths for digest, paths in by_hash.items() if len(paths) > 1}
records = [
FileRecord(path='demo_scan_dir/a.txt', size=15, digest=hash_a),
FileRecord(path='demo_scan_dir/b.txt', size=15, digest=hash_b),
FileRecord(path='demo_scan_dir/c.txt', size=15, digest=hash_c),
]
duplicates = find_duplicates(records)
print(len(duplicates))
print(sorted(list(duplicates.values())[0]))
Part 6: setup_logging() Using logging#
import logging
def setup_logging(verbose=False):
level = logging.DEBUG if verbose else logging.INFO
logger = logging.getLogger('dupe_finder')
logger.setLevel(level)
logger.handlers.clear()
handler = logging.StreamHandler()
handler.setFormatter(logging.Formatter('%(levelname)s: %(message)s'))
logger.addHandler(handler)
return logger
log = setup_logging(verbose=True)
log.debug('scanning started')
log.info(f'found {len(records)} files')
log.warning('this is just a demonstration warning')
Part 7: generate_json_report() Using json and datetime#
import json
from datetime import datetime, timezone
def generate_json_report(duplicates, output_path):
report = {
'generated_at': datetime.now(timezone.utc).isoformat(),
'duplicate_group_count': len(duplicates),
'groups': [{'digest': d, 'paths': p} for d, p in duplicates.items()],
}
with open(output_path, 'w') as f:
json.dump(report, f, indent=2)
return report
json_report = generate_json_report(duplicates, 'demo_report.json')
print(json_report['duplicate_group_count'])
print(os.path.exists('demo_report.json'))
Part 8: generate_csv_report() Using csv#
import csv
def generate_csv_report(duplicates, output_path):
with open(output_path, 'w', newline='') as f:
writer = csv.writer(f)
writer.writerow(['group_digest', 'path'])
for digest, paths in duplicates.items():
for path in paths:
writer.writerow([digest[:12], path])
generate_csv_report(duplicates, 'demo_report.csv')
with open('demo_report.csv') as f:
reader = csv.reader(f)
for row in reader:
print(row)
Part 9: main(argv) - Wiring Everything Together#
def main(argv):
parser = build_argparser()
args = parser.parse_args(argv)
log = setup_logging(verbose=True)
log.info(f'scanning {args.directory} for pattern {args.pattern}')
paths = discover_files(args.directory, args.pattern)
records = [FileRecord(path=str(p), size=p.stat().st_size, digest=hash_file(p)) for p in paths]
log.info(f'hashed {len(records)} files')
dupes = find_duplicates(records)
log.info(f'found {len(dupes)} duplicate groups')
generate_json_report(dupes, args.json_out)
generate_csv_report(dupes, args.csv_out)
return dupes
result = main(['demo_scan_dir', '--pattern', '*.txt', '--json-out', 'demo_main_report.json', '--csv-out', 'demo_main_report.csv'])
print(len(result))
Part 10: Verifying the Full Run and Reviewing the Series#
print(os.path.exists('demo_main_report.json'))
print(os.path.exists('demo_main_report.csv'))
with open('demo_main_report.json') as f:
final_report = json.load(f)
print(final_report['duplicate_group_count'])
print(sorted(final_report['groups'][0]['paths']))
print('Modules genuinely exercised in this single capstone tool:')
for m in MODULES_USED:
print(f' - {m}')
Wrap-Up: What You Learned, and Series Complete#
- This capstone combined argparse, pathlib, glob, hashlib, collections, logging, dataclasses, json, csv, and datetime, into one real, working command-line duplicate-file finder.
- A real tool is built from small, individually testable pieces: an arg parser, a discovery function, a hashing function, a grouping function, and report writers, composed together in one main(argv).
- Accepting an explicit argv list in main(), rather than relying on real sys.argv, keeps the whole tool genuinely testable without ever needing a real subprocess.
- dataclasses gave every discovered file a clean, typed record; collections.defaultdict made grouping duplicates a few clean lines.
- logging, not print, is how a real CLI tool reports its own progress; json and csv are how it reports its actual findings.
- This wraps up the entire twenty-five-part Python Standard Library Deep-Dive series.
- From os and pathlib through datetime, collections, itertools, functools, re, json, math, subprocess, threading, asyncio, logging, argparse, pickle, hashlib, contextlib, dataclasses, typing, unittest, shutil, io, uuid, abc, and inspect, every genuinely commonly-used core module has now been covered in real depth.
- Thanks for following along through all twenty-five videos.
Found this useful?
All lessons, notebooks and datasets here are free. If they helped you, a coffee keeps new lessons coming.



