Files & data
Read and write files safely; parse JSON, CSV, and common text formats.
Build a Python Script That Detects and Deletes Empty Files Across Folders
A Python script that recursively finds and removes all zero-byte files across nested directories, returning a list of deleted paths.
import os
from pathlib import Path
def find_and_delete_empty_files(root_dir: str) -> list:
"""Find and delete all empty files under root_dir. Returns list of deleted paths."""
deleted = []
for file_path in Path(root_dir).rglob('*'):
if file_path.is_file() and file_path.stat().st_size == 0:
…
Build a Simple ETL Pipeline in Python
A simple ETL pipeline that reads JSON Lines, transforms records with filtering and normalization, and writes the result to JSON.
import json
from pathlib import Path
def read_input(file_path: Path) -> list[dict]:
"""Read JSON lines file into list of dicts."""
with file_path.open("r", encoding="utf-8") as f:
return [json.loads(line) for line in f if line.strip()]
def transform(records: list[dict]) -> list[dict]:
"""Transf…
Compress and Extract ZIP Files Programmatically in Python
Create a ZIP archive with in-memory files and extract its contents to a directory using Python's stdlib zipfile and pathlib modules.
import zipfile
from pathlib import Path
import tempfile
import os
def create_sample_zip(zip_path: str, files: dict) -> None:
"""Create a ZIP file containing the given files (name -> content mapping)."""
with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zf:
for filename, content in files.ite…
Extract a Single Member from a ZIP Archive in Python
Extract one specific file from a ZIP archive to an output directory using the standard zipfile and pathlib modules.
import zipfile
from pathlib import Path
def extract_single_member(zip_path: str, member_name: str, output_dir: str = ".") -> Path:
"""Extract a single member from a zip archive to the output directory."""
with zipfile.ZipFile(zip_path, "r") as archive:
archive.extract(member_name, output_dir)
retu…
How to Check Disk Free Space in Python with shutil.disk_usage
This Python script uses the standard library shutil.disk_usage to report total, used, and free disk space in bytes, plus a percentage usage figure.
import shutil
def check_disk_free_space(path="/"):
"""Return a tuple of total, used, and free disk space in bytes."""
usage = shutil.disk_usage(path)
return usage.total, usage.used, usage.free
if __name__ == "__main__":
total, used, free = check_disk_free_space()
print(f"Total: {total:,} bytes"…
How to Compress a String to Gzip Bytes in Python
Compress a string into gzip-compressed bytes entirely in memory using the standard library gzip module.
import gzip
def compress_to_gzip_bytes(data: str, encoding: str = "utf-8") -> bytes:
"""Compress a string to gzip-compressed bytes in memory."""
return gzip.compress(data.encode(encoding))
if __name__ == "__main__":
original = "Hello, world! " * 10
compressed = compress_to_gzip_bytes(original)
pr…
How to Decompress a gzip File in Python
This code provides a function to decompress a .gz file, writing the decompressed content to a new file and returning the text, using the gzip standard library module.
import gzip
from pathlib import Path
def decompress_gzip(filepath: str, output_path: str | None = None) -> str:
"""Decompress a .gz file and return the decompressed content."""
input_path = Path(filepath)
if output_path is None:
output_path = str(input_path.with_suffix(""))
with gzip.open…
How to Extract IP Address Counts from Access Logs in Python
Read a web server access log, count occurrences of each IP address using regex and Counter, and print the ranked results.
import re
from collections import Counter
from pathlib import Path
def extract_ip_counts(log_file_path):
ip_pattern = r'^(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})'
ip_counter = Counter()
with open(log_file_path, 'r') as file:
for line in file:
match = re.match(ip_pattern, line)
…
How to Load a YAML Subset in Python Without PyYAML
Parse a flat, key-value YAML file with the Python standard library (re and pathlib), handling comments, quotes, and inline comments while skipping nested structures.
import re
from pathlib import Path
def load_yaml_subset(path):
"""Load a flat YAML file (key: value) without external dependencies."""
data = {}
with open(path, 'r', encoding='utf-8') as f:
for line in f:
# Skip empty lines and comments
line = line.strip()
if no…
How to Sanitize Filenames in Python
Strip illegal filename characters and clean up names for safe filesystem use.
import re
from pathlib import Path
def sanitize_filename(filename: str, replacement: str = "_") -> str:
"""
Remove illegal characters from a filename.
Illegal characters: / \\ : * ? " < > |
Also strips leading/trailing spaces and dots.
"""
# Remove illegal characters
sanitized = re.su…
How to Serialize a Python Object to Pickle Bytes in Memory
Serialize a Python object to pickle bytes in memory with pickle.dumps, then deserialize it back with pickle.loads and verify the roundtrip.
import pickle
class Person:
def __init__(self, name, age, skills):
self.name = name
self.age = age
self.skills = skills
def main():
person = Person("Alice", 30, ["Python", "SQL", "Docker"])
# Serialize to bytes in memory
pickle_bytes = pickle.dumps(person)
print(…
How to Split a Large File into Fixed-Size Parts in Python
Splits any binary or text file into multiple part files of a fixed byte size using Python's standard library.
import os
import math
def split_file(filepath, chunk_size_bytes):
"""Split a file into parts of fixed size (bytes). Creates part files in same directory."""
filepath = os.path.abspath(filepath)
basename = os.path.basename(filepath)
file_size = os.path.getsize(filepath)
num_parts = math.ceil(file_s…
How to Strip BOM When Reading UTF-8 Files in Python
Read a UTF-8 text file with Python's pathlib while automatically stripping the Byte Order Mark (BOM) so the first character isn't a hidden glyph.
from pathlib import Path
def read_text_without_bom(file_path):
"""Read a UTF-8 text file, stripping the BOM if present."""
return Path(file_path).read_text(encoding='utf-8-sig')
if __name__ == "__main__":
# Create a sample file with BOM for demonstration
sample_path = Path("sample_with_bom.txt")
…
Merge Multiple PDF Files into One Document in Python
Combines multiple PDF files into a single PDF document using the PyPDF2 library's PdfMerger class.
import PyPDF2
def merge_pdfs(input_paths, output_path):
merger = PyPDF2.PdfMerger()
for path in input_paths:
merger.append(path)
merger.write(output_path)
merger.close()
print(f"Merged {len(input_paths)} PDFs into '{output_path}'.")
if __name__ == "__main__":
files = ["file1.pdf", "fi…
Split CSV Files into Smaller Chunks in Python
Splits a large CSV file into multiple smaller chunk files, preserving the header row in each chunk.
import csv
import os
def split_csv(input_file, chunk_size=1000, output_prefix="chunk"):
"""Split a large CSV file into smaller chunks."""
with open(input_file, 'r', newline='') as infile:
reader = csv.reader(infile)
header = next(reader)
file_count = 1
row_count = 0
…
Sync only changed files between two folders in Python
This code compares two folders and copies only the new or modified files from source to destination, skipping unchanged ones by comparing SHA-256 hashes.
import hashlib
from pathlib import Path
import shutil
def file_hash(path: Path, chunk_size: int = 8192) -> str:
hasher = hashlib.sha256()
with path.open("rb") as f:
for chunk in iter(lambda: f.read(chunk_size), b""):
hasher.update(chunk)
return hasher.hexdigest()
def sync_files(src: s…
Browse by section
Each section groups closely related Python snippets.
Files & data — Python code examples
What you will find here
This page collects files & data snippets — short, copy-ready Python you can paste into our free online IDE and run without installing anything. Each sample includes a plain-English explanation and the full source code.
Samples vs tutorials and challenges
Samples are quick reference — one concept per page. For step-by-step teaching, use our Python tutorials. To test yourself, try quizzes or coding challenges. Clean up style with the Python formatter.