Files & data
Read and write files safely; parse JSON, CSV, and common text formats.
How to Split Files by Extension in Python
Group files in a folder by their file extension into a dictionary using pathlib.
from pathlib import Path
def split_files_by_extension(folder_path):
folder = Path(folder_path)
files_by_ext = {}
for file_path in folder.iterdir():
if file_path.is_file():
ext = file_path.suffix.lower() or "no_extension"
files_by_ext.setdefault(ext, []).append(file_path.na…
How to Split a Large File into Fixed-Size Parts in Python
Splits any binary or text file into multiple part files of a fixed byte size using Python's standard library.
import os
import math
def split_file(filepath, chunk_size_bytes):
"""Split a file into parts of fixed size (bytes). Creates part files in same directory."""
filepath = os.path.abspath(filepath)
basename = os.path.basename(filepath)
file_size = os.path.getsize(filepath)
num_parts = math.ceil(file_s…
How to Stream Large CSV Files in Python
Process a large CSV file in memory-efficient chunks using Python's csv module, yielding batches of rows instead of loading everything at once.
import csv
from pathlib import Path
def process_csv_in_chunks(file_path, chunk_size=1000):
"""Yield rows from a large CSV file in chunks without loading all into memory."""
with open(file_path, 'r', newline='') as f:
reader = csv.DictReader(f)
chunk = []
for row in reader:
…
How to Strip BOM When Reading UTF-8 Files in Python
Read a UTF-8 text file with Python's pathlib while automatically stripping the Byte Order Mark (BOM) so the first character isn't a hidden glyph.
from pathlib import Path
def read_text_without_bom(file_path):
"""Read a UTF-8 text file, stripping the BOM if present."""
return Path(file_path).read_text(encoding='utf-8-sig')
if __name__ == "__main__":
# Create a sample file with BOM for demonstration
sample_path = Path("sample_with_bom.txt")
…
How to Sum a CSV Column by Group in Python
This code reads a CSV string and sums a specified column for each unique value of a group key using the csv module and defaultdict.
import csv
from collections import defaultdict
from io import StringIO
def aggregate_csv(csv_data, group_key, sum_column):
totals = defaultdict(float)
reader = csv.DictReader(StringIO(csv_data))
for row in reader:
key = row[group_key]
totals[key] += float(row[sum_column])
return dict(t…
How to Sync Two Folders in Python (Lightweight Backup)
A Python script that synchronizes a source folder to a destination folder, copying new or updated files and removing files that no longer exist in the source.
import os
import shutil
import sys
from pathlib import Path
def sync_folders(src: Path, dst: Path):
"""Sync src folder to dst folder, copying missing/updated files."""
dst.mkdir(parents=True, exist_ok=True)
for src_path in src.rglob("*"):
relative = src_path.relative_to(src)
dst_path = ds…
How to Transcode a File from Latin-1 to UTF-8 in Python
Read a latin1-encoded text file and rewrite it as UTF-8 using Python's pathlib and encoding parameters.
from pathlib import Path
def transcode_to_utf8(input_path, output_path):
"""Read a latin1-encoded file and write it as UTF-8."""
source = Path(input_path)
target = Path(output_path)
with source.open(encoding='latin1') as infile:
content = infile.read()
with target.open('w', encod…
How to Use fcntl for Exclusive File Locking in Python
This code demonstrates how to acquire an exclusive advisory lock on a file using fcntl.flock with a non-blocking flag, simulate work, then release the lock.
import fcntl
import os
import tempfile
import time
def acquire_exclusive_lock(filepath):
fd = os.open(filepath, os.O_RDWR | os.O_CREAT)
try:
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
print(f"Exclusive lock acquired on {filepath}")
time.sleep(1) # Simulate work while holding the l…
How to Validate JSON Schema Shape in Python
Validate JSON data against a schema using manual checks for required fields, types, and constraints.
import json
from typing import Any, Dict
def validate_person_schema(data: Dict[str, Any]) -> bool:
"""Validate a person object against expected schema shape."""
if not isinstance(data, dict):
return False
# Required fields check
required_fields = {"name", "age", "email"}
if not requir…
How to Validate a JSON File in Python
A beginner-friendly Python helper that reads a JSON file, catches common errors, and returns a status dictionary.
import json
from pathlib import Path
def get_valid_json_data(file_path: str) -> dict:
file = Path(file_path)
if not file.exists():
return {"status": "error", "message": f"File not found: {file_path}"}
try:
data = json.loads(file.read_text())
except json.JSONDecodeError as e:
…
How to Walk a Directory Tree with os.walk in Python
A generator function that recursively walks a directory tree and yields every file path found using the os.walk generator.
import os
def walk_directory_tree(root_path: str):
"""Walk a directory tree and yield file paths using os.walk generator."""
for dirpath, dirnames, filenames in os.walk(root_path):
for filename in filenames:
yield os.path.join(dirpath, filename)
if __name__ == "__main__":
# Create a…
How to Watch a Directory for New Files in Python
Poll a directory at regular intervals and detect newly added files, printing each one as it appears.
import time
import os
from pathlib import Path
WATCH_DIR = Path("watched_files")
def watch_for_new_files(directory: Path, sleep_time: float = 1.0, max_iterations: int = 10):
"""Poll a directory for new files and print when one appears."""
directory.mkdir(exist_ok=True)
existing = set(os.listdir(directory…
How to Write Bytes to a File in Python with 'wb'
Write a bytearray buffer to a binary file using Python's open() in 'wb' mode, then read it back to confirm the data.
data = bytearray([0x48, 0x65, 0x6c, 0x6c, 0x6f, 0x20, 0x57, 0x6f, 0x72, 0x6c, 0x64])
with open("output.bin", "wb") as f:
f.write(data)
with open("output.bin", "rb") as f:
content = f.read()
print(f"Written {len(data)} bytes: {content}")
print(f"As string: {content.decode('ascii')}")
How to Write Simple XML Documents with ElementTree in Python
Create well-structured XML documents in memory using Python's built-in ElementTree module, complete with nested elements, attributes, and text content.
import xml.etree.ElementTree as ET
def create_xml_document():
# Create root element
root = ET.Element("catalog")
# Create a book element with attributes and children
book1 = ET.SubElement(root, "book", id="bk101")
ET.SubElement(book1, "author").text = "Gambardella, Matthew"
ET.SubElement(…
How to Write a Dict to a Pretty JSON File with Indent in Python
Serializes a Python dictionary to a readable JSON file using json.dump with indentation and sorted keys, then prints the file contents to stdout.
import json
from pathlib import Path
data = {
"name": "Python",
"version": 3.12,
"features": ["simple", "readable", "powerful"],
"nested": {"creator": "Guido van Rossum", "year": 1991}
}
output_path = Path("output.json")
with output_path.open("w", encoding="utf-8") as f:
json.dump(data, f, inden…
How to Write a List of Lines to a Text File Safely in Python
This code atomically writes a list of strings as lines to a text file using a temporary file and os.replace to prevent corruption.
from pathlib import Path
import tempfile
import os
def write_lines_safely(lines: list[str], filepath: str | Path) -> None:
"""Write lines to a text file atomically to avoid corruption."""
path = Path(filepath)
path.parent.mkdir(parents=True, exist_ok=True)
fd, temp_path = tempfile.mkstemp(dir=str…
How to check file data in Python
Check if a file exists and is a regular file, then return its name, size, line count, and first line.
def check_file_data(file_path):
from pathlib import Path
path = Path(file_path)
if not path.exists():
return f"File '{file_path}' does not exist."
if not path.is_file():
return f"'{file_path}' is not a regular file."
size = path.stat().st_size
lines = path.read_text(encodin…
How to resolve a symlink to its real path in Python with pathlib
Use Path.resolve() to turn a symlink path into its absolute target path, handling relative symlinks and eliminating symbolic links.
from pathlib import Path
def resolve_symlink(path):
p = Path(path)
return str(p.resolve())
if __name__ == "__main__":
# Create a symlink to demonstrate the resolution
target = Path("/tmp/real_target.txt")
target.write_text("hello")
link = Path("/tmp/my_link.txt")
try:
link.symlink…
How to write an INI config section with configparser in Python
Create an INI configuration file with sections using Python's configparser module and write it to disk.
import configparser
config = configparser.ConfigParser()
config["General"] = {
"host": "localhost",
"port": "8080",
"debug": "true"
}
config["Database"] = {
"name": "appdb",
"user": "admin",
"password": "secret"
}
with open("example.ini", "w") as file:
config.write(file)
with open("examp…
Join two CSV files on shared key column in Python
Merge rows from two CSV files by a common key column, outputting combined records to a new file.
import csv
def join_csv(file1, file2, key, output="joined.csv"):
# Read first CSV into dict keyed by the join column
with open(file1, newline="") as f1:
reader1 = csv.DictReader(f1)
data1 = {row[key]: row for row in reader1}
# Read second CSV and merge matching rows
with open(file2, n…
Merge Multiple PDF Files into One Document in Python
Combines multiple PDF files into a single PDF document using the PyPDF2 library's PdfMerger class.
import PyPDF2
def merge_pdfs(input_paths, output_path):
merger = PyPDF2.PdfMerger()
for path in input_paths:
merger.append(path)
merger.write(output_path)
merger.close()
print(f"Merged {len(input_paths)} PDFs into '{output_path}'.")
if __name__ == "__main__":
files = ["file1.pdf", "fi…
Move a file to an archive folder with shutil.move in Python
Move a file to an archive folder with shutil.move, creating the folder if needed, and return the new path.
from pathlib import Path
import shutil
def move_file_to_archive(source_file: str, archive_folder: str) -> Path:
"""Move a file to the archive folder, creating it if needed."""
src = Path(source_file)
archive = Path(archive_folder)
archive.mkdir(parents=True, exist_ok=True)
destination = archive / …
Normalize CSV Column Names to snake_case in Python
Convert CSV header names to snake_case using a regular expression and write the updated file in place.
import csv
import re
import sys
def to_snake_case(header):
header = re.sub(r"(?<=[a-z0-9])(?=[A-Z])", "_", header)
header = re.sub(r"[^a-zA-Z0-9]+", "_", header).strip("_").lower()
return header
def normalize_csv_headers(input_path, output_path=None):
with open(input_path, newline="", encoding="utf…
Parameterize SQL queries in Python to prevent SQL injection
Safely fetch users from a SQLite database using parameterized queries to prevent SQL injection attacks.
import sqlite3
def get_users_by_name(name):
"""Fetch users safely using parameterized query."""
conn = sqlite3.connect(':memory:')
cursor = conn.cursor()
# Create sample table and data
cursor.execute('CREATE TABLE users (id INTEGER, name TEXT)')
cursor.executemany('INSERT INTO users (name…
Browse by section
Each section groups closely related Python snippets.
Files & data — Python code examples
What you will find here
This page collects files & data snippets — short, copy-ready Python you can paste into our free online IDE and run without installing anything. Each sample includes a plain-English explanation and the full source code.
Samples vs tutorials and challenges
Samples are quick reference — one concept per page. For step-by-step teaching, use our Python tutorials. To test yourself, try quizzes or coding challenges. Clean up style with the Python formatter.