Automation & scripting
CLI tools, scheduled jobs, filesystem tasks, and glue scripts that save time.
Build a Complete Website Sitemap Generator Without External Services
Crawl a website recursively using only Python's standard library to generate a structured sitemap of internal links.
import json
from urllib.parse import urlparse, urljoin
from collections import deque
import urllib.request
import urllib.error
import re
from html.parser import HTMLParser
class SitemapParser(HTMLParser):
def __init__(self, base_url):
super().__init__()
self.base_url = base_url
self.links …
Build a Python Tool to Find All API Endpoints on a Website
A Python script that crawls a website, searches for common API endpoint patterns in HTML and JavaScript, and returns all discovered public API URLs.
import re
import requests
from urllib.parse import urljoin, urlparse
from collections import deque
def find_api_endpoints(base_url, max_pages=10):
visited = set()
queue = deque([base_url])
api_endpoints = set()
api_patterns = [
r'/api/[a-zA-Z0-9_/-]+',
r'/v[0-9]+/[a-zA-Z0-9_/-]+',…
Build a Website Accessibility Scanner Using Python
Scans a webpage for common accessibility issues like missing alt text, headings, labels, and landmarks using only Python.
import requests
from urllib.parse import urljoin
from html.parser import HTMLParser
import re
class AccessibilityParser(HTMLParser):
def __init__(self):
super().__init__()
self.images_without_alt = []
self.missing_headings = True
self.has_main_tag = False
self.label_for_inp…
Build an RSS feed from markdown blog posts in Python
Scans a folder of markdown files, extracts titles, dates, and excerpts, and generates a valid RSS 2.0 XML feed.
import re
from pathlib import Path
from xml.etree.ElementTree import Element, SubElement, tostring
from datetime import datetime, timezone
from xml.dom import minidom
def build_rss(blog_dir, site_url="https://example.com"):
feed = Element("rss", version="2.0")
channel = SubElement(feed, "channel")
SubElem…
Convert DOCX to Text by Unzipping XML in Python
Extract plain text from a .docx file by unzipping the container and parsing word/document.xml with regex, using only Python's standard library.
import zipfile
import re
from pathlib import Path
def docx_to_text_unzip_xml(docx_path: str) -> str:
"""Extract plain text from a .docx file by unzipping and parsing document.xml."""
docx_path = Path(docx_path)
if not docx_path.exists():
raise FileNotFoundError(f"File not found: {docx_path}")
…
Convert HTML Tables to Excel Reports in Python
Convert HTML tables into formatted Excel reports using BeautifulSoup and Pandas with auto-adjusted column widths.
import pandas as pd
from bs4 import BeautifulSoup
from pathlib import Path
def html_table_to_excel(html_file: str, excel_file: str) -> None:
"""Convert HTML table to formatted Excel report."""
with open(html_file, 'r', encoding='utf-8') as f:
html_content = f.read()
soup = BeautifulSoup(html_…
Convert Markdown to HTML in Python (Batch)
Convert every Markdown file in a directory to HTML with the Python markdown library, saving each result with an .html extension.
import markdown
from pathlib import Path
def convert_md_to_html(source_dir: str, dest_dir: str) -> list[str]:
src = Path(source_dir)
dst = Path(dest_dir)
dst.mkdir(parents=True, exist_ok=True)
converted_files = []
for md_file in src.glob("*.md"):
html_content = markdown.markdown(md_file.…
Create a Python Script That Detects Website Technology Stack Automatically
This script sends an HTTP request to a URL and inspects headers and HTML content to identify technologies like servers, frameworks, and JavaScript libraries.
import requests
from re import search
def detect_tech_stack(url):
tech_stack = []
try:
response = requests.get(url, timeout=5, headers={'User-Agent': 'Mozilla/5.0'})
headers = response.headers
html = response.text.lower() if response.text else ''
# Check server header
…
Discover RSS Feeds From Any Website in Python
Scrape a website's HTML to automatically find all linked RSS or Atom feed URLs using requests, BeautifulSoup, and regex.
import requests
import re
from urllib.parse import urljoin, urlparse
from bs4 import BeautifulSoup
def discover_rss_feeds(url):
"""Discover all RSS/Atom feeds linked from a given website."""
try:
headers = {'User-Agent': 'Mozilla/5.0 (compatible; RSSDiscovery/1.0)'}
response = requests.get(url…
Extract Every Open Graph and Social Media Meta Tag from Web Pages in Python
A Python script that fetches a webpage and extracts all Open Graph, Twitter Card, Facebook, and Article meta tags using the standard library HTML parser.
from html.parser import HTMLParser
import re
from urllib.request import urlopen
from urllib.parse import urlparse
class MetaExtractor(HTMLParser):
def __init__(self):
super().__init__()
self.meta_tags = []
def handle_starttag(self, tag, attrs):
if tag == 'meta':
attrs_…
Fetch weather API mock and write dashboard HTML in Python
This script fetches a mock weather API response as a Python dict, builds a simple HTML dashboard, writes it to a file, and prints both the file path and JSON payload.
from datetime import datetime
import json
import os
def fetch_weather_mock(city: str) -> dict:
"""Return a mock weather payload for a given city."""
return {
"city": city,
"temperature_c": 21.5,
"condition": "Partly Cloudy",
"humidity": 58,
"wind_kph": 12.3,
"u…
How to Bump Version in pyproject.toml Using Regex in Python
Updates the version field in a pyproject.toml file using a regex substitution with the Python standard library.
import re
from pathlib import Path
def bump_version(pyproject_path: str, new_version: str) -> None:
"""Update version in pyproject.toml using regex."""
path = Path(pyproject_path)
content = path.read_text()
# Match version = "x.y.z" (simple or PEP 440 with pre-release)
pattern = r'^version\s*=\s*…
How to Detect Unused Images in a Project with Python
A Python script that scans a website project folder, identifies all image files, and checks HTML/CSS/JS files to find which images are never referenced.
import os
import re
from pathlib import Path
def find_unused_images(project_path):
image_exts = {'.png', '.jpg', '.jpeg', '.gif', '.svg', '.webp'}
used_images = set()
all_images = set()
# Find all image files
for root, _, files in os.walk(project_path):
for file in files:
…
How to apply Kubernetes YAML files from a folder in Python
Uses the Kubernetes Python client to apply all YAML manifests in a directory, with sorted processing and per-file error handling.
import os
import yaml
from kubernetes import client, config
from kubernetes.utils import create_from_yaml
def apply_yaml_folder(folder_path):
"""Apply all YAML files in a folder using the Kubernetes mock client."""
# Load mock configuration
config.load_kube_config()
k8s_client = client.ApiClient()
…
Scrape HTML Tables in Python with html.parser
Extract data from HTML tables using Python's built-in html.parser module, without third-party dependencies, by overriding callback methods to track table, row, and cell states.
import html.parser
from urllib.request import urlopen
class TableParser(html.parser.HTMLParser):
def __init__(self):
super().__init__()
self.in_table = False
self.in_row = False
self.in_cell = False
self.current_cell = []
self.rows = []
self.row = []
d…
Browse by section
Each section groups closely related Python snippets.
Automation & scripting — Python code examples
What you will find here
This page collects automation & scripting snippets — short, copy-ready Python you can paste into our free online IDE and run without installing anything. Each sample includes a plain-English explanation and the full source code.
Samples vs tutorials and challenges
Samples are quick reference — one concept per page. For step-by-step teaching, use our Python tutorials. To test yourself, try quizzes or coding challenges. Clean up style with the Python formatter.