Files & data
Read and write files safely; parse JSON, CSV, and common text formats.
Convert All Markdown Files in a Folder to HTML in Python
Batch convert every .md file in a folder to .html using the `markdown` library with the 'extra' extensions.
import os
import markdown
from pathlib import Path
def convert_md_folder_to_html(input_folder="markdown_files", output_folder="html_pages"):
input_path = Path(input_folder)
output_path = Path(output_folder)
output_path.mkdir(exist_ok=True)
for md_file in input_path.glob("*.md"):
with open…
Extract Hyperlinks from Word Documents in Python
Parses a .docx file using Python's standard library to extract every hyperlink's display text and target URL.
import zipfile
from pathlib import Path
import xml.etree.ElementTree as ET
def extract_hyperlinks_from_docx(filepath: str) -> list[dict]:
"""
Extract all hyperlinks from a .docx file.
Returns a list of dicts with 'text' and 'target' keys.
"""
hyperlinks = []
with zipfile.ZipFile(Path(filepath)…
Find Duplicate Web Pages by Content Similarity in Python
Compute SHA-256 hashes of file contents to detect and report duplicate HTML pages or any files in a directory.
import hashlib
import os
from collections import defaultdict
def get_file_hash(filepath):
"""Compute SHA-256 hash of file contents."""
sha256 = hashlib.sha256()
with open(filepath, 'rb') as f:
for chunk in iter(lambda: f.read(4096), b''):
sha256.update(chunk)
return sha256.hexdiges…
How to Find HTML Elements by Tag, Class, ID, CSS Selector, and Attribute in BeautifulSoup
Parse an HTML string with BeautifulSoup and demonstrate five distinct ways to locate elements: by tag name, by class, by ID, by CSS selector, and by attribute.
from bs4 import BeautifulSoup
html_content = """
<html><body>
<h1 id="title" class="heading">Hello World</h1>
<p class="content">First paragraph</p>
<p class="content special">Second paragraph</p>
<a href="https://example.com" class="link">Click here</a>
<div id="footer">
<p>© 2024</p>
…
How to Load a YAML Subset in Python Without PyYAML
Parse a flat, key-value YAML file with the Python standard library (re and pathlib), handling comments, quotes, and inline comments while skipping nested structures.
import re
from pathlib import Path
def load_yaml_subset(path):
"""Load a flat YAML file (key: value) without external dependencies."""
data = {}
with open(path, 'r', encoding='utf-8') as f:
for line in f:
# Skip empty lines and comments
line = line.strip()
if no…
How to Parse XML Attributes into a Flat Dictionary in Python
Parses XML elements and attributes using ElementTree, building a flat dictionary keyed by element attributes.
import xml.etree.ElementTree as ET
xml_data = """<root>
<book id="1" category="fiction" price="9.99">
<title>The Catcher</title>
</book>
<book id="2" category="nonfiction" price="12.50">
<title>Deep Learning</title>
</book>
</root>"""
def parse_xml_attributes(xml_string):
root = E…
How to Scrape Headlines from a News Website Using Beautiful Soup in Python
Scrape headline text from a news website using requests and Beautiful Soup with a CSS selector.
import requests
from bs4 import BeautifulSoup
def scrape_headlines(url: str, selector: str) -> list:
"""
Scrape headlines from a news website using Beautiful Soup.
Args:
url: The URL of the news website.
selector: CSS selector for headline elements.
Returns:
List of h…
How to Write Simple XML Documents with ElementTree in Python
Create well-structured XML documents in memory using Python's built-in ElementTree module, complete with nested elements, attributes, and text content.
import xml.etree.ElementTree as ET
def create_xml_document():
# Create root element
root = ET.Element("catalog")
# Create a book element with attributes and children
book1 = ET.SubElement(root, "book", id="bk101")
ET.SubElement(book1, "author").text = "Gambardella, Matthew"
ET.SubElement(…
How to resolve a symlink to its real path in Python with pathlib
Use Path.resolve() to turn a symlink path into its absolute target path, handling relative symlinks and eliminating symbolic links.
from pathlib import Path
def resolve_symlink(path):
p = Path(path)
return str(p.resolve())
if __name__ == "__main__":
# Create a symlink to demonstrate the resolution
target = Path("/tmp/real_target.txt")
target.write_text("hello")
link = Path("/tmp/my_link.txt")
try:
link.symlink…
Read an XML File with xml.etree.ElementTree in Python
Parse an XML file and print its root and child elements using the standard library's xml.etree.ElementTree module.
import xml.etree.ElementTree as ET
def read_xml_file(file_path):
"""Read an XML file and print its structure."""
tree = ET.parse(file_path)
root = tree.getroot()
print(f"Root element: {root.tag}")
for child in root:
print(f"Child element: {child.tag}, text: {child.text}")
if __name__ ==…
Scrape HTML Tables and Convert Them to CSV Using Beautiful Soup in Python
Scrape a Wikipedia table with Beautiful Soup and write the data to a CSV file using the csv module.
import requests
from bs4 import BeautifulSoup
import csv
url = "https://en.wikipedia.org/wiki/List_of_countries_by_GDP_(nominal)"
response = requests.get(url)
soup = BeautifulSoup(response.text, 'html.parser')
tables = soup.find_all('table', {'class': 'wikitable'})
if tables:
target_table = tables[2]
rows =…
Browse by section
Each section groups closely related Python snippets.
Files & data — Python code examples
What you will find here
This page collects files & data snippets — short, copy-ready Python you can paste into our free online IDE and run without installing anything. Each sample includes a plain-English explanation and the full source code.
Samples vs tutorials and challenges
Samples are quick reference — one concept per page. For step-by-step teaching, use our Python tutorials. To test yourself, try quizzes or coding challenges. Clean up style with the Python formatter.