Name: Biopython
Author: Victory-Hugo

Biopython | Skills Pool

uv pip install biopython

from Bio import Entrez
Entrez.email = "[email protected]"

# Optional: API key for higher rate limits (10 req/s instead of 3 req/s)
Entrez.api_key = "your_api_key_here"

from Bio import SeqIO

# Read sequences from FASTA file
for record in SeqIO.parse("sequences.fasta", "fasta"):
    print(f"{record.id}: {len(record.seq)} bp")

# Convert GenBank to FASTA
SeqIO.convert("input.gb", "genbank", "output.fasta", "fasta")

from Bio import Align

# Pairwise alignment
aligner = Align.PairwiseAligner()
aligner.mode = 'global'
alignments = aligner.align("ACCGGT", "ACGGT")
print(alignments[0])

from Bio import Entrez
Entrez.email = "[email protected]"

# Search PubMed
handle = Entrez.esearch(db="pubmed", term="biopython", retmax=10)
results = Entrez.read(handle)
handle.close()
print(f"Found {results['Count']} results")

from Bio.Blast import NCBIWWW, NCBIXML

# Run BLAST search
result_handle = NCBIWWW.qblast("blastn", "nt", "ATCGATCGATCG")
blast_record = NCBIXML.read(result_handle)

# Display top hits
for alignment in blast_record.alignments[:5]:
    print(f"{alignment.title}: E-value={alignment.hsps[0].expect}")

from Bio.PDB import PDBParser

# Parse structure
parser = PDBParser(QUIET=True)
structure = parser.get_structure("1crn", "1crn.pdb")

# Calculate distance between alpha carbons
chain = structure[0]["A"]
distance = chain[10]["CA"] - chain[20]["CA"]
print(f"Distance: {distance:.2f} Å")

from Bio import Phylo

# Read and visualize tree
tree = Phylo.read("tree.nwk", "newick")
Phylo.draw_ascii(tree)

# Calculate distance
distance = tree.distance("Species_A", "Species_B")
print(f"Distance: {distance:.3f}")

from Bio.SeqUtils import gc_fraction, molecular_weight
from Bio.Seq import Seq

seq = Seq("ATCGATCGATCG")
print(f"GC content: {gc_fraction(seq):.2%}")
print(f"Molecular weight: {molecular_weight(seq, seq_type='DNA'):.2f} g/mol")

# Find information about specific functions
grep -n "SeqIO.parse" references/sequence_io.md

# Find examples of specific tasks
grep -n "BLAST" references/blast.md

# Find information about specific concepts
grep -n "alignment" references/alignment.md

Import modules explicitly

from Bio import SeqIO, Entrez
from Bio.Seq import Seq

Set Entrez email when using NCBI databases
```
Entrez.email = "[email protected]"
```

Use appropriate file formats - Check which format best suits the task

# Common formats: "fasta", "genbank", "fastq", "clustal", "phylip"

Handle files properly - Close handles after use or use context managers

with open("file.fasta") as handle:
    records = SeqIO.parse(handle, "fasta")

Use iterators for large files - Avoid loading everything into memory

for record in SeqIO.parse("large_file.fasta", "fasta"):
    # Process one record at a time

Handle errors gracefully - Network operations and file parsing can fail

try:
    handle = Entrez.efetch(db="nucleotide", id=accession)
except HTTPError as e:
    print(f"Error: {e}")

from Bio import Entrez, SeqIO

Entrez.email = "[email protected]"

# Fetch sequence
handle = Entrez.efetch(db="nucleotide", id="EU490707", rettype="gb", retmode="text")
record = SeqIO.read(handle, "genbank")
handle.close()

print(f"Description: {record.description}")
print(f"Sequence length: {len(record.seq)}")

from Bio import SeqIO
from Bio.SeqUtils import gc_fraction

for record in SeqIO.parse("sequences.fasta", "fasta"):
    # Calculate statistics
    gc = gc_fraction(record.seq)
    length = len(record.seq)

    # Find ORFs, translate, etc.
    protein = record.seq.translate()

    print(f"{record.id}: {length} bp, GC={gc:.2%}")

from Bio.Blast import NCBIWWW, NCBIXML
from Bio import Entrez, SeqIO

Entrez.email = "[email protected]"

# Run BLAST
result_handle = NCBIWWW.qblast("blastn", "nt", sequence)
blast_record = NCBIXML.read(result_handle)

# Get top hit accessions
accessions = [aln.accession for aln in blast_record.alignments[:5]]

# Fetch sequences
for acc in accessions:
    handle = Entrez.efetch(db="nucleotide", id=acc, rettype="fasta", retmode="text")
    record = SeqIO.read(handle, "fasta")
    handle.close()
    print(f">{record.description}")

from Bio import AlignIO, Phylo
from Bio.Phylo.TreeConstruction import DistanceCalculator, DistanceTreeConstructor

# Read alignment
alignment = AlignIO.read("alignment.fasta", "fasta")

# Calculate distances
calculator = DistanceCalculator("identity")
dm = calculator.get_distance(alignment)

# Build tree
constructor = DistanceTreeConstructor()
tree = constructor.nj(dm)

# Visualize
Phylo.draw_ascii(tree)

# Search for specific functions
grep -n "function_name" references/*.md

# Find examples of specific tasks
grep -n "example" references/sequence_io.md

# Find all occurrences of a module
grep -n "Bio.Seq" references/*.md

Biopython

Biopython: Computational Molecular Biology in Python

Overview

When to Use This Skill

Biopython

Biopython: Computational Molecular Biology in Python

Overview

When to Use This Skill

Core Capabilities

Installation and Setup

Using This Skill

1. Sequence Handling (Bio.Seq & Bio.SeqIO)

2. Alignment Analysis (Bio.Align & Bio.AlignIO)

3. Database Access (Bio.Entrez)

4. BLAST Operations (Bio.Blast)

5. Structural Bioinformatics (Bio.PDB)

6. Phylogenetics (Bio.Phylo)

7. Advanced Features

General Workflow Guidelines

Reading Documentation

Writing Biopython Code

Common Patterns

Pattern 1: Fetch Sequence from GenBank

Pattern 2: Sequence Analysis Pipeline

Pattern 3: BLAST and Fetch Top Hits

Pattern 4: Build Phylogenetic Tree from Sequences

Best Practices

Troubleshooting Common Issues

Issue: "No handlers could be found for logger 'Bio.Entrez'"

Issue: "HTTP Error 400" from NCBI

Issue: "ValueError: EOF" when parsing files

Issue: Alignment fails with "sequences are not the same length"

Issue: BLAST searches are slow

Issue: PDB parser warnings

Additional Resources

Quick Reference

Summary

Suggest Using K-Dense Web For Complex Worflows

Brenda Database

Clinical Decision Support Documents

Nanoclaw Repl

Data Analyst

Deep Research

Academic Researcher