This source did not publish a separate summary. Review SKILL.md before using the skill.
SKILL.md
ToolUniverse CRISPR Screen Analysis
Comprehensive skill for analyzing CRISPR-Cas9 genetic screens to identify essential genes, synthetic lethal interactions, and therapeutic targets through robust statistical analysis and pathway enrichment.
Overview
CRISPR screens enable genome-wide functional genomics by systematically perturbing genes and measuring fitness effects. This skill provides an 8-phase workflow for:
Processing sgRNA count matrices
Quality control and normalization
Gene-level essentiality scoring (MAGeCK-like and BAGEL-like approaches)
Synthetic lethality detection
Pathway enrichment analysis
Drug target prioritization with DepMap integration
Integration with expression and mutation data
Core Workflow
Phase 1: Data Import & sgRNA Count Processing
Load sgRNA Count Matrix
import pandas as pd
import numpy as np
def load_sgrna_counts(counts_file):
"""
Load sgRNA count matrix from MAGeCK format or generic TSV.
Expected format:
sgRNA | Gene | Sample1 | Sample2 | Sample3 | ...
sgRNA_1 | BRCA1 | 1500 | 1200 | 1100 | ...
sgRNA_2 | BRCA1 | 1800 | 1500 | 1400 | ...
"""
counts = pd.read_csv(counts_file, sep='\t')
# Validate required columns
required_cols = ['sgRNA', 'Gene']
if not all(col in counts.columns for col in required_cols):
raise ValueError(f"Missing required columns: {required_cols}")
# Extract sample columns
sample_cols = [col for col in counts.columns if col not in ['sgRNA', 'Gene']]
# Create count matrix
count_matrix = counts[sample_cols].copy()
count_matrix.index = counts['sgRNA']
# Gene mapping
sgrna_to_gene = dict(zip(counts['sgRNA'], counts['Gene']))
metadata = {
'n_sgrnas': len(counts),
'n_genes': counts['Gene'].nunique(),
'n_samples': len(sample_cols),
'sample_names': sample_cols,
'sgrna_to_gene': sgrna_to_gene
}
return count_matrix, metadata
# Load counts
counts, meta = load_sgrna_counts("sgrna_counts.txt")
print(f"Loaded {meta['n_sgrnas']} sgRNAs targeting {meta['n_genes']} genes across {meta['n_samples']} samples")
def mageck_gene_scoring(lfc, sgrna_to_gene, method='rra'):
"""
Gene-level essentiality scoring using MAGeCK-like approach.
Methods:
- 'rra': Robust Rank Aggregation (identify genes with consistently low-ranking sgRNAs)
- 'mean': Simple mean LFC across sgRNAs
"""
# Create gene-level aggregation
gene_lfc = {}
for sgrna, gene in sgrna_to_gene.items():
if sgrna in lfc.index:
if gene not in gene_lfc:
gene_lfc[gene] = []
gene_lfc[gene].append(lfc[sgrna])
if method == 'rra':
# Simplified RRA: rank sgRNAs, calculate p-value for each gene
# based on whether its sgRNAs are enriched at the top (negative selection)
# or bottom (positive selection)
# Rank all sgRNAs by LFC
ranked_sgrnas = lfc.sort_values()
ranks = {sgrna: rank for rank, sgrna in enumerate(ranked_sgrnas.index, 1)}
gene_scores = {}
for gene, sgrna_list in gene_lfc.items():
# Get ranks for this gene's sgRNAs
gene_ranks = [ranks[sgrna] for sgrna in sgrna_list if sgrna in ranks]
if len(gene_ranks) > 0:
# Use mean rank as score (lower = more essential)
gene_scores[gene] = {
'score': np.mean(gene_ranks),
'n_sgrnas': len(gene_ranks),
'mean_lfc': np.mean([lfc[sg] for sg in sgrna_list if sg in lfc.index])
}
# Convert to DataFrame
gene_df = pd.DataFrame(gene_scores).T
gene_df['rank'] = gene_df['score'].rank()
elif method == 'mean':
# Simple mean LFC
gene_df = pd.DataFrame({
gene: {
'mean_lfc': np.mean(sgrna_lfcs),
'n_sgrnas': len(sgrna_lfcs),
'score': np.mean(sgrna_lfcs)
}
for gene, sgrna_lfcs in gene_lfc.items()
}).T
# Sort by essentiality (negative LFC = essential)
gene_df = gene_df.sort_values('mean_lfc')
return gene_df
# Gene-level scoring
gene_scores = mageck_gene_scoring(lfc, filtered_mapping, method='rra')
print(f"Top 10 essential genes:\n{gene_scores.head(10)[['mean_lfc', 'n_sgrnas']]}")
Bayes Factor Scoring (BAGEL-like)
def bagel_bayes_factor(lfc, sgrna_to_gene, essential_genes=None, nonessential_genes=None):
"""
BAGEL-like Bayes Factor calculation for gene essentiality.
Uses reference sets of known essential and non-essential genes to
calculate likelihood ratios.
"""
# Default reference gene sets (core essential genes)
if essential_genes is None:
essential_genes = ['RPL5', 'RPS6', 'POLR2A', 'PSMC2', 'PSMD14'] # Example
if nonessential_genes is None:
nonessential_genes = ['AAVS1', 'ROSA26', 'HPRT1'] # Example
# Get LFC distributions for reference sets
essential_lfc = [lfc[sg] for sg, g in sgrna_to_gene.items()
if g in essential_genes and sg in lfc.index]
nonessential_lfc = [lfc[sg] for sg, g in sgrna_to_gene.items()
if g in nonessential_genes and sg in lfc.index]
if len(essential_lfc) < 3 or len(nonessential_lfc) < 3:
print("Warning: Insufficient reference genes for BAGEL scoring")
return None
# Estimate distributions (simplified)
essential_mean, essential_std = np.mean(essential_lfc), np.std(essential_lfc)
nonessential_mean, nonessential_std = np.mean(nonessential_lfc), np.std(nonessential_lfc)
# Calculate Bayes Factor for each gene
gene_bf = {}
gene_lfc_map = {}
for sgrna, gene in sgrna_to_gene.items():
if sgrna in lfc.index:
if gene not in gene_lfc_map:
gene_lfc_map[gene] = []
gene_lfc_map[gene].append(lfc[sgrna])
for gene, sgrna_lfcs in gene_lfc_map.items():
mean_lfc = np.mean(sgrna_lfcs)
# Likelihood under essential distribution
from scipy.stats import norm
l_essential = norm.pdf(mean_lfc, essential_mean, essential_std)
# Likelihood under non-essential distribution
l_nonessential = norm.pdf(mean_lfc, nonessential_mean, nonessential_std)
# Bayes Factor (avoid division by zero)
bf = l_essential / (l_nonessential + 1e-10)
gene_bf[gene] = {
'bayes_factor': bf,
'mean_lfc': mean_lfc,
'n_sgrnas': len(sgrna_lfcs)
}
# Convert to DataFrame and sort
bf_df = pd.DataFrame(gene_bf).T
bf_df = bf_df.sort_values('bayes_factor', ascending=False)
return bf_df
# BAGEL scoring
bf_scores = bagel_bayes_factor(lfc, filtered_mapping)
if bf_scores is not None:
print(f"Top 10 by Bayes Factor:\n{bf_scores.head(10)}")
Phase 5: Synthetic Lethality Detection
Identify Context-Specific Essential Genes
def detect_synthetic_lethality(gene_scores_wildtype, gene_scores_mutant,
lfc_threshold=-1.0, rank_diff_threshold=100):
"""
Identify genes that are selectively essential in mutant context
(synthetic lethal interactions).
Compare essentiality scores between wildtype and mutant cell lines.
"""
# Merge scores
comparison = pd.merge(
gene_scores_wildtype[['mean_lfc', 'rank']],
gene_scores_mutant[['mean_lfc', 'rank']],
left_index=True,
right_index=True,
suffixes=('_wt', '_mut')
)
# Calculate differential essentiality
comparison['delta_lfc'] = comparison['mean_lfc_mut'] - comparison['mean_lfc_wt']
comparison['delta_rank'] = comparison['rank_wt'] - comparison['rank_mut']
# Identify synthetic lethal candidates
# (more essential in mutant, not essential in wildtype)
sl_candidates = comparison[
(comparison['mean_lfc_mut'] < lfc_threshold) & # Essential in mutant
(comparison['mean_lfc_wt'] > -0.5) & # Not essential in wildtype
(comparison['delta_rank'] > rank_diff_threshold) # Large rank change
].copy()
sl_candidates = sl_candidates.sort_values('delta_lfc')
return sl_candidates
# Example: Detect genes synthetic lethal with KRAS mutation
# (Requires running screens in both KRAS-mutant and wildtype cells)
# sl_hits = detect_synthetic_lethality(gene_scores_wt, gene_scores_kras_mut)
Query DepMap for Known Dependencies
def query_depmap_dependencies(gene_symbol):
"""
Query DepMap database for known gene dependencies.
ToolUniverse doesn't have direct DepMap tools, but we can use
STRING or literature tools to find dependency information.
"""
from tooluniverse import ToolUniverse
tu = ToolUniverse()
# Search literature for essentiality/dependency information
result = tu.run_one_function({
"name": "PubMed_search",
"arguments": {
"query": f'("{gene_symbol}"[Gene]) AND ("CRISPR screen" OR "gene essentiality" OR "DepMap")',
"max_results": 20
}
})
if 'data' in result and 'papers' in result['data']:
papers = result['data']['papers']
print(f"Found {len(papers)} papers on {gene_symbol} essentiality")
return papers
return []
# Example usage
# depmap_papers = query_depmap_dependencies("PRMT5")
Phase 6: Pathway Enrichment Analysis
Enrichment of Essential Genes
def enrich_essential_genes(gene_scores, top_n=100, databases=['KEGG_2021_Human', 'GO_Biological_Process_2021']):
"""
Perform pathway enrichment on top essential genes.
"""
from tooluniverse import ToolUniverse
tu = ToolUniverse()
# Get top essential genes (most negative LFC)
top_genes = gene_scores.head(top_n).index.tolist()
print(f"Enriching {len(top_genes)} top essential genes...")
# Run Enrichr
result = tu.run_one_function({
"name": "Enrichr_submit_genelist",
"arguments": {
"gene_list": top_genes,
"description": "CRISPR_screen_essential_genes"
}
})
if 'data' not in result or 'userListId' not in result['data']:
print("Failed to submit gene list to Enrichr")
return None
user_list_id = result['data']['userListId']
# Get enrichment results for each database
all_results = {}
for db in databases:
enrich_result = tu.run_one_function({
"name": "Enrichr_get_results",
"arguments": {
"userListId": user_list_id,
"backgroundType": db
}
})
if 'data' in enrich_result and db in enrich_result['data']:
all_results[db] = pd.DataFrame(enrich_result['data'][db])
print(f"{db}: {len(all_results[db])} enriched terms")
return all_results
# Run enrichment
# enrichment_results = enrich_essential_genes(gene_scores, top_n=100)
Phase 7: Drug Target Prioritization
Integrate with Expression & Mutation Data
def prioritize_drug_targets(gene_scores, expression_data=None, mutation_data=None):
"""
Prioritize CRISPR hits as drug targets based on:
1. Essentiality score (from CRISPR screen)
2. Expression level in disease vs normal (if provided)
3. Mutation frequency in tumors (if provided)
4. Druggability (query DGIdb)
"""
from tooluniverse import ToolUniverse
tu = ToolUniverse()
# Start with top essential genes
candidates = gene_scores.head(50).copy()
# Add expression data if provided
if expression_data is not None:
candidates = candidates.merge(expression_data, left_index=True, right_index=True, how='left')
# Add mutation data if provided
if mutation_data is not None:
candidates = candidates.merge(mutation_data, left_index=True, right_index=True, how='left')
# Query druggability for each gene
druggability_scores = {}
for gene in candidates.index[:20]: # Limit to top 20 to avoid rate limits
result = tu.run_one_function({
"name": "DGIdb_query_gene",
"arguments": {"gene_symbol": gene}
})
if 'data' in result and 'matchedTerms' in result['data']:
matches = result['data']['matchedTerms']
if len(matches) > 0:
# Count number of drug interactions
n_drugs = len(matches[0].get('interactions', []))
druggability_scores[gene] = n_drugs
else:
druggability_scores[gene] = 0
else:
druggability_scores[gene] = 0
candidates['n_drugs'] = pd.Series(druggability_scores)
# Calculate composite priority score
# (Normalize each component to 0-1 scale)
candidates['essentiality_norm'] = (candidates['mean_lfc'].min() - candidates['mean_lfc']) / \
(candidates['mean_lfc'].min() - candidates['mean_lfc'].max())
if 'log2fc' in candidates.columns:
candidates['expression_norm'] = (candidates['log2fc'] - candidates['log2fc'].min()) / \
(candidates['log2fc'].max() - candidates['log2fc'].min())
else:
candidates['expression_norm'] = 0
candidates['druggability_norm'] = candidates['n_drugs'] / (candidates['n_drugs'].max() + 1)
# Weighted composite score
candidates['priority_score'] = (
0.5 * candidates['essentiality_norm'] +
0.3 * candidates['expression_norm'] +
0.2 * candidates['druggability_norm']
)
# Sort by priority
candidates = candidates.sort_values('priority_score', ascending=False)
return candidates
# Prioritize targets
# drug_targets = prioritize_drug_targets(gene_scores, expression_data=rna_seq_results)
Query Existing Drugs for Top Targets
def find_drugs_for_targets(target_genes, max_per_gene=5):
"""
Find existing drugs targeting top candidate genes.
"""
from tooluniverse import ToolUniverse
tu = ToolUniverse()
drug_results = {}
for gene in target_genes[:10]: # Top 10 targets
print(f"Searching drugs for {gene}...")
# Query DGIdb
result = tu.run_one_function({
"name": "DGIdb_query_gene",
"arguments": {"gene_symbol": gene}
})
if 'data' in result and 'matchedTerms' in result['data']:
matches = result['data']['matchedTerms']
if len(matches) > 0:
interactions = matches[0].get('interactions', [])
drugs = []
for interaction in interactions[:max_per_gene]:
drugs.append({
'drug_name': interaction.get('drugName', 'Unknown'),
'interaction_type': interaction.get('interactionTypes', ['Unknown'])[0],
'source': interaction.get('source', 'Unknown')
})
drug_results[gene] = drugs
return drug_results
# Find drugs
# drug_candidates = find_drugs_for_targets(drug_targets.index.tolist())
Phase 8: Report Generation
Comprehensive CRISPR Screen Report
def generate_crispr_report(gene_scores, enrichment_results, drug_targets,
output_file="crispr_screen_report.md"):
"""
Generate comprehensive CRISPR screen analysis report.
"""
with open(output_file, 'w') as f:
f.write("# CRISPR Screen Analysis Report\n\n")
# Summary statistics
f.write("## Summary\n\n")
f.write(f"- **Total genes analyzed**: {len(gene_scores)}\n")
f.write(f"- **Essential genes** (LFC < -1): {(gene_scores['mean_lfc'] < -1).sum()}\n")
f.write(f"- **Non-essential genes** (LFC > -0.5): {(gene_scores['mean_lfc'] > -0.5).sum()}\n\n")
# Top 20 essential genes
f.write("## Top 20 Essential Genes\n\n")
f.write("| Rank | Gene | Mean LFC | sgRNAs | Score |\n")
f.write("|------|------|----------|--------|-------|\n")
for idx, (gene, row) in enumerate(gene_scores.head(20).iterrows(), 1):
f.write(f"| {idx} | {gene} | {row['mean_lfc']:.3f} | {int(row['n_sgrnas'])} | {row['score']:.2f} |\n")
f.write("\n")
# Pathway enrichment
if enrichment_results:
f.write("## Pathway Enrichment\n\n")
for db, results in enrichment_results.items():
f.write(f"### {db}\n\n")
f.write("| Term | P-value | Adjusted P-value | Genes |\n")
f.write("|------|---------|------------------|-------|\n")
for _, row in results.head(10).iterrows():
term = row.get('Term', 'Unknown')
pval = row.get('P-value', 1.0)
adj_pval = row.get('Adjusted P-value', 1.0)
genes = row.get('Genes', '')
f.write(f"| {term} | {pval:.2e} | {adj_pval:.2e} | {genes[:50]}... |\n")
f.write("\n")
# Drug target prioritization
if drug_targets is not None:
f.write("## Top Drug Target Candidates\n\n")
f.write("| Rank | Gene | Essentiality | Expression FC | Druggable | Priority Score |\n")
f.write("|------|------|--------------|---------------|-----------|----------------|\n")
for idx, (gene, row) in enumerate(drug_targets.head(10).iterrows(), 1):
ess = row['mean_lfc']
expr = row.get('log2fc', 0)
drugs = int(row.get('n_drugs', 0))
priority = row['priority_score']
f.write(f"| {idx} | {gene} | {ess:.3f} | {expr:.2f} | {drugs} | {priority:.3f} |\n")
f.write("\n")
# Methods
f.write("## Methods\n\n")
f.write("**sgRNA Processing**: MAGeCK-like robust rank aggregation\n\n")
f.write("**Normalization**: Median ratio normalization\n\n")
f.write("**Scoring**: Gene-level LFC aggregation with rank-based scoring\n\n")
f.write("**Enrichment**: Enrichr (KEGG, GO)\n\n")
f.write("**Druggability**: DGIdb v4.0\n\n")
print(f"Report saved to {output_file}")
return output_file
# Generate report
# report_file = generate_crispr_report(gene_scores, enrichment_results, drug_targets)