SKILL.md
Digital Humanities Guide
A skill for applying computational and quantitative methods to humanities research. Covers text mining, network analysis, spatial humanities, and digital archival methods. Designed for researchers bridging traditional humanities with data-driven approaches.
Text Mining and Distant Reading
Corpus Preparation
import re
from collections import Counter
def prepare_corpus(texts: list[str], stopwords: set = None) -> list[list[str]]:
"""
Tokenize and clean a corpus of texts for analysis.
Args:
texts: List of raw text strings
stopwords: Set of words to remove
Returns:
List of tokenized, cleaned documents
"""
if stopwords is None:
stopwords = {'the', 'a', 'an', 'and', 'or', 'but', 'in', 'on',
'at', 'to', 'for', 'of', 'with', 'is', 'was', 'are'}
processed = []
for text in texts:
# Lowercase and remove punctuation
tokens = re.findall(r'\b[a-z]+\b', text.lower())
# Remove stopwords and short tokens
tokens = [t for t in tokens if t not in stopwords and len(t) > 2]
processed.append(tokens)
return processed
def compute_tfidf(corpus: list[list[str]]) -> dict:
"""Compute TF-IDF scores for term importance analysis."""
import math
n_docs = len(corpus)
# Document frequency
df = Counter()
for doc in corpus:
df.update(set(doc))
# TF-IDF per document
tfidf_scores = []
for doc in corpus:
tf = Counter(doc)
total = len(doc)
scores = {}
for term, count in tf.items():
tf_val = count / total
idf_val = math.log(n_docs / (1 + df[term]))
scores[term] = tf_val * idf_val
tfidf_scores.append(scores)
return tfidf_scores
Topic Modeling
Apply Latent Dirichlet Allocation (LDA) to discover thematic structures in large text corpora:
from gensim import corpora, models
def run_topic_model(corpus: list[list[str]], n_topics: int = 10,
passes: int = 15) -> models.LdaModel:
"""
Train an LDA topic model on a preprocessed corpus.
"""
dictionary = corpora.Dictionary(corpus)
dictionary.filter_extremes(no_below=5, no_above=0.5)
bow_corpus = [dictionary.doc2bow(doc) for doc in corpus]
lda_model = models.LdaModel(
bow_corpus,
num_topics=n_topics,
id2word=dictionary,
passes=passes,
random_state=42,
alpha='auto',
eta='auto'
)
return lda_model
# Print top words per topic
# for idx, topic in lda_model.print_topics(-1):
# print(f"Topic {idx}: {topic}")
