NLTK & spaCy Cheat Sheet
This cheat sheet provides a quick reference for common NLTP tasks using NLTK and spaCy in Python.
I. NLTK (Natural Language Toolkit)
import nltk
from nltk.tokenize import word_tokenize, sent_tokenize
from nltk.corpus import stopwords
from nltk.stem import PorterStemmer, WordNetLemmatizer
from nltk.probability import FreqDist
nltk.download('punkt') # Download necessary data (run once)
nltk.download('stopwords')
nltk.download('wordnet')
nltk.download('omw-1.4') #Open Multilingual WordNet
text = "This is a sample sentence. It has multiple words."
# 1. Tokenization: Splitting text into words or sentences
words = word_tokenize(text) # Word tokenization
sentences = sent_tokenize(text) # Sentence tokenization
print(f"Words: {words}")
print(f"Sentences: {sentences}")
# 2. Stop Word Removal: Removing common words (e.g., "the", "a", "is")
stop_words = set(stopwords.words('english'))
filtered_words = [w for w in words if not w.lower() in stop_words]
print(f"Filtered words: {filtered_words}")
# 3. Stemming: Reducing words to their root form (e.g., "running" -> "run")
stemmer = PorterStemmer()
stemmed_words = [stemmer.stem(w) for w in filtered_words]
print(f"Stemmed words: {stemmed_words}")
# 4. Lemmatization: Reducing words to their dictionary form (lemma)
lemmatizer = WordNetLemmatizer()
lemmatized_words = [lemmatizer.lemmatize(w) for w in filtered_words]
print(f"Lemmatized words: {lemmatized_words}")
# 5. Frequency Distribution: Counting word occurrences
fdist = FreqDist(words)
print(fdist.most_common(5)) # Top 5 most frequent words
fdist.plot(20, cumulative=False) # Plot the frequency distribution
# 6. Part-of-Speech (POS) Tagging: Assigning grammatical tags to words
nltk.download('averaged_perceptron_tagger') # Download POS tagger data
tagged_words = nltk.pos_tag(words)
print(f"Tagged words: {tagged_words}") # Output: [('This', 'DT'), ('is', 'VBZ'), ...]
# 7. N-grams: Sequences of n words
from nltk.util import ngrams
bigrams = list(ngrams(words, 2))
print(f"Bigrams: {bigrams}")
# 8. Chunking (Shallow Parsing): Grouping words into phrases
# (Requires POS tagging first)
grammar = "NP: {<DT>?<JJ>*<NN>}" # Define a simple grammar for noun phrases
cp = nltk.RegexpParser(grammar)
tree = cp.parse(tagged_words)
print(tree)
#tree.draw() # Uncomment to draw the tree (requires graphviz)
# 9. Named Entity Recognition (NER) (using a pre-trained model)
nltk.download('maxent_ne_chunker')
nltk.download('words')
tree = nltk.ne_chunk(tagged_words)
print(tree)
#tree.draw() # Uncomment to draw the tree (requires graphviz)
II. spaCy
import spacy
# Load a spaCy language model (download if needed: python -m spacy download en_core_web_sm)
nlp = spacy.load("en_core_web_sm") # or en_core_web_trf for transformer model
text = "This is a sample sentence. It has multiple words."
doc = nlp(text)
# 1. Tokenization: Accessing tokens
for token in doc:
print(token.text, token.lemma_, token.pos_, token.tag_, token.dep_, token.shape_, token.is_alpha, token.is_stop)
# 2. Stop Word Removal: Checking if a token is a stop word
filtered_tokens = [token for token in doc if not token.is_stop]
print(f"Filtered tokens: {[token.text for token in filtered_tokens]}")
# 3. Lemmatization: Accessing the lemma
for token in doc:
print(token.text, token.lemma_)
# 4. Part-of-Speech (POS) Tagging: Accessing POS tags
for token in doc:
print(token.text, token.pos_, token.tag_)
# 5. Named Entity Recognition (NER): Accessing named entities
for ent in doc.ents:
print(ent.text, ent.label_)
# 6. Dependency Parsing: Accessing dependencies
for token in doc:
print(token.text, token.dep_, token.head.text)
# 7. Sentence Segmentation: Accessing sentences
for sent in doc.sents:
print(sent.text)
# 8. Word Vectors & Similarity: (If the model has word vectors)
nlp_lg = spacy.load('en_core_web_lg') # Load a larger model with vectors.
doc1 = nlp_lg("king")
doc2 = nlp_lg("queen")
print(doc1.similarity(doc2))
# 9. Text Classification (using a pre-trained pipeline or training your own)
# spaCy's TextCategorizer component can be used for text classification.
# See spaCy's documentation for details on training and using it.
# 10. Rule-based Matching:
from spacy.matcher import Matcher
matcher = Matcher(nlp)
pattern = [{"TEXT": "sample"}, {"TEXT": "sentence"}] # Define pattern
matcher.add("SampleSentence", [pattern]) # Add pattern to matcher
matches = matcher(doc) # Find matches
for match_id, start, end in matches:
string_id = nlp.vocab.strings[match_id] # Get string ID
span = doc[start:end] # The matched span
print(string_id, span.text)
Key Differences:
-
NLTK: More focused on a wider range of NLP tasks and algorithms, often used for research and learning. Provides more control over individual steps.
-
spaCy: Designed for production use, emphasizing speed and efficiency. Offers pre-trained models and pipelines optimized for common tasks. Generally faster and more accurate for tasks like NER and dependency parsing.
Remember to install the necessary libraries: pip install nltk spacy and download the required data/models. Refer to the official docs NLTK Official Documentation & spaCy Official Documentation for more advanced features and details. This cheat sheet is just a starting point!