# working-w.text
course: Academy — 54-Next-Gen-AI-GenAI-Agents-Future-Trends
module: Academy/54-Next-Gen-AI-GenAI-Agents-Future-Trends
type: notebook
source_url: https://personal-learn.armco.dev/files/Academy/54-Next-Gen-AI-GenAI-Agents-Future-Trends/folders/Session-2-NLP_Folder/working-w.text.ipynb
---
[cell 1 markdown]
# Working with Text
- Using [spaCy](https://spacy.iostions
[cell 2 code]
## Import Libraries
%matplotlib inline
import matplotlib.pyplot as plt
from collections import Counter
import regex as re
import spacy
[cell 3 code]
## Loading the data
input_file = '../DATA/DATA/ncc-1701-D.txt'
with open(input_file, 'r') as f:
text = f.read()
[cell 4 code]
#inspect the data
print(text[:500])
[cell 5 code]
!python -m spacy download en_core_web_sm
[cell 6 code]
# load spaCy and the English model
#!python -m spacy download en_core_web_sm
nlp = spacy.load('en_core_web_sm')
# process the text
doc = nlp(text)
[cell 7 code]
# Tokenize
for i, t in enumerate(doc):
print('%2d| %r' % (i+1, t.text))
if t.text == '.':
break
[cell 8 code]
## Remove stop words
print('i | with stop words without')
print('--| --------------- ------------')
# for all the tokens
for i, t in enumerate(doc):
print('%2d| %-15r %r' % (i+1, t.text, ('' if t.is_stop else t.text)))
# break after the first sentence
if t.text == '.':
break
[cell 9 code]
## Check Part of Speech
for i, t in enumerate(doc):
print('%2d|%-12r : %-5s %s' % (i+1, t.text, t.pos_, t.tag_))
if t.text == '.':
break
[cell 10 code]
## Lemmatization
print('i | Token Lemma')
print('--| ------------ ------------')
for i, t in enumerate(doc):
print('%2d| %-12r %r' % (i+1, t.text, t.lemma_))
if t.text == '.':
break
[cell 11 code]
## Identify Entities
for i, s in enumerate(doc.sents):
print('%2d: %s' % (i, re.sub(r'\n+', '', s.text)))
if s.as_doc().ents:
print('-'*80)
for e in s.as_doc().ents:
print('%-11s: %s' % (e.label_, re.sub(r'\n+', '', e.text)))
print('='*80)