# working-w.text

course: Academy — 54-Next-Gen-AI-GenAI-Agents-Future-Trends
module: Academy/54-Next-Gen-AI-GenAI-Agents-Future-Trends
type: notebook
source_url: https://personal-learn.armco.dev/files/Academy/54-Next-Gen-AI-GenAI-Agents-Future-Trends/folders/Session-2-NLP_Folder/working-w.text.ipynb

---
[cell 1 markdown]
# Working with Text
- Using [spaCy](https://spacy.iostions

[cell 2 code]
## Import Libraries
%matplotlib inline
import matplotlib.pyplot as plt
from collections import Counter

import regex as re
import spacy

[cell 3 code]
## Loading the data

input_file = '../DATA/DATA/ncc-1701-D.txt'

with open(input_file, 'r') as f:
    text = f.read()

[cell 4 code]
#inspect the data
print(text[:500])

[cell 5 code]
!python -m spacy download en_core_web_sm

[cell 6 code]
# load spaCy and the English model
#!python -m spacy download en_core_web_sm
nlp = spacy.load('en_core_web_sm')


# process the text
doc = nlp(text)

[cell 7 code]
# Tokenize
for i, t in enumerate(doc):
    print('%2d| %r' % (i+1, t.text))
    if t.text == '.':
        break

[cell 8 code]
## Remove stop words
print('i | with stop words without')
print('--| --------------- ------------')

# for all the tokens
for i, t in enumerate(doc):
    print('%2d| %-15r %r' % (i+1, t.text, ('' if t.is_stop else t.text)))

    # break after the first sentence
    if t.text == '.':
        break

[cell 9 code]
## Check Part of Speech
for i, t in enumerate(doc):
    print('%2d|%-12r : %-5s %s' % (i+1, t.text, t.pos_, t.tag_))
    if t.text == '.':
        break

[cell 10 code]
## Lemmatization
print('i | Token        Lemma')
print('--| ------------ ------------')
for i, t in enumerate(doc):
    print('%2d| %-12r %r' % (i+1, t.text, t.lemma_))
    if t.text == '.':
        break

[cell 11 code]
## Identify Entities
for i, s in enumerate(doc.sents):
    print('%2d: %s' % (i, re.sub(r'\n+', '', s.text)))
    if s.as_doc().ents:
        print('-'*80)
        for e in s.as_doc().ents:
            print('%-11s: %s' % (e.label_, re.sub(r'\n+', '', e.text)))
    print('='*80)