★
RNN在NLP中的3个小应用(PyTorch实现)
自然语言处理( Natural Language Processing, NLP)是计算机科学领域与人工智能领域中的一个重要方向。
前提:GPU版PyTorch已安装
查看方法:
import torch
print(torch.__version__) # 查看Pytorch版本
# 1.7.1
print(torch.cuda.is_available()) # 验证GPU版是否可用
# True
1 词嵌入 - PyTorch实现
import torch
from torch import nn
from torch.autograd import Variable
word_to_ix = {'hello': 0, 'world': 1}
embeds = nn.Embedding(2, 5)
hello_idx = torch.LongTensor([word_to_ix['hello']])
hello_idx = Variable(hello_idx)
hello_embed = embeds(hello_idx)
print(hello_embed)
输出结果
tensor([[ 1.0135, -0.2385, 0.2811, 0.3737, 0.8869]],
grad_fn=<EmbeddingBackward>)
2 单词预测 - N Gram语言模型
词嵌入是如何更新的,以及它如何结合N Gram语言模型进行预测
马尔可夫假设:某个待预测单词只与所在位置前面的几个单词有关系,并不是和前面所有的词都有关系
条件概率的传统解决方法:统计语料中每个单词出行的频率,据此来估计这个条件概率(根据贝叶斯定理来估计这个条件概率)
使用词嵌入的方法:直接在语料中计算这个条件概率,然后最大化条件概率从而优化词向量,据此进行预测。
即用词嵌入对估计条件概率的方法进行代替,然后使用 RNN 进行条件概率的计算,然后最大化这个条件概率不仅修改词嵌入,同时能够使得模型可以依据计算的条件概率对其中的一个单词进行预测。
import torch
from torch import nn
import torch.nn.functional as F
from torch.autograd import Variable
# CONTEXT_SIZE 表示我们希望由前面几个单词来预测这个单词,这里使用两个单词,EMBEDDING_DIM 表示词嵌入的维度。
CONTEXT_SIZE = 2
EMBEDDING_DIM = 10
# We will use Shakespeare Sonnet 2
test_sentence = """When forty winters shall besiege thy brow,
And dig deep trenches in thy beauty's field,
Thy youth's proud livery so gazed on now,
Will be a totter'd weed of small worth held:
Then being asked, where all thy beauty lies,
Where all the treasure of thy lusty days;
To say, within thine own deep sunken eyes,
Were an all-eating shame, and thriftless praise.
How much more praise deserv'd thy beauty's use,
If thou couldst answer 'This fair child of mine
Shall sum my count, and make my old excuse,'
Proving his beauty by succession thine!
This were to be new made when thou art old,
And see thy blood warm when thou feel'st it cold.""".split()
Step1 建立训练集
# 需要将单词分为3个组,每个组前两个作为传入的数据,而最后一个作为预测的结果
trigram = [((test_sentence[i], test_sentence[i+1]), test_sentence[i+2]) for i in range(len(test_sentence) - 2)]
变量查看
# 总的数据量
len(trigram) # 113
变量查看
# 取出第一个数据看看
trigram[0] # (('When', 'forty'), 'winters')
Step2 将每个单词编码
# 将每个单词编码,即用数字来表示每个单词,只有这样猜能够传入nn.Embedding得到词向量
vocb = set(test_sentence) # 通过set将重复的单词去掉
word_to_idx = {word: i for i, word in enumerate(vocb)}
idx_to_word = {word_to_idx[word]: word for word in word_to_idx}
变量查看
idx_to_word # 可以看到每个词都对应一个数字,且这里的单词都各不相同
变量查看 -> Output
{0: 'on',
1: 'answer',
2: 'sunken',
3: 'livery',
4: 'Proving',
5: 'new',
6: 'old',
7: "feel'st",
8: 'couldst',
9: 'This',
10: 'now,',
11: 'proud',
12: 'To',
13: 'a',
14: "deserv'd",
15: 'it',
16: 'thine!',
17: 'when',
18: 'count,',
19: 'an',
20: 'treasure',
21: 'to',
22: "'This",
23: 'How',
24: 'field,',
25: 'being',
26: 'fair',
27: 'be',
28: 'more',
29: 'see',
30: 'all-eating',
31: 'in',
32: 'lies,',
33: 'dig',
34: 'make',
35: 'If',
36: 'beauty',
37: 'thriftless',
38: 'small',
39: "excuse,'",
40: 'were',
41: 'And',
42: 'of',
43: 'deep',
44: 'and',
45: "beauty's",
46: "youth's",
47: 'days;',
48: 'gazed',
49: 'shame,',
50: "totter'd",
51: 'shall',
52: 'lusty',
53: 'thou',
54: 'When',
55: 'succession',
56: 'the',
57: 'his',
58: 'forty',
59: 'Will',
60: 'Shall',
61: 'use,',
62: 'warm',
63: 'thy',
64: 'winters',
65: 'weed',
66: 'besiege',
67: 'made',
68: 'asked,',
69: 'Then',
70: 'cold.',
71: 'brow,',
72: 'art',
73: 'say,',
74: 'child',
75: 'where',
76: 'praise.',
77: 'own',
78: 'much',
79: 'within',
80: 'sum',
81: 'trenches',
82: 'mine',
83: 'Thy',
84: 'blood',
85: 'Were',
86: 'by',
87: 'my',
88: 'thine',
89: 'praise',
90: 'old,',
91: 'all',
92: 'worth',
93: 'Where',
94: 'eyes,',
95: 'held:',
96: 'so'}
Step3 定义 N Gram 模型
模型的输入就是前面的两个词,输出就是预测单词的概率
# 原书代码 + 小修 by tao
class NgramModel(nn.Module):
def __init__(self, vocb_size, context_size=CONTEXT_SIZE, n_dim=EMBEDDING_DIM):
super(NgramModel, self).__init__()
self.n_word = vocb_size
self.embedding = nn.Embedding(self.n_word, n_dim)
self.linear1 = nn.Linear(context_size * n_dim, 128)
self.linear2 = nn.Linear(128, self.n_word)
def forward(self, x):
emb = self.embedding(x)
emb = emb.view(1, -1)
out = self.linear1(emb)
out = F.relu(out)
out = self.linear2(out)
return out
# log_prob = F.log_softmax(out)
# return log_prob
作者代码 v2.0更新版
# 定义 N Gram模型
# 模型的输入就是前面的两个词,输出就是预测单词的概率
class n_gram(nn.Module):
def __init__(self, vocab_size, context_size=CONTEXT_SIZE, n_dim=EMBEDDING_DIM):
super(n_gram, self).__init__()
self.embed = nn.Embedding(vocab_size, n_dim)
self.classify = nn.Sequential(
nn.Linear(context_size * n_dim, 128),
nn.ReLU(True),
nn.Linear(128, vocab_size)
)
def forward(self, x):
voc_embed = self.embed(x) # 得到词嵌入
voc_embed = voc_embed.view(1, -1) # 将两个词向量拼在一起
out = self.classify(voc_embed)
return out
Step4 开始训练
最后我们输出就是条件概率,相当于是一个分类问题,我们可以使用交叉熵来方便地衡量误差
net = NgramModel(len(word_to_idx))
criterion = nn.CrossEntropyLoss()
optimizer = torch.optim.SGD(net.parameters(), lr=1e-2, weight_decay=1e-5)
for e in range(100):
train_loss = 0
for word, label in trigram: # 使用前 100 个作为训练集
word = Variable(torch.LongTensor([word_to_idx[i] for i in word])) # 将两个词作为输入
label = Variable(torch.LongTensor([word_to_idx[label]]))
# 前向传播
out = net(word)
loss = criterion(out, label)
train_loss += loss.item()
# 反向传播
optimizer.zero_grad()
loss.backward()
optimizer.step()
if (e + 1) % 20 == 0:
print('epoch: {}, Loss: {:.6f}'.format(e + 1, train_loss / len(trigram)))
输出结果
epoch: 20, Loss: 0.738774
epoch: 40, Loss: 0.140150
epoch: 60, Loss: 0.089768
epoch: 80, Loss: 0.072926
epoch: 100, Loss: 0.064004
Step5 模型测试
net = net.eval()
# 测试一下结果
word, label = trigram[19]
print('input: {}'.format(word))
print('label: {}'.format(label))
print()
word = Variable(torch.LongTensor([word_to_idx[i] for i in word]))
out = net(word)
pred_label_idx = out.max(1)[1].item()
predict_word = idx_to_word[pred_label_idx]
print('real word is {}, predicted word is {}'.format(label, predict_word))
输出结果 例1
input: ('so', 'gazed')
label: on
real word is on, predicted word is on
word, label = trigram[75]
print('input: {}'.format(word))
print('label: {}'.format(label))
print()
word = Variable(torch.LongTensor([word_to_idx[i] for i in word]))
out = net(word)
pred_label_idx = out.max(1)[1].item()
predict_word = idx_to_word[pred_label_idx]
print('real word is {}, predicted word is {}'.format(label, predict_word))
输出结果 例2
input: ("'This", 'fair')
label: child
real word is child, predicted word is child
Step6 结果分析
结果: 在训练集上基本能够预测准确;
可能存在的问题: 不过这里样本太少,特别容易过拟合。
3 词性预测/判断 - 基于LSTM
前面两部分主要介绍词嵌入 & N Gram模型,接下来,结合 1 & 2,利用RNN(借助LSTM)来进行词性预测/判断
分析:
一个相同的单词可以表示两种不同的词性,比如 book 既可以表示名词,也可以表示动词,所以到底这个词是什么词性需要结合前后文来具体判断。
对于一个单词,会有这不同的词性,首先能够根据一个单词的后缀来初步判断,比如 -ly 这种后缀,很大概率是一个副词。
结论:可以使用 lstm 模型来进行预测
首先对于一个单词,可以将其看作一个序列,比如 apple 是由 a p p l e 这 5 个单词构成,这就形成了 5 的序列,我们可以对这些字符构建词嵌入;
然后输入 lstm,就像 lstm 做图像分类一样,只取最后一个输出作为预测结果,整个单词的字符串能够形成一种记忆的特性,帮助我们更好的预测词性;
接着我们把这个单词和其前面几个单词构成序列,可以对这些单词构建新的词嵌入;
最后输出结果是单词的词性,也就是根据前面几个词的信息对这个词的词性进行分类。
import torch
from torch import nn
from torch.autograd import Variable
Step1 导入训练集(简单两句话)
training_data = [("The dog ate the apple".split(),
["DET", "NN", "V", "DET", "NN"]),
("Everybody read that book".split(),
["NN", "V", "DET", "NN"])]
Step2 对单词和标签进行编码
word_to_idx = {}
tag_to_idx = {}
for context, tag in training_data:
for word in context:
if word.lower() not in word_to_idx:
word_to_idx[word.lower()] = len(word_to_idx)
for label in tag:
if label.lower() not in tag_to_idx:
tag_to_idx[label.lower()] = len(tag_to_idx)
变量查看
word_to_idx
变量查看 -> Output
{'the': 0,
'dog': 1,
'ate': 2,
'apple': 3,
'everybody': 4,
'read': 5,
'that': 6,
'book': 7}
变量查看
tag_to_idx # {'det': 0, 'nn': 1, 'v': 2}
# 对字母进行编码
alphabet = 'abcdefghijklmnopqrstuvwxyz'
char_to_idx = {}
for i in range(len(alphabet)):
char_to_idx[alphabet[i]] = i
变量查看
char_to_idx
变量查看 -> Output
{'a': 0,
'b': 1,
'c': 2,
'd': 3,
'e': 4,
'f': 5,
'g': 6,
'h': 7,
'i': 8,
'j': 9,
'k': 10,
'l': 11,
'm': 12,
'n': 13,
'o': 14,
'p': 15,
'q': 16,
'r': 17,
's': 18,
't': 19,
'u': 20,
'v': 21,
'w': 22,
'x': 23,
'y': 24,
'z': 25}
Step3 构建训练数据
def make_sequence(x, dic): # 字符编码
idx = [dic[i.lower()] for i in x]
idx = torch.LongTensor(idx)
return idx
变量查看
make_sequence('apple', char_to_idx) # tensor([ 0, 15, 15, 11, 4])
变量查看
training_data[1][0] # ['Everybody', 'read', 'that', 'book']
变量查看
make_sequence(training_data[1][0], word_to_idx) # tensor([4, 5, 6, 7])
Step4 构建单个字符的 lstm 模型
class char_lstm(nn.Module):
def __init__(self, n_char, char_dim, char_hidden):
super(char_lstm, self).__init__()
self.char_embed = nn.Embedding(n_char, char_dim)
self.lstm = nn.LSTM(char_dim, char_hidden)
def forward(self, x):
x = self.char_embed(x)
out, _ = self.lstm(x)
return out[-1] # (batch, hidden)
Step5 构建词性分类的 lstm 模型
class lstm_tagger(nn.Module):
def __init__(self, n_word, n_char, char_dim, word_dim,
char_hidden, word_hidden, n_tag):
super(lstm_tagger, self).__init__()
self.word_embed = nn.Embedding(n_word, word_dim)
self.char_lstm = char_lstm(n_char, char_dim, char_hidden)
self.word_lstm = nn.LSTM(word_dim + char_hidden, word_hidden)
self.classify = nn.Linear(word_hidden, n_tag)
def forward(self, x, word):
char = []
for w in word: # 对于每个单词做字符的 lstm
char_list = make_sequence(w, char_to_idx)
char_list = char_list.unsqueeze(1) # (seq, batch, feature) 满足 lstm 输入条件
char_infor = self.char_lstm(Variable(char_list)) # (batch, char_hidden)
char.append(char_infor)
char = torch.stack(char, dim=0) # (seq, batch, feature)
x = self.word_embed(x) # (batch, seq, word_dim)
x = x.permute(1, 0, 2) # 改变顺序
x = torch.cat((x, char), dim=2) # 沿着特征通道将每个词的词嵌入和字符 lstm 输出的结果拼接在一起
x, _ = self.word_lstm(x)
s, b, h = x.shape
x = x.view(-1, h) # 重新 reshape 进行分类线性层
out = self.classify(x)
return out
net = lstm_tagger(len(word_to_idx), len(char_to_idx), 10, 100, 50, 128, len(tag_to_idx))
criterion = nn.CrossEntropyLoss()
optimizer = torch.optim.SGD(net.parameters(), lr=1e-2)
Step6 开始训练
for e in range(300):
train_loss = 0
for word, tag in training_data:
word_list = make_sequence(word, word_to_idx).unsqueeze(0) # 添加第一维 batch
tag = make_sequence(tag, tag_to_idx)
word_list = Variable(word_list)
tag = Variable(tag)
# 前向传播
out = net(word_list, word)
loss = criterion(out, tag)
train_loss += loss.item()
# 反向传播
optimizer.zero_grad()
loss.backward()
optimizer.step()
if (e + 1) % 50 == 0:
print('Epoch: {}, Loss: {:.5f}'.format(e + 1, train_loss / len(training_data)))
输出结果
Epoch: 50, Loss: 0.91244
Epoch: 100, Loss: 0.71033
Epoch: 150, Loss: 0.50279
Epoch: 200, Loss: 0.33263
Epoch: 250, Loss: 0.21957
Epoch: 300, Loss: 0.15108
Step7 进行预测
net = net.eval()
test_sent = 'Everybody ate the apple'
test = make_sequence(test_sent.split(), word_to_idx).unsqueeze(0)
out = net(Variable(test), test_sent.split())
print(tag_to_idx)
print("=============")
print(out)
输出结果
{'det': 0, 'nn': 1, 'v': 2}
=============
tensor([[-0.9732, 1.7271, -0.7776],
[-1.0362, -0.6175, 1.5954],
[ 1.6785, -0.5793, -0.9897],
[-0.2566, 1.6937, -1.4471]], grad_fn=<AddmmBackward>)
Step8 结果解释
因为最后一层的线性层没有使用
softmax,所以数值不太像一个概率;但是每一行数值最大的就表示属于该类,可以看到第一个单词 ‘Everybody’ 属于
nn,第二个单词 ‘ate’ 属于v,第三个单词 ‘the’ 属于det,第四个单词 ‘apple’ 属于nn;所以得到的这个预测结果是正确的