接下来我们将介绍三种文本分类的算法模型,其中已经完成了前两种模型,本文将详细介绍最后一种算法模型。
我们定义了一个名为 DeepCbow 的类,该类继承自 nn.Module。以下是该类的详细说明:
```python class DeepCbow(nn.Module): def init(self, nwords, ntags, nlayers, embsize, hidsize): super(DeepCbow, self).init() self.nlayers = nlayers self.embedding = nn.Embedding(nwords, embsize) nn.init.xavieruniform_(self.embedding.weight)
self.linears = nn.ModuleList(
[nn.Linear(emb_size if i == 0 else hid_size, hid_size) for i in range(nlayers)]
)
for i in range(nlayers):
nn.init.xavier_uniform_(self.linears[i].weight)
self.out = nn.Linear(hid_size, ntags)
nn.init.xavier_uniform_(self.out.weight)
def forward(self, words):
emb = self.embedding(words)
emb_sum = torch.sum(emb, dim=0)
h = emb_sum.view(1, -1)
for i in range(self.nlayers):
h = self.linears[i](h)
h = torch.tanh(h)
out = self.out(h)
return out
```
self.linears = nn.ModuleList([nn.Linear(emb_size if i == 0 else hid_size, hid_size) for i in range(nlayers)])
这行代码表示要构建一个包含多个隐藏层的神经网络。第一层的输入神经元数为 emb_size,而后续各层的输入神经元数均为 hid_size。
words 的维度为 [文本单词数]embedding 层处理后,维度变为 [文本单词数, embedding 编码维度]emb 在 dim=0 维度求和后,维度变为 [embedding 编码维度][1, embedding 编码维度][1, ntags]以下是完整的代码实现:
```python import torch from torch import nn import random import numpy as np from collections import defaultdict
class DeepCbow(nn.Module): def init(self, nwords, ntags, nlayers, embsize, hidsize): super(DeepCbow, self).init() self.nlayers = nlayers self.embedding = nn.Embedding(nwords, embsize) nn.init.xavieruniform_(self.embedding.weight)
self.linears = nn.ModuleList(
[nn.Linear(emb_size if i == 0 else hid_size, hid_size) for i in range(nlayers)]
)
for i in range(nlayers):
nn.init.xavier_uniform_(self.linears[i].weight)
self.out = nn.Linear(hid_size, ntags)
nn.init.xavier_uniform_(self.out.weight)
def forward(self, words):
emb = self.embedding(words)
emb_sum = torch.sum(emb, dim=0)
h = emb_sum.view(1, -1)
for i in range(self.nlayers):
h = self.linears[i](h)
h = torch.tanh(h)
out = self.out(h)
return out
w2i = defaultdict(lambda: len(w2i))
t2i = defaultdict(lambda: len(t2i))
UNK = w2i["
def read_dataset(filename): with open(filename, "r") as f: for line in f: tag, words = line.lower().strip().split(" ||| ") yield ([w2i[x] for x in words.split(" ")], t2i[tag])
train = list(readdataset("train.txt")) w2i = defaultdict(lambda: UNK, w2i) dev = list(readdataset("test.txt"))
nwords = len(w2i) ntags = len(t2i) EMBSIZE = 64 HIDSIZE = 64 NLAYERS = 2
model = DeepCbow(nwords, ntags, NLAYERS, EMBSIZE, HIDSIZE) criterion = nn.CrossEntropyLoss() optimizer = torch.optim.Adam(model.parameters())
for epoch in range(100): random.shuffle(train) train_loss = 0.0
for words, tag in train:
words = torch.tensor(words)
tag = torch.tensor([tag])
score = model(words)
loss = criterion(score, tag)
train_loss += loss.item()
optimizer.zero_grad()
loss.backward()
optimizer.step()
print("current epoch %r, current loss %.4f" % (epoch, train_loss / len(train)))
test_correct = 0.0
for words, tag in dev:
words = torch.tensor(words)
scores = model(words).detach().numpy()
predict = np.argmax(scores)
if predict == tag:
test_correct += 1
print("current epoch %r, test accuracy %.4f" % (epoch, test_correct / len(dev)))
```
以上是对原文内容的改写,旨在保留原文的核心信息,同时提高文章的可读性和紧凑性。