Word2vec Example
Loading Text
For this example, we use the text of the Shakespeare play Hamlet, which can be downloaded from Folger.edu
'''
download shakespear at https://www.folger.edu/explore/shakespeares-works/download/
'''
with open("./text/hamlet.txt", "r", encoding="utf-8") as file:
content = file.read().lower()
Tokenizing
'''
tokenize the text
'''
punctuation_pattern = re.escape(string.punctuation)
pattern = f"[{re.escape(string.punctuation)}\\s]"
tokens = [
p
for p in re.split(pattern, content)
if p not in [',','.',' ','\t','\n','"',"'",'']
]
'''
remove duplicates dna then sort
'''
token_set = sorted(set(tokens))
'''
assign an integer to each token
'''
indexed = [
{'index':i, 'token':token}
for i, token in enumerate(token_set)
]
'''
create a lookkup map for tokens
'''
map = {}
for item in indexed:
map[item['token']] = {'token':item, 'context':[]}
Creating the Inputs and Outputs
inputs = []
outputs = []
for index, item in enumerate(tokens):
if index>2 and index < len(tokens)-2:
item2 = map[item]
item2['context'].append(tokens[index-2])
item2['context'].append(tokens[index-1])
item2['context'].append(tokens[index+1])
item2['context'].append(tokens[index+2])
pass
pass
'''
create a one hot encoded tensor for an index. that is, the number i is mapped to a tensor of length n,
where n is the number of tokens, filled with zeros and with a 1 in the ith index
'''
def create_tensor(index):
n = len(indexed)
tensor = torch.zeros(n)
tensor[index] = 1.0
return tensor
for key in map:
item = map[key]
item['token']['tensor'] = create_tensor(item['token']['index'])
pass
'''
construct the input and output tensors
'''
for key in map:
item = map[key]
input_tensor = item['token']['tensor']
for word in item['context']:
output_tensor = map[word]['token']['tensor']
inputs.append(input_tensor)
outputs.append(output_tensor)
pass
pass
Creating the Model
'''
construct the model
'''
network = nn.Sequential(
nn.Linear(len(indexed), 20),
nn.Tanh(),
nn.Linear(20,len(indexed)),
nn.Sigmoid()
)
opt = op.SGD(network.parameters(), lr=0.01)
err = nn.CrossEntropyLoss()
index = 1
def callback(current_loss):
global index
print('run = '+str(index)+' and current error is '+str(current_loss.item()) )
index += 1
pass
#torch.stack will convert the list of tensors to a tensor
tc.train(network, opt, err, torch.stack(inputs, dim=0), torch.stack(outputs,dim=0), 3, callback=callback)
PUlling apart the Model
'''
now we want to strip away the second layer. The word embedding is the result of applying
the first layer of the trained network to the token
'''
subset = network[:2]
saved = []
for key in map:
output = subset(map[key]['token']['tensor'])
map[key]['embedding'] = output
list_1d = output.tolist()
saved.append({
'token': map[key]['token']['token'],
'index':map[key]['token']['index'],
'embedding': list_1d
})
print(map[key]['token']['token'] +' = '+ str(list_1d))
pass
with open("./text/output.json", "w") as file:
file.write(json.dumps(saved))