Word2vec Example

Loading Text

For this example, we use the text of the Shakespeare play Hamlet, which can be downloaded from Folger.edu
''' download shakespear at https://www.folger.edu/explore/shakespeares-works/download/ ''' with open("./text/hamlet.txt", "r", encoding="utf-8") as file: content = file.read().lower()

Tokenizing

''' tokenize the text ''' punctuation_pattern = re.escape(string.punctuation) pattern = f"[{re.escape(string.punctuation)}\\s]" tokens = [ p for p in re.split(pattern, content) if p not in [',','.',' ','\t','\n','"',"'",''] ] ''' remove duplicates dna then sort ''' token_set = sorted(set(tokens)) ''' assign an integer to each token ''' indexed = [ {'index':i, 'token':token} for i, token in enumerate(token_set) ] ''' create a lookkup map for tokens ''' map = {} for item in indexed: map[item['token']] = {'token':item, 'context':[]}

Creating the Inputs and Outputs

inputs = [] outputs = [] for index, item in enumerate(tokens): if index>2 and index < len(tokens)-2: item2 = map[item] item2['context'].append(tokens[index-2]) item2['context'].append(tokens[index-1]) item2['context'].append(tokens[index+1]) item2['context'].append(tokens[index+2]) pass pass ''' create a one hot encoded tensor for an index. that is, the number i is mapped to a tensor of length n, where n is the number of tokens, filled with zeros and with a 1 in the ith index ''' def create_tensor(index): n = len(indexed) tensor = torch.zeros(n) tensor[index] = 1.0 return tensor for key in map: item = map[key] item['token']['tensor'] = create_tensor(item['token']['index']) pass ''' construct the input and output tensors ''' for key in map: item = map[key] input_tensor = item['token']['tensor'] for word in item['context']: output_tensor = map[word]['token']['tensor'] inputs.append(input_tensor) outputs.append(output_tensor) pass pass

Creating the Model

''' construct the model ''' network = nn.Sequential( nn.Linear(len(indexed), 20), nn.Tanh(), nn.Linear(20,len(indexed)), nn.Sigmoid() ) opt = op.SGD(network.parameters(), lr=0.01) err = nn.CrossEntropyLoss() index = 1 def callback(current_loss): global index print('run = '+str(index)+' and current error is '+str(current_loss.item()) ) index += 1 pass #torch.stack will convert the list of tensors to a tensor tc.train(network, opt, err, torch.stack(inputs, dim=0), torch.stack(outputs,dim=0), 3, callback=callback)

PUlling apart the Model

''' now we want to strip away the second layer. The word embedding is the result of applying the first layer of the trained network to the token ''' subset = network[:2] saved = [] for key in map: output = subset(map[key]['token']['tensor']) map[key]['embedding'] = output list_1d = output.tolist() saved.append({ 'token': map[key]['token']['token'], 'index':map[key]['token']['index'], 'embedding': list_1d }) print(map[key]['token']['token'] +' = '+ str(list_1d)) pass with open("./text/output.json", "w") as file: file.write(json.dumps(saved))

Full Script/Desktop