Word embedding demo
UMass CS 485, 2026-09-30
In [24]:
import numpy as np
import matplotlib.pyplot as plt
plt.rcParams['figure.figsize'] = [5.2,5] ## fairly square plots
#plt.rcParams['figure.figsize'] = [14,6] ## fills notebook width
%matplotlib inline
Load the data. Downloaded a while ago from https://nlp.stanford.edu/projects/glove/
You may need to use http://web.archive.org (highly recommended!). The direct link seems to be: http://downloads.cs.stanford.edu/nlp/data/glove.6B.zip
Trained on -- English Wikipedia (2014) and Gigaword 5 (English newspaper and newswire), totalling 6 billion word tokens in length (dozens of GB of text)
There are other copies on the web and various libraries to help load these for you, too.
For actual use I'd recommend using one of the larger embeddings versions listed on that webpage.
In [2]:
lines = open("/users/brenocon/data/lexical/glove/glove.6B.50d.txt").readlines()
In [3]:
len(lines)
Out[3]:
400000
In [4]:
lines[200]
Out[4]:
'according 0.3675 0.17162 0.45661 0.22694 0.50477 -0.16938 -0.72449 -0.60276 0.25607 -0.67345 0.3297 -0.28103 0.15122 -0.64325 1.0454 0.0028958 -0.51234 -0.33298 -0.092862 0.24603 0.31475 -0.020641 0.55353 -0.19807 0.11941 -1.327 -0.65037 -0.46369 -0.86273 0.38967 3.32 -0.73484 0.10476 -0.62037 -0.25884 -0.39999 0.14253 -0.11855 0.62405 0.70724 -0.11078 0.29246 0.49381 -0.2496 0.0020108 -0.4103 -0.62928 0.78374 0.17455 0.17664\n'
In [5]:
vocab = np.array( [L.split()[0] for L in lines] )
In [11]:
vocab[:100]
Out[11]:
array(['the', ',', '.', 'of', 'to', 'and', 'in', 'a', '"', "'s", 'for',
'-', 'that', 'on', 'is', 'was', 'said', 'with', 'he', 'as', 'it',
'by', 'at', '(', ')', 'from', 'his', "''", '``', 'an', 'be', 'has',
'are', 'have', 'but', 'were', 'not', 'this', 'who', 'they', 'had',
'i', 'which', 'will', 'their', ':', 'or', 'its', 'one', 'after',
'new', 'been', 'also', 'we', 'would', 'two', 'more', "'", 'first',
'about', 'up', 'when', 'year', 'there', 'all', '--', 'out', 'she',
'other', 'people', "n't", 'her', 'percent', 'than', 'over', 'into',
'last', 'some', 'government', 'time', '$', 'you', 'years', 'if',
'no', 'world', 'can', 'three', 'do', ';', 'president', 'only',
'state', 'million', 'could', 'us', 'most', '_', 'against', 'u.s.'],
dtype='<U68')
In [7]:
#L=lines[0]
wordvecs = np.array([ np.array([float(x) for x in L.split()[1:]]) for L in lines ])
In [10]:
wordvecs.shape
Out[10]:
(400000, 50)
In [16]:
i=np.where(vocab=="cookie")[0][0]
In [18]:
wordvecs[i,:]
Out[18]:
array([-0.076743, -0.21017 , 0.56356 , 0.057243, 1.2726 , 0.67451 ,
-0.651 , -0.38204 , 0.27345 , -0.62413 , -0.66457 , 0.20545 ,
0.24752 , 1.3365 , -0.10062 , 0.010924, -0.036977, -0.028287,
0.26451 , -0.85008 , 0.68399 , -0.33976 , 1.038 , 0.43267 ,
-0.12916 , 0.074176, -1.5373 , 0.41251 , 1.3431 , -0.46741 ,
0.90981 , -0.42715 , -0.89915 , 1.7175 , -0.10247 , 0.4782 ,
-0.66754 , 1.0398 , 1.0425 , -0.4286 , 1.1889 , -0.056237,
-0.19045 , 0.023564, 0.19073 , 0.65252 , 0.38651 , -0.019387,
0.29446 , 0.12667 ])
In [19]:
cookie = wordvecs[np.where(vocab=="cookie")[0][0],:]
In [20]:
cake = wordvecs[np.where(vocab=="cake")[0][0],:]
In [21]:
cake
Out[21]:
array([ 0.069443 , 0.76897 , -0.52978 , 0.084142 , 1.1945 ,
0.37776 , -0.34614 , 0.089619 , 0.030415 , 0.14648 ,
-0.51823 , -0.11561 , 1.2622 , 1.0821 , 0.35248 ,
0.10658 , 0.0088652, -0.039198 , 0.52789 , -1.1781 ,
0.79956 , -0.49779 , 1.1323 , 0.06991 , 0.0053372,
0.049828 , -1.0271 , 1.0131 , 1.2817 , -0.45057 ,
1.529 , -0.31407 , -1.4749 , 1.7196 , -0.29891 ,
0.20966 , -0.3686 , 1.333 , 0.97123 , -0.50428 ,
0.66221 , 0.10779 , -0.69917 , -0.62049 , 0.25486 ,
0.84102 , 0.20349 , -1.0346 , -0.074467 , -0.23133 ])
In [27]:
def cossim(x, y):
return x.dot(y) / np.linalg.norm(x) / np.linalg.norm(y)
In [28]:
cossim( np.array([ 1, 2, 3 ]), np.array([1, 2, 3]) )
Out[28]:
np.float64(1.0)
In [29]:
cossim( np.array([ -1, -2, -3 ]), np.array([1, 2, 3]) )
Out[29]:
np.float64(-1.0)
In [30]:
cossim( np.array([ 1, 2, 3 ]), np.array([1000, 2000, 3000]) )
Out[30]:
np.float64(1.0000000000000002)
In [44]:
scores = np.array([ cossim(wordvecs[i,:], cookie) for i in range(400000) ])
In [51]:
for i in np.argsort(-scores)[:100]:
print(i, vocab[i], cossim(wordvecs[i,:], cookie))
13816 cookie 1.0 11641 cookies 0.8521579988114755 7931 cake 0.824885196612333 9246 pie 0.8141830427262035 18929 pastry 0.810629886423002 10057 baking 0.7960145776566127 12985 dough 0.7904830518825011 18952 scoop 0.7772820020957749 13268 baked 0.7597561174563356 13418 bake 0.7458634266887119 6242 chocolate 0.7419693977999394 7232 recipe 0.7412312765067288 5872 bread 0.7386142198142993 26858 crumbs 0.7349171655049351 35795 oatmeal 0.7306888379571089 6458 butter 0.7299392093902305 93018 shortbread 0.7235800260444947 22894 loaf 0.7224630753808289 39472 cheesecake 0.7192391621131444 31942 biscuit 0.717413608187205 41238 muffin 0.7161582616402903 53454 tins 0.7160981146439117 42917 brownies 0.7158605592492544 30751 pancakes 0.714306221016465 35280 custard 0.7120371307971101 9388 pizza 0.7112754980915486 62513 crusts 0.708829236246976 10425 potato 0.7067397673276038 10542 slice 0.7049850871215563 17024 cakes 0.7037498409976596 5795 cheese 0.7010597126793617 12611 sandwich 0.7009506154406124 25241 crackers 0.699790523264658 14794 crust 0.6988571974107497 13649 slices 0.6971828820145313 13687 rack 0.6959803312819235 7260 dish 0.6947048051651263 17943 pumpkin 0.6926724754349487 15703 dessert 0.6925657459816987 22264 jelly 0.6882782257797659 8629 candy 0.6880942605990722 8233 soup 0.6870660976645149 52881 buttered 0.6832512021759706 30887 tortilla 0.6814832775514051 39396 pizzas 0.6799239269135327 94149 phyllo 0.6760117041138255 15169 spoon 0.6753455361704355 9622 salad 0.6739908724895783 43922 greased 0.6731985387856234 25019 pies 0.6713404034350747 33706 casserole 0.6706461256616207 7474 rolls 0.6684518550260069 5161 cream 0.6677637677462508 6593 egg 0.6669444516763129 12617 pasta 0.6668348607637908 6109 sheet 0.6661055892604821 8212 flour 0.6642538523311995 62981 oreo 0.6603668624260539 39469 muffins 0.6583729564662416 9007 wrap 0.6581200011348253 34221 pancake 0.6576640859358657 4769 bag 0.6573292024567081 16093 toast 0.6568793888449682 9944 nuts 0.6558123039019043 21720 pudding 0.6539730226990624 24304 cubes 0.6529105324307047 9189 bits 0.6526665209084127 16762 sandwiches 0.6525294969921975 30598 fudge 0.6503681444372923 20198 popcorn 0.6503159917974084 9058 oven 0.6501222475782046 13673 refrigerator 0.6491479113311854 10503 fried 0.6487658049947218 16720 jar 0.6486079184923222 15526 chunks 0.648133238968578 23569 tart 0.6469861406360465 40852 frosting 0.6462411269841789 9477 potatoes 0.6462160037997554 13943 peanut 0.6461506991575239 41345 loaves 0.6442525679014376 35122 pastries 0.6441062014201944 16200 snack 0.6403232754536714 62386 omelet 0.6401126600319323 41721 crispy 0.6390364972640357 40747 brownie 0.6386874893091524 341570 lavagetto 0.6373377079021145 10554 stuffed 0.6373041170878982 40530 crunchy 0.6352865807424108 25676 puff 0.6350712176003129 18913 fries 0.6347752931568809 17814 sprinkle 0.6340061368350367 46074 9-inch 0.6337338024773103 27734 mashed 0.6334646159349792 44386 toaster 0.632672404320008 12109 nail 0.6326031388761977 51200 pretzels 0.6322528936751379 24825 drawer 0.6308033860130403 24562 toasted 0.6295039360182955 8628 recipes 0.6295038457524453 7046 ingredients 0.6282410392750363
In [52]:
for i in np.argsort(scores)[:30]:
print(i, vocab[i], cossim(wordvecs[i,:], cookie))
195696 unley -0.6299441852777353 166684 thurstan -0.6269932685734875 282095 kashkar -0.6187193498769015 291205 ieronymos -0.6138883747435774 200261 mmrda -0.5914216762569736 145520 lanfranc -0.5874432178940485 336344 alkazi -0.581175302374551 287533 116.33 -0.580878465822245 249909 gongga -0.579532162899631 261768 shastriji -0.5760134362582102 266535 soldan -0.5735783453241056 41932 monash -0.5717222420267948 321555 martanda -0.570822332942928 195564 bonello -0.5662643970723427 210220 dyche -0.5638633806128981 355728 shadle -0.5628398774197689 244974 zhijiang -0.561271124839506 179584 ncpa -0.5603920905443377 150961 esmor -0.5595999561878388 75474 ninoy -0.5591931962068704 155495 fantino -0.5583264672744387 304099 sji -0.5566332234506424 180289 antagonised -0.5563596608166864 132152 iakovos -0.5552821676183515 371831 miege -0.5548839261224043 308871 interboro -0.550245941322153 362611 yvr -0.5482235408421388 207978 nease -0.5474516468776084 399943 usapa -0.5469145241904138 336963 anthimos -0.5451467346864656
In [56]:
cookie = wordvecs[np.where(vocab=="the")[0][0],:]
scores = np.array([ cossim(wordvecs[i,:], cookie) for i in range(400000) ])
for i in np.argsort(-scores)[:30]:
print(i, vocab[i], cossim(wordvecs[i,:], cookie))
0 the 0.9999999999999998 42 which 0.9221877446660318 153 part 0.917894962483752 6 in 0.9029428959951388 3 of 0.9026352642565101 13 on 0.898413735707049 48 one 0.8948692060759184 2 . 0.8917523356721194 19 as 0.8904381583327057 37 this 0.8828657685049587 47 its 0.8809497266446554 215 same 0.8806194279516006 58 first 0.8699569156034566 1452 entire 0.8633796092751588 52 also 0.8608381051410171 20 it 0.8606576015236609 4 to 0.8574301557746279 170 another 0.857357900216889 263 came 0.8573306305565251 10 for 0.8567915433130374 143 well 0.8531814427988991 111 where 0.8528323228983445 7 a 0.8517428747944916 212 however 0.8502730154375981 91 only 0.8491234815664997 25 from 0.8487151699209832 75 into 0.8469710492995769 492 taken 0.8420961510774191 376 although 0.8419191964895506 17 with 0.8403741373821758
In [57]:
cookie = wordvecs[np.where(vocab=="monster")[0][0],:]
scores = np.array([ cossim(wordvecs[i,:], cookie) for i in range(400000) ])
for i in np.argsort(-scores)[:30]:
print(i, vocab[i], cossim(wordvecs[i,:], cookie))
7519 monster 1.0 11655 beast 0.8600439679726953 12956 monsters 0.7986564332992634 7431 ghost 0.7932728226748019 20374 zombie 0.7740900547773115 11539 spider 0.7675895947822594 5450 cat 0.7619763141083461 10988 creature 0.7613341462149754 11408 monkey 0.7421374822269433 11035 bug 0.7341485426748917 13455 villain 0.7332179473125586 12425 rabbit 0.7275375811568532 18561 superhero 0.7215602916882004 8439 sequel 0.7164626221739362 1005 movie 0.7140901448072705 5613 hell 0.7120097716350353 2926 dog 0.7105765777783042 5578 crazy 0.7054880720420486 11782 batman 0.7014593775556807 7394 dragon 0.6994238353530752 11868 vampire 0.6942202787085183 6092 animated 0.6929530046563378 4313 kid 0.6927114302659942 4898 killer 0.6923206193095414 5988 horror 0.6907340601896874 9247 robot 0.6849970904290731 9517 snake 0.6803458183344245 7571 mouse 0.676889541292644 10212 scary 0.6735640965964927 13055 smash 0.6732490766548851
In [58]:
cookie = wordvecs[np.where(vocab=="nerd")[0][0],:]
scores = np.array([ cossim(wordvecs[i,:], cookie) for i in range(400000) ])
for i in np.argsort(-scores)[:30]:
print(i, vocab[i], cossim(wordvecs[i,:], cookie))
35465 nerd 1.0 26309 geek 0.8293729096567657 44144 slacker 0.8090723871724442 58089 geeky 0.7969193370608488 29931 crazed 0.7569688157208042 47541 nerdy 0.7433485866982084 86555 curmudgeon 0.7201585094656451 96817 dorky 0.697263024686298 34082 geeks 0.696192619699131 4313 kid 0.6950220419462729 33201 gamer 0.6936665222970874 27401 fanatic 0.6884175011489712 13436 mentality 0.6786048674653181 63449 scrawny 0.675444114807518 61248 gizmo 0.675061173756852 65316 gunslinger 0.6668263550512463 26436 cheerleader 0.6654754514214731 24737 gadget 0.6635588542490874 50127 yuppie 0.661955757500423 44895 nerds 0.655154160883919 88432 techie 0.6516363275250835 93930 dork 0.6506518672656352 12594 obsessed 0.6474508306931893 43645 archetype 0.6440522960792431 58701 wisecracking 0.6417392135467914 16689 instinct 0.6412341448745975 49153 hipster 0.6404485715650188 21110 stereotype 0.6395275993326107 78844 waif 0.6391300010438783 94896 gawky 0.638806184430464