Word embedding demo

UMass CS 485, 2026-09-30

In [24]:
import numpy as np

import matplotlib.pyplot as plt
plt.rcParams['figure.figsize'] = [5.2,5] ## fairly square plots
#plt.rcParams['figure.figsize'] = [14,6]  ## fills notebook width
%matplotlib inline

Load the data. Downloaded a while ago from https://nlp.stanford.edu/projects/glove/

You may need to use http://web.archive.org (highly recommended!). The direct link seems to be: http://downloads.cs.stanford.edu/nlp/data/glove.6B.zip

Trained on -- English Wikipedia (2014) and Gigaword 5 (English newspaper and newswire), totalling 6 billion word tokens in length (dozens of GB of text)

There are other copies on the web and various libraries to help load these for you, too.

For actual use I'd recommend using one of the larger embeddings versions listed on that webpage.

In [2]:
lines = open("/users/brenocon/data/lexical/glove/glove.6B.50d.txt").readlines()
In [3]:
len(lines)
Out[3]:
400000
In [4]:
lines[200]
Out[4]:
'according 0.3675 0.17162 0.45661 0.22694 0.50477 -0.16938 -0.72449 -0.60276 0.25607 -0.67345 0.3297 -0.28103 0.15122 -0.64325 1.0454 0.0028958 -0.51234 -0.33298 -0.092862 0.24603 0.31475 -0.020641 0.55353 -0.19807 0.11941 -1.327 -0.65037 -0.46369 -0.86273 0.38967 3.32 -0.73484 0.10476 -0.62037 -0.25884 -0.39999 0.14253 -0.11855 0.62405 0.70724 -0.11078 0.29246 0.49381 -0.2496 0.0020108 -0.4103 -0.62928 0.78374 0.17455 0.17664\n'
In [5]:
vocab = np.array(  [L.split()[0] for L in lines]  )
In [11]:
vocab[:100]
Out[11]:
array(['the', ',', '.', 'of', 'to', 'and', 'in', 'a', '"', "'s", 'for',
       '-', 'that', 'on', 'is', 'was', 'said', 'with', 'he', 'as', 'it',
       'by', 'at', '(', ')', 'from', 'his', "''", '``', 'an', 'be', 'has',
       'are', 'have', 'but', 'were', 'not', 'this', 'who', 'they', 'had',
       'i', 'which', 'will', 'their', ':', 'or', 'its', 'one', 'after',
       'new', 'been', 'also', 'we', 'would', 'two', 'more', "'", 'first',
       'about', 'up', 'when', 'year', 'there', 'all', '--', 'out', 'she',
       'other', 'people', "n't", 'her', 'percent', 'than', 'over', 'into',
       'last', 'some', 'government', 'time', '$', 'you', 'years', 'if',
       'no', 'world', 'can', 'three', 'do', ';', 'president', 'only',
       'state', 'million', 'could', 'us', 'most', '_', 'against', 'u.s.'],
      dtype='<U68')
In [7]:
#L=lines[0]
wordvecs = np.array([ np.array([float(x) for x in L.split()[1:]])  for L in lines ])
In [10]:
wordvecs.shape
Out[10]:
(400000, 50)
In [16]:
i=np.where(vocab=="cookie")[0][0]
In [18]:
wordvecs[i,:]
Out[18]:
array([-0.076743, -0.21017 ,  0.56356 ,  0.057243,  1.2726  ,  0.67451 ,
       -0.651   , -0.38204 ,  0.27345 , -0.62413 , -0.66457 ,  0.20545 ,
        0.24752 ,  1.3365  , -0.10062 ,  0.010924, -0.036977, -0.028287,
        0.26451 , -0.85008 ,  0.68399 , -0.33976 ,  1.038   ,  0.43267 ,
       -0.12916 ,  0.074176, -1.5373  ,  0.41251 ,  1.3431  , -0.46741 ,
        0.90981 , -0.42715 , -0.89915 ,  1.7175  , -0.10247 ,  0.4782  ,
       -0.66754 ,  1.0398  ,  1.0425  , -0.4286  ,  1.1889  , -0.056237,
       -0.19045 ,  0.023564,  0.19073 ,  0.65252 ,  0.38651 , -0.019387,
        0.29446 ,  0.12667 ])
In [19]:
cookie = wordvecs[np.where(vocab=="cookie")[0][0],:]
In [20]:
cake = wordvecs[np.where(vocab=="cake")[0][0],:]
In [21]:
cake
Out[21]:
array([ 0.069443 ,  0.76897  , -0.52978  ,  0.084142 ,  1.1945   ,
        0.37776  , -0.34614  ,  0.089619 ,  0.030415 ,  0.14648  ,
       -0.51823  , -0.11561  ,  1.2622   ,  1.0821   ,  0.35248  ,
        0.10658  ,  0.0088652, -0.039198 ,  0.52789  , -1.1781   ,
        0.79956  , -0.49779  ,  1.1323   ,  0.06991  ,  0.0053372,
        0.049828 , -1.0271   ,  1.0131   ,  1.2817   , -0.45057  ,
        1.529    , -0.31407  , -1.4749   ,  1.7196   , -0.29891  ,
        0.20966  , -0.3686   ,  1.333    ,  0.97123  , -0.50428  ,
        0.66221  ,  0.10779  , -0.69917  , -0.62049  ,  0.25486  ,
        0.84102  ,  0.20349  , -1.0346   , -0.074467 , -0.23133  ])
In [27]:
def cossim(x, y):
    return x.dot(y) / np.linalg.norm(x) / np.linalg.norm(y)
In [28]:
cossim( np.array([  1, 2, 3 ]),  np.array([1, 2, 3])  )  
Out[28]:
np.float64(1.0)
In [29]:
cossim( np.array([  -1, -2, -3 ]),  np.array([1, 2, 3])  )  
Out[29]:
np.float64(-1.0)
In [30]:
cossim( np.array([  1, 2, 3 ]),  np.array([1000, 2000, 3000])  )  
Out[30]:
np.float64(1.0000000000000002)
In [44]:
scores = np.array([  cossim(wordvecs[i,:], cookie) for i in range(400000)  ])
In [51]:
for i in np.argsort(-scores)[:100]:
    print(i, vocab[i],  cossim(wordvecs[i,:],  cookie))
13816 cookie 1.0
11641 cookies 0.8521579988114755
7931 cake 0.824885196612333
9246 pie 0.8141830427262035
18929 pastry 0.810629886423002
10057 baking 0.7960145776566127
12985 dough 0.7904830518825011
18952 scoop 0.7772820020957749
13268 baked 0.7597561174563356
13418 bake 0.7458634266887119
6242 chocolate 0.7419693977999394
7232 recipe 0.7412312765067288
5872 bread 0.7386142198142993
26858 crumbs 0.7349171655049351
35795 oatmeal 0.7306888379571089
6458 butter 0.7299392093902305
93018 shortbread 0.7235800260444947
22894 loaf 0.7224630753808289
39472 cheesecake 0.7192391621131444
31942 biscuit 0.717413608187205
41238 muffin 0.7161582616402903
53454 tins 0.7160981146439117
42917 brownies 0.7158605592492544
30751 pancakes 0.714306221016465
35280 custard 0.7120371307971101
9388 pizza 0.7112754980915486
62513 crusts 0.708829236246976
10425 potato 0.7067397673276038
10542 slice 0.7049850871215563
17024 cakes 0.7037498409976596
5795 cheese 0.7010597126793617
12611 sandwich 0.7009506154406124
25241 crackers 0.699790523264658
14794 crust 0.6988571974107497
13649 slices 0.6971828820145313
13687 rack 0.6959803312819235
7260 dish 0.6947048051651263
17943 pumpkin 0.6926724754349487
15703 dessert 0.6925657459816987
22264 jelly 0.6882782257797659
8629 candy 0.6880942605990722
8233 soup 0.6870660976645149
52881 buttered 0.6832512021759706
30887 tortilla 0.6814832775514051
39396 pizzas 0.6799239269135327
94149 phyllo 0.6760117041138255
15169 spoon 0.6753455361704355
9622 salad 0.6739908724895783
43922 greased 0.6731985387856234
25019 pies 0.6713404034350747
33706 casserole 0.6706461256616207
7474 rolls 0.6684518550260069
5161 cream 0.6677637677462508
6593 egg 0.6669444516763129
12617 pasta 0.6668348607637908
6109 sheet 0.6661055892604821
8212 flour 0.6642538523311995
62981 oreo 0.6603668624260539
39469 muffins 0.6583729564662416
9007 wrap 0.6581200011348253
34221 pancake 0.6576640859358657
4769 bag 0.6573292024567081
16093 toast 0.6568793888449682
9944 nuts 0.6558123039019043
21720 pudding 0.6539730226990624
24304 cubes 0.6529105324307047
9189 bits 0.6526665209084127
16762 sandwiches 0.6525294969921975
30598 fudge 0.6503681444372923
20198 popcorn 0.6503159917974084
9058 oven 0.6501222475782046
13673 refrigerator 0.6491479113311854
10503 fried 0.6487658049947218
16720 jar 0.6486079184923222
15526 chunks 0.648133238968578
23569 tart 0.6469861406360465
40852 frosting 0.6462411269841789
9477 potatoes 0.6462160037997554
13943 peanut 0.6461506991575239
41345 loaves 0.6442525679014376
35122 pastries 0.6441062014201944
16200 snack 0.6403232754536714
62386 omelet 0.6401126600319323
41721 crispy 0.6390364972640357
40747 brownie 0.6386874893091524
341570 lavagetto 0.6373377079021145
10554 stuffed 0.6373041170878982
40530 crunchy 0.6352865807424108
25676 puff 0.6350712176003129
18913 fries 0.6347752931568809
17814 sprinkle 0.6340061368350367
46074 9-inch 0.6337338024773103
27734 mashed 0.6334646159349792
44386 toaster 0.632672404320008
12109 nail 0.6326031388761977
51200 pretzels 0.6322528936751379
24825 drawer 0.6308033860130403
24562 toasted 0.6295039360182955
8628 recipes 0.6295038457524453
7046 ingredients 0.6282410392750363
In [52]:
for i in np.argsort(scores)[:30]:
    print(i, vocab[i],  cossim(wordvecs[i,:],  cookie))
195696 unley -0.6299441852777353
166684 thurstan -0.6269932685734875
282095 kashkar -0.6187193498769015
291205 ieronymos -0.6138883747435774
200261 mmrda -0.5914216762569736
145520 lanfranc -0.5874432178940485
336344 alkazi -0.581175302374551
287533 116.33 -0.580878465822245
249909 gongga -0.579532162899631
261768 shastriji -0.5760134362582102
266535 soldan -0.5735783453241056
41932 monash -0.5717222420267948
321555 martanda -0.570822332942928
195564 bonello -0.5662643970723427
210220 dyche -0.5638633806128981
355728 shadle -0.5628398774197689
244974 zhijiang -0.561271124839506
179584 ncpa -0.5603920905443377
150961 esmor -0.5595999561878388
75474 ninoy -0.5591931962068704
155495 fantino -0.5583264672744387
304099 sji -0.5566332234506424
180289 antagonised -0.5563596608166864
132152 iakovos -0.5552821676183515
371831 miege -0.5548839261224043
308871 interboro -0.550245941322153
362611 yvr -0.5482235408421388
207978 nease -0.5474516468776084
399943 usapa -0.5469145241904138
336963 anthimos -0.5451467346864656
In [56]:
cookie = wordvecs[np.where(vocab=="the")[0][0],:]
scores = np.array([  cossim(wordvecs[i,:], cookie) for i in range(400000)  ])
for i in np.argsort(-scores)[:30]:
    print(i, vocab[i],  cossim(wordvecs[i,:],  cookie))
0 the 0.9999999999999998
42 which 0.9221877446660318
153 part 0.917894962483752
6 in 0.9029428959951388
3 of 0.9026352642565101
13 on 0.898413735707049
48 one 0.8948692060759184
2 . 0.8917523356721194
19 as 0.8904381583327057
37 this 0.8828657685049587
47 its 0.8809497266446554
215 same 0.8806194279516006
58 first 0.8699569156034566
1452 entire 0.8633796092751588
52 also 0.8608381051410171
20 it 0.8606576015236609
4 to 0.8574301557746279
170 another 0.857357900216889
263 came 0.8573306305565251
10 for 0.8567915433130374
143 well 0.8531814427988991
111 where 0.8528323228983445
7 a 0.8517428747944916
212 however 0.8502730154375981
91 only 0.8491234815664997
25 from 0.8487151699209832
75 into 0.8469710492995769
492 taken 0.8420961510774191
376 although 0.8419191964895506
17 with 0.8403741373821758
In [57]:
cookie = wordvecs[np.where(vocab=="monster")[0][0],:]
scores = np.array([  cossim(wordvecs[i,:], cookie) for i in range(400000)  ])
for i in np.argsort(-scores)[:30]:
    print(i, vocab[i],  cossim(wordvecs[i,:],  cookie))
7519 monster 1.0
11655 beast 0.8600439679726953
12956 monsters 0.7986564332992634
7431 ghost 0.7932728226748019
20374 zombie 0.7740900547773115
11539 spider 0.7675895947822594
5450 cat 0.7619763141083461
10988 creature 0.7613341462149754
11408 monkey 0.7421374822269433
11035 bug 0.7341485426748917
13455 villain 0.7332179473125586
12425 rabbit 0.7275375811568532
18561 superhero 0.7215602916882004
8439 sequel 0.7164626221739362
1005 movie 0.7140901448072705
5613 hell 0.7120097716350353
2926 dog 0.7105765777783042
5578 crazy 0.7054880720420486
11782 batman 0.7014593775556807
7394 dragon 0.6994238353530752
11868 vampire 0.6942202787085183
6092 animated 0.6929530046563378
4313 kid 0.6927114302659942
4898 killer 0.6923206193095414
5988 horror 0.6907340601896874
9247 robot 0.6849970904290731
9517 snake 0.6803458183344245
7571 mouse 0.676889541292644
10212 scary 0.6735640965964927
13055 smash 0.6732490766548851
In [58]:
cookie = wordvecs[np.where(vocab=="nerd")[0][0],:]
scores = np.array([  cossim(wordvecs[i,:], cookie) for i in range(400000)  ])
for i in np.argsort(-scores)[:30]:
    print(i, vocab[i],  cossim(wordvecs[i,:],  cookie))
35465 nerd 1.0
26309 geek 0.8293729096567657
44144 slacker 0.8090723871724442
58089 geeky 0.7969193370608488
29931 crazed 0.7569688157208042
47541 nerdy 0.7433485866982084
86555 curmudgeon 0.7201585094656451
96817 dorky 0.697263024686298
34082 geeks 0.696192619699131
4313 kid 0.6950220419462729
33201 gamer 0.6936665222970874
27401 fanatic 0.6884175011489712
13436 mentality 0.6786048674653181
63449 scrawny 0.675444114807518
61248 gizmo 0.675061173756852
65316 gunslinger 0.6668263550512463
26436 cheerleader 0.6654754514214731
24737 gadget 0.6635588542490874
50127 yuppie 0.661955757500423
44895 nerds 0.655154160883919
88432 techie 0.6516363275250835
93930 dork 0.6506518672656352
12594 obsessed 0.6474508306931893
43645 archetype 0.6440522960792431
58701 wisecracking 0.6417392135467914
16689 instinct 0.6412341448745975
49153 hipster 0.6404485715650188
21110 stereotype 0.6395275993326107
78844 waif 0.6391300010438783
94896 gawky 0.638806184430464