{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"# IMet Word Embedding\nHere are my tests for using NLP to encode labels.\nHope you find it helpful."},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm_notebook as tqdm\n\nGLOVE = '../input/glove840b300dtxt/glove.840B.300d.txt' #'../input/glove.840B.300d.txt'\nLABELS = '../input/imet-2019-fgvc6/labels.csv'\nTRAIN = '../input/imet-2019-fgvc6/train.csv'\nTRAIN_IMG = '../input/imet-2019-fgvc6/train/{}.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_coefs(word, *arr):\n    return word, np.asarray(arr, dtype='float32')\ndef load_embeddings(path):\n    with open(path) as f:\n        return dict(get_coefs(*line.strip().split(' ')) for line in tqdm(f))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"all_words = [\"abruzzi\",\"achaemenid\",\"aegean\",\"afghan\",\"after british\",\"after german\",\"after german original\",\"after italian\",\"after russian original\",\"akkadian\",\"alexandria-hadra\",\"algerian\",\"alsace\",\"american\",\"american or european\",\"amsterdam\",\"ansbach\",\"antwerp\",\"apulian\",\"arabian\",\"aragon\",\"arica\",\"asia minor\",\"assyrian\",\"atlantic watershed\",\"attic\",\"augsburg\",\"augsburg decoration\",\"augsburg original\",\"austrian\",\"avignon\",\"avon\",\"aztec\",\"babylonian\",\"babylonian or kassite\",\"bactria-margiana archaeological complex\",\"balinese\",\"bavaria\",\"bayreuth\",\"beautiran\",\"beauvais\",\"belgian\",\"berlin\",\"birmingham\",\"boeotian\",\"bohemian\",\"bologna\",\"bordeaux\",\"bow\",\"brescia\",\"bristol\",\"british\",\"british or french\",\"british or scottish\",\"brunswick\",\"brussels\",\"burma\",\"burslem\",\"byzantine\",\"calima\",\"cambodia\",\"campanian\",\"canaanite\",\"canosan\",\"castel durante\",\"catalan\",\"catalonia\",\"caucasian\",\"caughley\",\"central asia\",\"central european\",\"central highlands\",\"central italian\",\"chalcidian\",\"chantilly\",\"chaumont-sur-loire\",\"chelsea\",\"chelsea-derby\",\"chimu\",\"china\",\"chinese with dutch decoration\",\"chinese with european decoration\",\"chinese with french mounts\",\"chiriqui\",\"chorrera\",\"chupicuaro\",\"colima\",\"colombian\",\"colonial\",\"colonial american\",\"copenhagen\",\"corinthian\",\"coromandel coast\",\"costa rica\",\"costa rica or panama\",\"cretan\",\"crete\",\"cyclades\",\"cycladic\",\"cypriot\",\"cypriot or phoenician\",\"czech\",\"danish\",\"deccan\",\"dehua\",\"delft\",\"derby\",\"deruta\",\"devonshire\",\"dresden\",\"dublin\",\"dutch\",\"dyak\",\"east greek\",\"east greek/sardis\",\"eastern european\",\"eastern mediterranean\",\"eastern mediterranean or italian\",\"edinburgh\",\"edomite\",\"egypt\",\"egyptian\",\"elamite\",\"england\",\"etruria\",\"etruscan\",\"euboean\",\"european\",\"european bronze age\",\"faliscan\",\"ferrara\",\"flemish\",\"flemish or italian\",\"florence\",\"for american market\",\"for british market\",\"for continental market\",\"for danish market\",\"for european market\",\"for french market\",\"for iberian market\",\"for portuguese market\",\"for russian market\",\"for swedish market\",\"frankenthal\",\"frankish\",\"freiburg im breisgau\",\"french\",\"french or german\",\"french or italian\",\"french or swiss\",\"fulda\",\"furstenberg\",\"gaul\",\"geneva\",\"genoa\",\"german\",\"german or swiss\",\"ghassulian\",\"gnathian\",\"gonia\",\"greek\",\"greek islands\",\"greek or roman\",\"guanacaste-nicoya\",\"gubbio\",\"gurkha\",\"haida\",\"hanau\",\"hattian\",\"helladic\",\"hilt\",\"hittite\",\"hochst\",\"huastec\",\"hungarian\",\"hungary\",\"huron\",\"ica\",\"inca\",\"india\",\"indian or nepalese\",\"indonesia\",\"inuit\",\"iran\",\"irish\",\"isin-larsa\",\"isin-larsaold babylonian\",\"islamic\",\"italian\",\"italian or sicilian\",\"italian or spanish\",\"italic\",\"italic-native\",\"japan\",\"javanese\",\"jouy-en-josas\",\"kathmandu valley\",\"kazakhstan\",\"kholmogory\",\"kievan rus'\",\"konigsberg\",\"korea\",\"la rochelle\",\"laconian\",\"lambayeque\",\"lambeth\",\"langobardic\",\"leuven\",\"lille\",\"limoges\",\"liverpool\",\"london\",\"london original\",\"longton hall\",\"lowestoft\",\"ludwigsburg\",\"lydian\",\"lyons\",\"macao\",\"macaracas\",\"macedonian\",\"madrid\",\"malayan\",\"mali\",\"manteno\",\"maya\",\"meissen\",\"meissen with german\",\"mennecy\",\"mennecy or sceaux\",\"mexican\",\"mezcala\",\"michoacan\",\"milan\",\"mimbres\",\"minoan\",\"mitanni\",\"mixtec\",\"moche\",\"moche-wari\",\"montelupo\",\"moro\",\"moroccan\",\"moustiers\",\"mughal\",\"muisca\",\"munich\",\"mycenaean\",\"nabataean\",\"nailsea\",\"nantes\",\"naples\",\"nasca\",\"naxos\",\"nayarit\",\"neo-sumerian\",\"neolithic\",\"nepal\",\"netherlandish\",\"neuwied am rhein\",\"nevers\",\"nimes\",\"north china\",\"north indian\",\"north italian\",\"north netherlandish\",\"northern european\",\"northern india\",\"northern italian\",\"northwest china\",\"northwest china/eastern central asia\",\"norwegian\",\"nuremberg\",\"nymphenburg\",\"old assyrian trading colony\",\"olmec\",\"orleans\",\"ottonian\",\"padua\",\"pakistan\",\"palermo\",\"paracas\",\"paris\",\"parita\",\"parthian\",\"parthian or sasanian\",\"peruvian\",\"pesaro\",\"philippine\",\"phrygian\",\"piedmont\",\"polish\",\"populonia\",\"portuguese\",\"potsdam\",\"praenestine\",\"proto-elamite\",\"provincial\",\"ptolemaic\",\"qajar\",\"quechua\",\"remojadas\",\"rhenish\",\"roman\",\"roman egyptian\",\"rome\",\"rouen\",\"russian\",\"saint-cloud\",\"salinar\",\"salzburg\",\"san sabastian\",\"sasanian\",\"savoy\",\"saxony\",\"scandinavian\",\"sceaux\",\"scottish\",\"scythian\",\"seleucid\",\"seville\",\"sevres\",\"sheffield\",\"sicily\",\"siena\",\"silesia\",\"sinceny\",\"skyros\",\"smyrna\",\"south german\",\"south italian\",\"south netherlandish\",\"southall\",\"southern german\",\"spanish\",\"spitalfields\",\"sri lankan\",\"st. petersburg\",\"staffordshire\",\"stockholm\",\"stoke-on-trent\",\"strasbourg\",\"sulawesi\",\"sumatran\",\"sumerian\",\"surrey\",\"swedish\",\"swiss\",\"syrian\",\"tairona\",\"tarentine\",\"teano\",\"teotihuacan\",\"thailand\",\"thanjavur\",\"the hague\",\"thessaly\",\"thuringia\",\"tibet\",\"tibetan\",\"tiwanaku\",\"tlatilco\",\"tlingit\",\"tolita-tumaco\",\"topara\",\"tsimshian\",\"turin\",\"turkish\",\"turkish or venice\",\"ubaid\",\"umbria\",\"united states\",\"unknown\",\"urartian\",\"urbino\",\"urbino with gubbio luster\",\"valencia\",\"venice\",\"veracruz\",\"veraguas\",\"verona\",\"versailles\",\"vienna\",\"vietnam\",\"villanovan\",\"villeroy\",\"vincennes\",\"visigothic\",\"vulci\",\"wari\",\"west slavic\",\"western european\",\"worcester\",\"wurzburg\",\"zenu\",\"zoroastrian\",\"zurich\",\"abbies\",\"abraham\",\"abstraction\",\"acanthus\",\"acorns\",\"acrobats\",\"actors\",\"actresses\",\"adam\",\"admirals\",\"adonis\",\"adoration of the magi\",\"adoration of the sheperds\",\"air transports\",\"alexander the great\",\"altars\",\"amazons\",\"amulets\",\"amun\",\"ancient greek\",\"angels\",\"anger\",\"animals\",\"anklet\",\"annunciation\",\"aphrodite\",\"apocalypse\",\"apollo\",\"apostles\",\"apples\",\"arabic\",\"archangel gabriel\",\"arches\",\"architects\",\"architectural elements\",\"architectural fragments\",\"architecture\",\"ariadne\",\"armors\",\"army\",\"arrowheads\",\"arrows\",\"artemis\",\"artists\",\"assumption of the virgin\",\"astronomy\",\"athena\",\"athletes\",\"autumn\",\"avalokiteshvara\",\"axes\",\"bacchus\",\"badges\",\"bagpipes\",\"bakers\",\"balconies\",\"bamboo\",\"baptism of christ\",\"barns\",\"baseball\",\"basins\",\"bathing\",\"bathsheba\",\"bats\",\"battles\",\"beaches\",\"beads\",\"beakers\",\"bears\",\"bedrooms\",\"beds\",\"bees\",\"belts\",\"benches\",\"benjamin franklin\",\"bes\",\"bible\",\"bicycles\",\"billiards\",\"birds\",\"bishops\",\"boars\",\"boats\",\"bobbins\",\"bodhisattva\",\"bodies of water\",\"body parts\",\"books\",\"boots\",\"bottles\",\"bow and arrow\",\"bowls\",\"boxes\",\"boxing\",\"boys\",\"bracelets\",\"bridges\",\"brooches\",\"buckles\",\"buddha\",\"buddhism\",\"buddhist religious figures\",\"buffalos\",\"buildings\",\"buildings and structures\",\"bulls\",\"burial grounds\",\"burials\",\"butterflies\",\"buttons\",\"cabinets\",\"calendars\",\"camels\",\"cameos\",\"canals\",\"candelabra\",\"candles\",\"candlesticks\",\"cannons\",\"capitals\",\"carpets and rugs\",\"carriages\",\"cartouches\",\"caryatids\",\"castles\",\"cathedrals\",\"cats\",\"cauldrons\",\"caves\",\"celestial bodies\",\"censers\",\"centaurs\",\"ceremony\",\"ceres\",\"chairs\",\"chalices\",\"chariots\",\"chess\",\"chests\",\"chickens\",\"children\",\"chinese\",\"chinoiserie\",\"christ\",\"christian imagery\",\"christianity\",\"christmas\",\"churches\",\"circles\",\"circus\",\"cities\",\"civil war\",\"cleopatra\",\"clocks\",\"clothing and accessories\",\"clouds\",\"coat of arms\",\"coats\",\"coffeepots\",\"coffins\",\"coins\",\"columns\",\"commodes\",\"concerts\",\"contemplation\",\"coptic\",\"cornucopia\",\"corpses\",\"correspondence\",\"corsets\",\"costumes\",\"couches\",\"couples\",\"courtyards\",\"coverlets and quilts\",\"cows\",\"crabs\",\"cradles\",\"cranes\",\"crescents\",\"crocodiles\",\"cross\",\"crowd\",\"crucifixion\",\"cuneiform\",\"cupid\",\"cups\",\"curtains\",\"cutlery\",\"daggers\",\"daily life\",\"daisies\",\"dance\",\"dancers\",\"dancing\",\"david\",\"dawn\",\"death\",\"decorative designs\",\"decorative elements\",\"deer\",\"deities\",\"demons\",\"descent from the cross\",\"deserts\",\"design elements\",\"desks\",\"devil\",\"diadems\",\"diamonds\",\"diana\",\"dice\",\"dining\",\"dionysus\",\"dishes\",\"docks\",\"doctors\",\"documents\",\"dogs\",\"dolls\",\"dolphins\",\"domes\",\"donkeys\",\"doors\",\"doorways\",\"doves\",\"dragons\",\"drawing\",\"dresses\",\"drinking\",\"drinking glasses\",\"drums\",\"drunkenness\",\"ducks\",\"durga\",\"eagles\",\"earrings\",\"easter\",\"egg and dart\",\"elephants\",\"emblems\",\"embroidery\",\"emperor augustus\",\"entombment\",\"eros\",\"esther\",\"europa\",\"eve\",\"evening\",\"ewers\",\"eyes\",\"facades\",\"faces\",\"factories\",\"fairies\",\"falcons\",\"family\",\"fans\",\"farmers\",\"farms\",\"fathers\",\"fauns\",\"fear\",\"feathers\",\"feet\",\"female nudes\",\"fire\",\"firearms\",\"fireplaces\",\"fireworks\",\"fish\",\"fishing\",\"flags\",\"flowers\",\"flutes\",\"fluting\",\"food\",\"footwear\",\"forests\",\"fortification\",\"fountains\",\"foxes\",\"friezes\",\"frogs\",\"fruit\",\"funerals\",\"funerary objects\",\"furniture\",\"gadrooning\",\"galatea\",\"games\",\"gardeners\",\"gardens\",\"garlands\",\"gates\",\"generals\",\"genre scene\",\"geometric patterns\",\"george washington\",\"gingham pattern\",\"girls\",\"globes\",\"gloves\",\"goats\",\"goblets\",\"goddess\",\"gods\",\"grapes\",\"greek deities\",\"greek figures\",\"griffins\",\"grotesques\",\"guitars\",\"hair\",\"hammers\",\"hands\",\"harps\",\"hathor\",\"hats\",\"hawks\",\"heads\",\"hell\",\"helmets\",\"hercules\",\"hermes\",\"hexagons\",\"hieroglyphs\",\"hills\",\"hilts\",\"hindu religious figures\",\"hinduism\",\"historical figures\",\"holofernes\",\"holy family\",\"horns\",\"horse riding\",\"horses\",\"horus\",\"hospitals\",\"houses\",\"human figures\",\"hunting\",\"illness\",\"incense burners\",\"infants\",\"inns\",\"inscriptions\",\"insects\",\"insignia\",\"interiors\",\"isis\",\"jackets\",\"jainism\",\"jars\",\"jason\",\"jesus\",\"jewelry\",\"jockeys\",\"journals\",\"judith\",\"jugs\",\"julius caesar\",\"juno\",\"jupiter\",\"kettles\",\"keys\",\"kings\",\"kitchens\",\"knives\",\"krishna\",\"lace\",\"ladders\",\"ladles\",\"lakes\",\"lambs\",\"lamentation\",\"lamps\",\"landforms\",\"landscapes\",\"last judgement\",\"last supper\",\"law\",\"leaves\",\"leda\",\"leopards\",\"lighting\",\"lions\",\"literature\",\"liturgical objects\",\"living rooms\",\"lizards\",\"lobsters\",\"lockets\",\"lotuses\",\"louis xiv\",\"love\",\"lovers\",\"lutes\",\"madonna and child\",\"maenads\",\"magicians\",\"maitreya\",\"male nudes\",\"mandolins\",\"manjushri\",\"manuscripts\",\"maps\",\"mark antony\",\"markets\",\"mars\",\"mary magdalene\",\"masks\",\"massacres\",\"medallions\",\"medea\",\"men\",\"merchants\",\"mercury\",\"mice\",\"military\",\"military clothing\",\"military equipment\",\"minerva\",\"mirrors\",\"monkeys\",\"monks\",\"monsters\",\"monuments\",\"moon\",\"moses\",\"mosques\",\"mothers\",\"mountains\",\"muses\",\"music\",\"musical instruments\",\"musicians\",\"mythical creatures\",\"mythology\",\"napoleon i\",\"nativity\",\"navy\",\"necklaces\",\"necktie\",\"neptune\",\"nero\",\"netsuke\",\"new testament\",\"night\",\"nike\",\"nonrepresentational art\",\"nymphs\",\"obelisks\",\"occupations\",\"octagons\",\"octopus\",\"old testament\",\"olive trees\",\"opera\",\"organs\",\"ornament\",\"orpheus\",\"owls\",\"painting\",\"paisley\",\"palaces\",\"palmettes\",\"pants\",\"parks\",\"parrots\",\"party\",\"peaches\",\"peacocks\",\"pediments\",\"pendants\",\"pentecost\",\"peonies\",\"percussion instruments\",\"performance\",\"perseus\",\"pheasants\",\"pianos\",\"pigeons\",\"pigs\",\"pilasters\",\"pinecones\",\"pins\",\"pitchers\",\"plants\",\"playing\",\"playing cards\",\"pocket watches\",\"poetry\",\"poets\",\"polka-dot pattern\",\"pomegranates\",\"ponds\",\"popes\",\"portraits\",\"poseidon\",\"princes\",\"princesses\",\"prisms\",\"prisoners\",\"prisons\",\"profiles\",\"prostitutes\",\"psyche\",\"punishment\",\"purses\",\"putti\",\"pyramids\",\"queens\",\"qur'an\",\"rabbits\",\"railways\",\"rain\",\"rams\",\"reading\",\"rectangles\",\"religious events\",\"religious texts\",\"reliquaries\",\"riding\",\"rings\",\"rivers\",\"roads\",\"robes\",\"roman deities\",\"roosters\",\"rosaries\",\"roses\",\"rowing\",\"ruins\",\"sadness\",\"sailors\",\"saint anne\",\"saint anthony\",\"saint catherine\",\"saint francis\",\"saint george\",\"saint jerome\",\"saint john the baptist\",\"saint john the evangelist\",\"saint joseph\",\"saint lawrence\",\"saint mark\",\"saint matthew\",\"saint michael\",\"saint paul\",\"saint peter\",\"saints\",\"samples\",\"sarcophagus\",\"satire\",\"satyrs\",\"saucers\",\"scarabs\",\"scarves\",\"schools\",\"scorpions\",\"screens\",\"scrolls\",\"sculpture\",\"seals\",\"seas\",\"seascapes\",\"seating furniture\",\"self-portraits\",\"serpents\",\"servants\",\"shakespeare\",\"shakyamuni\",\"sheep\",\"shells\",\"shepherds\",\"shields\",\"ships\",\"shirts\",\"shiva\",\"shoes\",\"sibyl\",\"silenus\",\"singers\",\"singing\",\"skeletons\",\"skirts\",\"skulls\",\"sky\",\"slavery\",\"sleep\",\"smoking\",\"snails\",\"snakes\",\"snow\",\"soldiers\",\"spears\",\"spectators\",\"sphinx\",\"sports\",\"spring\",\"squares\",\"squirrels\",\"stairs\",\"stars\",\"still life\",\"stools\",\"storage furniture\",\"storms\",\"strapwork\",\"street scene\",\"streets\",\"stripes\",\"students\",\"suffering\",\"suits\",\"summer\",\"sun\",\"sundials\",\"sunflowers\",\"swans\",\"sword guards\",\"swords\",\"tabernacles\",\"tables\",\"tablets\",\"taoism\",\"tapestries\",\"taweret\",\"tea caddy\",\"tea drinking\",\"teachers\",\"teapots\",\"telescopes\",\"temples\",\"tents\",\"textile fragments\",\"textiles\",\"theatre\",\"tigers\",\"tombs\",\"tools and equipment\",\"towers\",\"towns\",\"toys\",\"trains\",\"transportation\",\"trays\",\"trees\",\"triangles\",\"tricorns\",\"triton\",\"trophies\",\"trumpets\",\"tulips\",\"tunics\",\"tureens\",\"turtles\",\"undergarment\",\"uniforms\",\"urns\",\"utilitarian objects\",\"vajrapani\",\"vase fragments\",\"vases\",\"vegetables\",\"venus\",\"vestments\",\"vests\",\"victory\",\"villages\",\"vines\",\"violas\",\"violins\",\"virgin mary\",\"vishnu\",\"volcanoes\",\"vulcan\",\"wagons\",\"walking\",\"wars\",\"washing\",\"watches\",\"waterfalls\",\"watermills\",\"waves\",\"weapons\",\"weights and measures\",\"wells\",\"wind\",\"windmills\",\"windows\",\"wine\",\"winter\",\"women\",\"working\",\"world war i\",\"worshiping\",\"wreaths\",\"writing\",\"writing implements\",\"writing systems\",\"zeus\",\"zigzag pattern\",\"zodiac\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_matrix(all_words, path, statistics=True, embedding_index=None):\n    if embedding_index is None: embedding_index = load_embeddings(path)\n    \n    if statistics:\n        good_words = 0\n        bad_words = 0\n        for word in all_words:\n            try:\n                embedding_index[word]\n                good_words = good_words+1\n            except Exception as e:\n                bad_words = bad_words+1\n        print(\"good {}, bad {}, percent {}\".format(good_words, bad_words, good_words/(good_words+bad_words)))\n    \n    embedding_matrix = dict()\n    unknown_words = []\n    text = \"\"\n    \n    for i, word in enumerate(all_words):\n        try:\n            embedding_matrix[word] = (np.array(embedding_index[word])).tolist()\n        except KeyError:\n            if \" \" in word or \"-\" in word:\n                try:\n                    words = word.replace(\"for \", \"\").replace(\" or \", \" \").replace(\" and \", \" \").replace(\" with \", \" \").replace(\" of \", \" \").replace(\"the \", \" \").split(\" \")\n                    while \"\" in words:\n                        words.remove(\"\")\n                    embedding_matrix[word] = (np.array([embedding_index[word] for word in words]).mean(axis=0)).tolist()\n#                     text = text + \"\\n\" + str(words)\n                    continue\n                except KeyError:\n                    try:\n                        words = (word.replace(\"for \", \"\").replace(\" or \", \" \").replace(\" and \", \" \").replace(\" with \", \" \").replace(\" of \", \" \").replace(\"the \", \" \").replace(\"'\", \"\").replace(\"/\", \" \") + \"[]\").replace(\"ish[]\", \"[]\").replace(\"s[]\", \"[]\").replace(\"[]\", \"\").split(\" \")\n                        while \"\" in words:\n                            words.remove(\"\")\n                        embedding_matrix[word] = (np.array([embedding_index[word] for word in words]).mean(axis=0)).tolist()\n                        text = text + \"\\n\" + str(words)\n                        continue\n                    except KeyError:\n                        try:\n                            words = (word.replace(\"for \", \"\").replace(\" or \", \" \").replace(\" and \", \" \").replace(\" with \", \" \").replace(\" of \", \" \").replace(\"the \", \" \").replace(\"'\", \"\").replace(\"/\", \" \") + \"[]\").replace(\"ish[]\", \"[]\").replace(\"s[]\", \"[]\").replace(\"[]\", \"\").replace(\"-\", \" \").split(\" \")\n                            while \"\" in words:\n                                words.remove(\"\")\n                            embedding_matrix[word] = (np.array([embedding_index[word] for word in words]).mean(axis=0)).tolist()\n                            text = text + \"\\n\" + str(words)\n                            continue\n                        except KeyError:\n                                try:\n                                    words = (word.replace(\"for \", \"\").replace(\" or \", \" \").replace(\" and \", \" \").replace(\" with \", \" \").replace(\" of \", \" \").replace(\"the \", \" \").replace(\"'\", \"\").replace(\"/\", \" \") + \"[]\").replace(\"ish[]\", \"[]\").replace(\"s[]\", \"[]\").replace(\"[]\", \"\").replace(\"-\", \"\").split(\" \")\n                                    while \"\" in words:\n                                        words.remove(\"\")\n                                    embedding_matrix[word] = (np.array([embedding_index[word] for word in words]).mean(axis=0)).tolist()\n                                    text = text + \"\\n\" + str(words)\n                                    continue\n                                except KeyError:\n                                    unknown_words.append(word)\n                                    continue\n                        continue\n            elif \"ish\" in word or \"s\" or \"''\" in word:\n                try:\n                    words = (word+\"[]\").replace(\"ish[]\", \"[]\").replace(\"s[]\", \"[]\").replace(\"[]\", \"\").replace(\"'\", \"\").split(\" \")\n                    while \"\" in words:\n                        words.remove(\"\")\n                    embedding_matrix[word] = (np.array([embedding_index[word] for word in words]).mean(axis=0)).tolist()\n                    text = text + \"\\n\" + str(words)\n                    continue\n                except KeyError:\n                    unknown_words.append(word)\n                    continue\n            unknown_words.append(word)\n    print(text)\n    return embedding_matrix, unknown_words\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embedding_index = load_embeddings(GLOVE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embedding_matrix, unknown_words = build_matrix(all_words, GLOVE, embedding_index=embedding_index)\ngood_count = len(embedding_matrix.keys())\nbad_count = len(unknown_words)\nprint(\"good {}, bad {}, percent {}\".format(good_count, bad_count, good_count/(good_count + bad_count)))\n# print(\"good {}, bad {}, percent {}\".format(len(embedding_matrix.keys())-len(unknown_words), len(unknown_words), (len(embedding_matrix)-len(unknown_words))/(len(embedding_matrix)+len(unknown_words))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unknown_words","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embedding_index['quran']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from scipy.spatial.distance import cosine\n\nprint(cosine(embedding_index['english'], embedding_index['chinese']), cosine(embedding_index['english'], embedding_index['atom']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import operator\n\ndef find_similar_word(target):\n    dic = dict()\n    if type(target) == str: target = embedding_matrix[target]\n    for word in embedding_matrix.keys():\n        dic[word] = cosine(target, embedding_matrix[word])\n    return sorted(dic.items(), key=operator.itemgetter(1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"find_similar_word(\"female nudes\")[:10] # well it is similar to virgin mary. The Russians love female nudes too!","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Russian to Nudes: {}, Chinese to Nudes: {}\".format(cosine(embedding_index[\"russian\"], embedding_index[\"nudes\"]), cosine(embedding_index[\"chinses\"], embedding_index[\"nudes\"])))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"find_similar_word(\"european bronze age\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"find_similar_word(\"zoroastrian\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# LABELS\nlabels = pd.read_csv(LABELS)\nlabels_attribute_id = [int(ids) for ids in labels.attribute_id]\nlabels_attribute_name = [name.replace(\"culture::\", \"\").replace(\"tag::\", \"\") for name in labels.attribute_name]\nids_2_names = dict(zip(list(labels_attribute_id), list(labels_attribute_name)))\nnames_2_ids = dict(zip(list(labels_attribute_name), list(labels_attribute_id)))\n\n# TRAIN\ntrain = pd.read_csv(TRAIN)\ntrain_ids = train.id\ntrain_attribute_ids = [   [int(i) for i in img_string.split(\" \")]   for img_string in train.attribute_ids]\ntrain_attribute_name = [   [ids_2_names[id]for id in ids]   for ids in train_attribute_ids]\n\n# fake embeddings for NaN targets\nfor name in names_2_ids.keys():\n    if name not in embedding_matrix.keys():\n        embedding_matrix[name] = np.zeros(300).tolist()\n\ntrain_attribute_embed = [   (np.array([embedding_matrix[name] for name in names]).sum(axis=0)/15).tolist()    for names in train_attribute_name]\ntrain_attribute_name[:2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_attribute_embed = [   (np.array([embedding_matrix[name] for name in names]).sum(axis=0)/15).tolist()    for names in train_attribute_name]\n\n\n\n# ids_2_embed = dict(zip(list(train_attribute_ids), list(train_attribute_embed)))\n# train_attribute_name[:2], train_attribute_embed[:2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\n\n%matplotlib inline\nimport matplotlib\nimport matplotlib.pyplot as plt\n\ndef find_similar_image(target, train_attribute_ids, train_attribute_embed):\n    target_embed = train_attribute_embed[train_ids.values.tolist().index(target)]\n    scores = []\n    ids = []\n    for i, embed in enumerate(train_attribute_embed):\n        ids.append(train_ids[i])\n        scores.append(cosine(embed, target_embed))\n    dic = dict(zip(ids, scores))\n    sort = sorted(dic.items(), key=operator.itemgetter(1))\n    return dict(sort)\n\n# def show_images(list_images_path):\n#     from IPython.display import Image, display\n#     for imageName in list_images_path:\n#         display(Image(filename=imageName))\n\n# credit: https://gist.github.com/soply/f3eec2e79c165e39c9d540e916142ae1\ndef show_images(list_images_path, cols = 1, titles = None):\n    \"\"\"Display a list of images in a single figure with matplotlib.\n    \n    Parameters\n    ---------\n    images: List of np.arrays compatible with plt.imshow.\n    \n    cols (Default = 1): Number of columns in figure (number of rows is \n                        set to np.ceil(n_images/float(cols))).\n    \n    titles: List of titles corresponding to each image. Must have\n            the same length as titles.\n    \"\"\"\n    images = [cv2.imread(img) for img in list_images_path]\n    assert((titles is None)or (len(images) == len(titles)))\n    n_images = len(images)\n    if titles is None: titles = ['Image (%d)' % i for i in range(1,n_images + 1)]\n    fig = plt.figure()\n    for n, (image, title) in enumerate(zip(images, titles)):\n        a = fig.add_subplot(cols, np.ceil(n_images/float(cols)), n + 1)\n        if image.ndim == 2:\n            plt.gray()\n        plt.imshow(image)\n        a.set_title(title)\n    fig.set_size_inches(np.array(fig.get_size_inches()) * n_images)\n    plt.show()\n\n# print(find_similar_image(\"1000483014d91860\", train_attribute_ids, train_attribute_embed))\ndic = find_similar_image(\"1000483014d91860\", train_attribute_ids, train_attribute_embed)\nshow_images([TRAIN_IMG.format(img) for img in list(dic.keys())[:10]], cols=2, titles=list(dic.values())[:10])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_images([TRAIN_IMG.format(img) for img in list(dic.keys())[10:20]], cols=2, titles=list(dic.values())[10:20])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dic = find_similar_image(\"101c8394ff6db02d\", train_attribute_ids, train_attribute_embed)\nprint(list(dic.values())[:10])\nshow_images([TRAIN_IMG.format(img) for img in list(dic.keys())[:10]], cols=2, titles=list(dic.values())[:10])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}