diff --git a/Chatbot.py b/Chatbot.py index e24115a..4bc1423 100644 --- a/Chatbot.py +++ b/Chatbot.py @@ -14,7 +14,7 @@ class Chatbot(object): np.load('markov_dict.npy') except FileNotFoundError: #If there's no initial file, make a new one: - initFile = {'Hoi':{self.END_OF_MESSAGE : 1}} + initFile = {'pairs':{'Hoi':{self.END_OF_MESSAGE : 1}}, 'trairs':{}} np.save("markov_dict.npy", initFile) except: print("ERROR! - Could not open or create Markov chat file") @@ -28,25 +28,45 @@ class Chatbot(object): for i in range(len(corpus)-1): yield (corpus[i], corpus[i+1]) + #Make word trairs. Yes, I completely made up "trair" + #It's like pairs, but with three, but not really, because there are two. OK? + #[1 0 1] + def makeTrairs(self, corpus): + for i in range(1, len(corpus)-1): + yield (corpus[i-1], corpus[i+1]) + #Learn from a human message #Returns an updated dict - def learnFrom(self, initialWordDict, message: str): - newWordDict = initialWordDict + def learnFrom(self, wordDict, message: str): + newPairDict = wordDict['pairs'] + newTrairDict = wordDict['trairs'] message = message + " " + self.END_OF_MESSAGE corpus = message.split() pairs = self.makePairs(corpus) + trairs = self.makeTrairs(corpus) + print(trairs) - #Add wordcounts to dict + #Add wordcounts to pair dict for word_1, word_2 in pairs: - if word_1 in newWordDict.keys(): - if word_2 in newWordDict[word_1].keys(): - newWordDict[word_1][word_2] += 1 + if word_1 in newPairDict.keys(): + if word_2 in newPairDict[word_1].keys(): + newPairDict[word_1][word_2] += 1 else: - newWordDict[word_1][word_2] = 1 + newPairDict[word_1][word_2] = 1 else: - newWordDict[word_1] = {word_2 : 1} - return newWordDict; + newPairDict[word_1] = {word_2 : 1} + #Add wordcounts to trair dict + for word_1, word_3 in trairs: + if word_1 in newTrairDict.keys(): + if word_3 in newTrairDict[word_1].keys(): + newTrairDict[word_1][word_3] += 1 + else: + newTrairDict[word_1][word_3] = 1 + else: + newTrairDict[word_1] = {word_3 : 1} + + return {'pairs':newPairDict, 'trairs':newTrairDict}; @commands.command(pass_context=True, hidden=True) @asyncio.coroutine @@ -91,6 +111,8 @@ class Chatbot(object): #Get dictionary/Markov chain wordDict = np.load('markov_dict.npy').item() + wordPairDict = wordDict['pairs'] + wordTrairDict = wordDict['trairs'] #TODO store in global variable to minimise IO? #Delete command @@ -103,19 +125,29 @@ class Chatbot(object): #Pick a random first word (Which is not an end-of-message) #TODO pick a word which responds to the previous message in the channel - first_word = np.random.choice(list(wordDict.keys())) + first_word = np.random.choice(list(wordPairDict.keys())) while first_word == self.END_OF_MESSAGE: #or first_word.islower() ??? - first_word = np.random.choice(list(wordDict.keys())) + first_word = np.random.choice(list(wordPairDict.keys())) chain = [first_word] #Go through the chain: - for i in range(MAX_N_WORDS): + for i in range(1, MAX_N_WORDS): if chain[-1] == self.END_OF_MESSAGE: break + + #Find the possibilities: + pairPossibilities = wordPairDict[chain[-1]].copy() + if i > 1: #we can only compare trairs if we have at least two words already + if chain[-2] in wordTrairDict: + trairPossibilities = wordTrairDict[chain[-2]].copy() + #If a trair matches, increase the frequency of the pair + for word2 in pairPossibilities.keys(): + if word2 in trairPossibilities.keys(): + pairPossibilities[word2] += trairPossibilities[word2] #Calculate probabilities and pick a next word: #https://stackoverflow.com/questions/835092/python-dictionary-are-keys-and-values-always-the-same-order - possible_words = list(wordDict[chain[-1]].keys()) - weights = np.array(list(wordDict[chain[-1]].values())) + possible_words = list(pairPossibilities.keys()) + weights = np.array(list(pairPossibilities.values())) probs = weights / weights.sum() chain.append(np.random.choice(possible_words, p=probs))