Commit ad4e8e75 authored by Victor Zimmermann's avatar Victor Zimmermann
Browse files

Added fully functional graph class, can now extract senses for given words, not yet disambiguate

parent 909c4e12
Loading
Loading
Loading
Loading
+123 −66
Changes for code/absinth.py: 123 added lines, 66 removed lines.
Original line number Diff line number Diff line
import os
import spacy
import networkx as nx
import matplotlib.pyplot as plt
nlp = spacy.load('en')

import os # for reading files
from tqdm import tqdm # for counting seconds
import spacy # for nlp
import networkx as nx # for visualisation
import matplotlib.pyplot as plt # for visualisation
import copy # for deepcopy
import numpy as np # for calculations
nlp = spacy.load('en') # standard english nlp

# wrapper class for nodes + functions on nodes
class Graph:

    # can be initialised with nodes
    def __init__(self, nodes = {}):
        self.nodes = nodes
    
    # 'key in Graph' returns True if node with key exists in Graph 
    def __contains__(self, key):
        return key in self.nodes.keys()
    
    # returns all nodes (not keys)
    def get_nodes(self):
        return self.nodes.values()
    
    def add_node(self, token):
        if token in self:
            self.nodes[token].freq += 1
    # adds node or ups frequency of node if already in graph
    def add_node(self, key):
        if key in self:
            self.nodes[key].freq += 1
        else:
            self.nodes[token] = Node(token)
            self.nodes[key] = Node(key)
    
    def remove_node(self, token):
        del self.nodes[token]
        for node in self.nodes.values():
            print('tada')
            node.remove_neighbor(token)
    # removes node (doesn't work)
    #def remove_node(self, key):
    #    del self.nodes[key]
    #    for node in self.nodes.values():
    #        node.remove_neighbor(key)
    
    def add_edge(self, from_token, to_token):
        self.nodes[from_token].add_neighbor(self.nodes[to_token])
    # adds neighbor to node
    def add_edge(self, from_key, to_key):
        self.nodes[from_key].add_neighbor(self.nodes[to_key])
    
    def build(self, corpus_path, word, filter_dict):
    #builds graph from corpus for target word with applied filters
    #filters: min_occurrences, min_cooccurrence, stop_words, allowed_tags, context_size, max_distance
    def build(self, corpus_path, word, filters):
        
        files = [corpus_path+'/'+f for f in os.listdir(corpus_path)]
        spaced_word = word.replace('_', ' ') #input words are seperated with underscores
        files = [corpus_path+'/'+f for f in os.listdir(corpus_path)] # list of file paths (note that no other files should be in this directory)
        spaced_word = word.replace('_', ' ') #input words are generally seperated with underscores
        
        for f in files:
        for f in tqdm(files[:]): #iterates over corpus
            with open(f, 'r') as source:
                
                try:
                try: #some decoding throws the iteration
                    for line in source:
                        line = line.lower()
                        if spaced_word in line:
                        if spaced_word in line: #greedy filter (no processing on most lines)
                            new_line = line.replace(spaced_word, word)
                            spacy_line = nlp(new_line)
                            if word in [token.text for token in spacy_line]:
                                tokens = list()
                            if word in [token.text for token in spacy_line]: #detailed filter on tokenised line
                                tokens = list() #collects tokens
                                for token in spacy_line:
                                    text = token.text
                                    tag = token.tag_
                                    if text != word and text not in filter_dict['stop_words'] and tag in filter_dict['allowed_tags'] :
                                    # if token is not a stop word and has right pos tag
                                    if text != word and text not in filters['stop_words'] and tag in filters['allowed_tags'] :
                                        tokens.append(token.text)
                                if len(tokens) >= filter_dict['context_size']:
                                    for node in tokens:
                                        self.add_node(node)
                                    for edge in [(x,y) for x in tokens for y in tokens]:
                                        from_token, to_token = edge
                                        self.add_edge(from_token, to_token)
                                # if paragraph is the right size after filters
                                if len(tokens) >= filters['context_size']:
                                    for key in set(tokens):
                                        self.add_node(key)
                                    for from_key, to_key in {(x,y) for x in tokens for y in tokens if x != y}:
                                        self.add_edge(from_key, to_key)
                                    
                except UnicodeDecodeError:
                    print('Failed to decode:', f)
         
        self.nodes = {key:value for key, value in self.nodes.items() if value.freq >= filter_dict['min_occurrences']}
        #removes tokens with too few occurences
        self.nodes = {key:value for key, value in self.nodes.items() if value.freq >= filters['min_occurrences']}
        
        #removes unneccessary edges and pairs with too few cooccurrences
        for node in self.nodes.values():
            node.neighbors = {key:value for key, value in node.neighbors.items() if value >= filter_dict['min_cooccurrence']}
            node.neighbors = {key:value for key, value in node.neighbors.items() if value >= filters['min_cooccurrence'] and key in self.nodes.keys() and node.weight(self.nodes[key])<=filters['max_distance']}
        
        #removes singletons
        self.nodes = {key:value for key, value in self.nodes.items() if len(value.neighbors) > 0}
    
    # finds a path from one node to another
    # Variation on function from https://www.python-course.eu/graphs_python.php
    def find_path(self, start, end, path=None):
        if path == None:
            path = []
@@ -81,59 +101,96 @@ class Graph:
                    return extended_path
        return None
    
    # variation on algorithm from Véronis (2004)
    def root_hubs(self, min_neighbors = 6, theshold = 0.8):
        G = copy.deepcopy(self)
        
        V = sorted(G.nodes.values(), key=lambda value: -1 * value.freq) # -1 to sort descending (...3 -> 2 -> 1...)
        H = []
        
        while V:
            v = V[0]
            if len(v.neighbors) >= min_neighbors:
                mfn = sorted(v.neighbors.keys(), key=lambda key: v.neighbors[key])[:min_neighbors] #mfn: most frequent neighbors
                if np.mean([v.weight(G.nodes[n]) for n in mfn]) < theshold:
                    H.append(v)
                G.nodes = {key:value for key, value in G.nodes.items() if key != v.key and key not in v.neighbors.keys()}
                for node in G.nodes.values():
                    node.neighbors = {key:value for key, value in node.neighbors.items() if key in G.nodes.keys()}
                V = sorted(G.nodes.values(), key=lambda value: -1 * value.freq)
            else:
                return H
        return H
    
    #presents nodes in format key --> (weight, neighbors)
    def view(self):
        for node in self.nodes.values():
            print(node.token,'-->',node.neighbors.keys())
            print(node.key,'-->',[(node.weight(self.nodes[key]), key) for key in node.neighbors.keys()])
            
    #draws graph using networkx
    def draw(self):
        G = nx.Graph()
        for node in self.nodes.values():
            G.add_node(node.token)
            G.add_edges_from([(node.token, y) for y in node.neighbors.keys()])
            G.add_node(node.key)
            G.add_edges_from([(node.key, y) for y in node.neighbors.keys()])
        nx.draw(G, with_labels=True)
        plt.show()
    

#class for single words with frequency and neighbors
class Node:
    
    def __init__(self, token):
        self.token = token
    #initialises node with key and frequency of 1
    def __init__(self, key):
        self.key = key
        self.freq = 1
        self.neighbors = dict()
        

    #adds neighbor to neighbors dict or ups its cooccurence frequency
    def add_neighbor(self, other):
        token = other.token
        
        if token in self.neighbors.keys():
            self.neighbors[token] += 1
        if other.key in self.neighbors.keys():
            self.neighbors[other.key] += 1
        else:
            self.neighbors[token] = 1
            self.neighbors[other.key] = 1
    
    def remove_neighbor(self, token):
        del self.neighbors[token]
    #removes neighbor from dictionary
    def remove_neighbor(self, key):
        del self.neighbors[key]
    
    #calculates weight between self and other node
    #if node is not neighbor, return 1 (no cooccurence), (0 would be complete cooccurrence)
    def weight(self, other):
        return 1 - max([self.neighbors[other.token]/other.count, other.neighbors[self.token]/self.count])
        if other.key in self.neighbors.keys():
            return 1 - max([self.neighbors[other.key]/other.freq, self.neighbors[other.key]/self.freq])
        else:
            return 1


#see Kruskal's algorithm
def minimum_spanning_tree(graph, target):
    pass

#Components algorithm from Véronis (2004), converts graph for target into a MST
def components(graph, target):
    pass

#Uses MST to disambiguate context, should ideally write to evaluator format
def disambiguation(mst, context):
    pass


if __name__ == '__main__':
    """
    Filter:
    min_occurrences: minimum occurences of given word
    min_cooccurrence: minimum cooccurences for word pairs
    stop_words: list of stop words
    allowed_tags: pos-tags of words
    context_size: minimum size for paragraphs after word filtering
    """
    filter_dict = {'min_occurrences' : 10, 'min_cooccurrence' : 5, 'stop_words' : [], 'allowed_tags' : ['NN', 'NNS', 'JJ', 'JJS', 'JJR', 'NNP'], 'context_size' : 4, 'max_distance' : 0.9}
    
    filters = {'min_occurrences' : 10, 'min_cooccurrence' : 5, 'stop_words' : [], 'allowed_tags' : ['NN', 'NNS', 'JJ', 'JJS', 'JJR', 'NNP'], 'context_size' : 4, 'max_distance' : 0.9}
    data_path = '/home/students/zimmermann/Courses/ws17/fsem/absinth/WSI-Evaluator/datasets/MORESQUE'
    corpus_path = '/home/students/zimmermann/Courses/ws17/fsem/absinth/test'
    #corpus_path = '/proj/absinth/wikipedia.txt.dump.20140615-en.SZTAKI'
    
    G = Graph()
    G.build(corpus_path, 'dog', filter_dict)
    G.view()
    print(G.find_path('english', 'kennel'))
    G.draw()
    #corpus_path = '/home/students/zimmermann/Courses/ws17/fsem/absinth/test'
    corpus_path = '/proj/absinth/wikipedia.txt.dump.20140615-en.SZTAKI'
    
    G = Graph() #initialises graph
    G.build(corpus_path, 'gay_bar', filters) #builds graph from corpus with target and filters
    
    for hub in G.root_hubs():
        print(hub.key,'-->', list(hub.neighbors.keys()), '\n') #prints senses
        
    #G.view()
    #print(G.find_path('english', 'kennel'))
    G.draw() #draws graph