• 17-08-2019, 03:22:45
    #1
    S.a. arkadaşlar,

    Şu meşhur lsi, işte bu kod bu işi yapıyor. Kodda alıntılar vardır. Fakat nerden aldığımı hatırlamıyorum. Gensim modülünü kurmanız gerekmektedir. Anaconda ile birlikte geliyor diye biliyorum. Anaconda kurulu ise sıkıntı çıkarmayacaktır. Gensim modülü kurmakta çok sıkıntılar çekmiştim ayrıca. Linux kullanıyorsanız sıkıntı yaşayacağınızı sanmıyorum. articles.txt içerisine makalenizi veya google dan çıkan sonuçları özenli şekilde yapıştırın. Türkçe dilde pek iyi çalışmıyor. Fakat çıktıyı yorumlarsanız mutlaka güzel kelimeler yakalarsınız. words=30 kısmını çok uzun tutmamanızı öneririm. Bilgisayarınız sut down olabilir. Ayrıca stopwords kelimeleri temizler, datanızı hazır hale getirir. Birde kelimeleri köklerine indirir. Neyse ya işine yarayan varsa buyursun kullansın. Helali hoş olsun. Lütfen bu konu ile ilgili yardım istemeyiniz.

    #!/usr/bin/env python
    # -*- coding: utf-8 -*-
    import os.path
    from gensim import corpora
    from gensim.models import LsiModel
    from gensim.models.ldamodel import LdaModel
    from nltk.tokenize import RegexpTokenizer
    from nltk.corpus import stopwords
    from nltk.stem.porter import PorterStemmer
    from gensim.models.coherencemodel import CoherenceModel
    import matplotlib.pyplot as plt
    
    
    
    
    def load_data(path,file_name):
    """
    Input : path and file_name
    Purpose: loading text file
    Output : list of paragraphs/documents and
    title(initial 100 words considred as title of document)
    """
    documents_list = []
    titles=[]
    with open( os.path.join(path, file_name) ,"r", encoding='ISO-8859-1') as fin:
    for line in fin.readlines():
    text = line.strip()
    documents_list.append(text)
    print("Total Number of Documents:",len(documents_list))
    titles.append( text[0:min(len(text),100)] )
    return documents_list,titles
    
    def preprocess_data(doc_set):
    """
    Input : docuemnt list
    Purpose: preprocess text (tokenize, removing stopwords, and stemming)
    Output : preprocessed text
    """
    # initialize regex tokenizer
    tokenizer = RegexpTokenizer(r'w+')
    # create English stop words list
    
    en_stop = set(stopwords.words('turkish'))
    # Create p_stemmer of class PorterStemmer
    p_stemmer = PorterStemmer()
    # list for tokenized documents in loop
    texts = []
    # loop through document list
    for i in doc_set:
    # clean and tokenize document string
    raw = i.lower()
    tokens = tokenizer.tokenize(raw)
    # remove stop words from tokens
    stopped_tokens = [i for i in tokens if not i in en_stop]
    # stem tokens
    stemmed_tokens = [p_stemmer.stem(i) for i in stopped_tokens]
    # add tokens to list
    texts.append(stemmed_tokens)
    return texts
    def prepare_corpus(doc_clean):
    """
    Input : clean document
    Purpose: create term dictionary of our courpus and Converting list of documents (corpus) into Document Term Matrix
    Output : term dictionary and Document Term Matrix
    """
    # Creating the term dictionary of our courpus, where every unique term is assigned an index. dictionary = corpora.Dictionary(doc_clean)
    dictionary = corpora.Dictionary(doc_clean)
    # Converting list of documents (corpus) into Document Term Matrix using dictionary prepared above.
    doc_term_matrix = [dictionary.doc2bow(doc) for doc in doc_clean]
    # generate LDA model
    return dictionary,doc_term_matrix
    def create_gensim_lsa_model(doc_clean,number_of_topics,words):
    """
    Input : clean document, number of topics and number of words associated with each topic
    Purpose: create LSA model using gensim
    Output : return LSA model
    """
    dictionary,doc_term_matrix=prepare_corpus(doc_clean)
    # generate LSA model
    lsamodel = LsiModel(doc_term_matrix, num_topics=number_of_topics, id2word = dictionary) # train model
    print(lsamodel.print_topics(num_topics=number_of_topics, num_words=words))
    return lsamodel
    def compute_coherence_values(dictionary, doc_term_matrix, doc_clean, stop, start=2, step=3):
    """
    Input : dictionary : Gensim dictionary
    corpus : Gensim corpus
    texts : List of input texts
    stop : Max num of topics
    purpose : Compute c_v coherence for various number of topics
    Output : model_list : List of LSA topic models
    coherence_values : Coherence values corresponding to the LDA model with respective number of topics
    """
    coherence_values = []
    model_list = []
    for num_topics in range(start, stop, step):
    # generate LSA model
    model = LsiModel(doc_term_matrix, num_topics=number_of_topics, id2word = dictionary) # train model
    model_list.append(model)
    coherencemodel = CoherenceModel(model=model, texts=doc_clean, dictionary=dictionary, coherence='c_v')
    coherence_values.append(coherencemodel.get_coherence())
    return model_list, coherence_values
    def plot_graph(doc_clean,start, stop, step):
    dictionary,doc_term_matrix=prepare_corpus(doc_clean)
    model_list, coherence_values = compute_coherence_values(dictionary, doc_term_matrix,doc_clean,
    stop, start, step)
    # Show graph
    x = range(start, stop, step)
    plt.plot(x, coherence_values)
    plt.xlabel("Number of Topics")
    plt.ylabel("Coherence score")
    plt.legend(("coherence_values"), loc='best')
    plt.show()
    
    #start,stop,step=2,12,1
    #plot_graph(clean_text,start,stop,step)
    
    # LSA Model
    number_of_topics=7
    words=30
    document_list,titles=load_data("","articles.txt")
    clean_text=preprocess_data(document_list)
    model=create_gensim_lsa_model(clean_text,number_of_topics,words)
    #start,stop,step=2,12,1
    #plot_graph(clean_text,start,stop,step)
  • 19-08-2019, 10:28:44
    #2
    Paylaşım için teşekkürler. Akşam deneyeceğim.