Spaces:

pszemraj
/

document-summarization

Running on CPU Upgrade

App Files Files Community

pszemraj commited on Apr 30, 2023

Commit

e9ed1f2

1 Parent(s): 9d26661

⚡️ drop nltk for kw

Browse files

Signed-off-by: peter szemraj <peterszemraj@gmail.com>

Files changed (2) hide show

app.py +1 -4
utils.py +32 -25

app.py CHANGED Viewed

@@ -36,10 +36,7 @@ _here = Path(__file__).parent
 # os.environ["NLTK_DATA"] = str(_here / "nltk_data")
 nltk.download("punkt", force=True, quiet=True)
-nltk.download(
-    "popular",
-    force=True,
-)
 MODEL_OPTIONS = [

 # os.environ["NLTK_DATA"] = str(_here / "nltk_data")
 nltk.download("punkt", force=True, quiet=True)
+nltk.download("popular", force=True, quiet=True)
 MODEL_OPTIONS = [

utils.py CHANGED Viewed

@@ -17,11 +17,11 @@ from nltk.corpus import stopwords
 from nltk.tokenize import sent_tokenize, word_tokenize
 from rapidfuzz import fuzz
-nltk.download("punkt", quiet=True)
-nltk.download(
-    "popular",
-    quiet=True,
-)
 def validate_pytorch2(torch_version: str = None):
@@ -101,44 +101,51 @@ def load_example_filenames(example_path: str or Path):
     return examples
-def extract_keywords(text: str, num_keywords: int = 3) -> List[str]:
     """
-    Extracts keywords from a text using the TextRank algorithm.
     Args:
         text: The text to extract keywords from.
         num_keywords: The number of keywords to extract. Default is 5.
     Returns:
         A list of strings, where each string is a keyword extracted from the input text.
     """
-    # Remove stopwords from the input text
-    stop_words = set(stopwords.words("english"))
-    text = " ".join([word for word in text.lower().split() if word not in stop_words])
-    # Tokenize the text into sentences and words
-    sentences = sent_tokenize(text)
-    words = [word_tokenize(sentence) for sentence in sentences]
-    # Filter out words that are shorter than 3 characters
-    words = [[word for word in sentence if len(word) >= 3] for sentence in words]
-    # Create a graph of word co-occurrences
     cooccur = defaultdict(lambda: defaultdict(int))
-    for sentence in words:
-        for w1, w2 in combinations(sentence, 2):
             cooccur[w1][w2] += 1
             cooccur[w2][w1] += 1
-    # Assign scores to words using the TextRank algorithm
     scores = defaultdict(float)
-    for i in range(10):
-        for word in cooccur:
-            score = 0.15 + 0.85 * sum(
                 cooccur[word][other] / sum(cooccur[other].values()) * scores[other]
-                for other in cooccur[word]
             )
-            scores[word] = score
     # Sort the words by score and return the top num_keywords keywords
     keywords = sorted(scores, key=scores.get, reverse=True)[:num_keywords]

 from nltk.tokenize import sent_tokenize, word_tokenize
 from rapidfuzz import fuzz
+import re
+from typing import List
+from itertools import islice
+from collections import defaultdict, deque
+from rapidfuzz import fuzz
 def validate_pytorch2(torch_version: str = None):
     return examples
+def extract_keywords(
+    text: str, num_keywords: int = 3, window_size: int = 5
+) -> List[str]:
     """
+    Extracts keywords from a text using a simplified TextRank algorithm.
     Args:
         text: The text to extract keywords from.
         num_keywords: The number of keywords to extract. Default is 5.
+        window_size: The number of words considered for co-occurrence. Default is 5.
     Returns:
         A list of strings, where each string is a keyword extracted from the input text.
     """
+    # Define stopwords
+    stop_words = set(
+        "a about above after again against all am an and any are aren't as at be because been before being below between both but by can't cannot could couldn't did didn't do does doesn't doing don't down during each few for from further had hadn't has hasn't have haven't having he he'd he'll he's her here here's hers herself him himself his how how's i i'd i'll i'm i've if in into is isn't it it's its itself let's me more most mustn't my myself no nor not of off on once only or other ought our ours ourselves out over own same shan't she she'd she'll she's should shouldn't so some such than that that's the their theirs them themselves then there there's these they they'd they'll they're they've this those through to too under until up very was wasn't we we'd we'll we're we've were weren't what what's when when's where where's which while who who's whom why why's with won't would wouldn't you you'd you'll you're you've your yours yourself yourselves".split()
+    )
+    # Remove stopwords and tokenize the text into words
+    words = [
+        word
+        for word in re.findall(r"\b\w{3,}\b", text.lower())
+        if word not in stop_words
+    ]
+    # Create a graph of word co-occurrences within a moving window of words
     cooccur = defaultdict(lambda: defaultdict(int))
+    deque_words = deque(maxlen=window_size)
+    for word in words:
+        for w1, w2 in combinations(deque_words, 2):
             cooccur[w1][w2] += 1
             cooccur[w2][w1] += 1
+        deque_words.append(word)
+    # Assign scores to words using a simplified TextRank algorithm
     scores = defaultdict(float)
+    for _ in range(10):
+        new_scores = defaultdict(float)
+        for word, co_words in cooccur.items():
+            new_scores[word] = 0.15 + 0.85 * sum(
                 cooccur[word][other] / sum(cooccur[other].values()) * scores[other]
+                for other in co_words
             )
+        scores = new_scores
     # Sort the words by score and return the top num_keywords keywords
     keywords = sorted(scores, key=scores.get, reverse=True)[:num_keywords]