JustPaste
HomeCategoriesAboutDonateContactTerms of UsePrivacy Policy
JustPaste

Free online notepad — write and share instantly

Navigate

  • Home
  • Timeline
  • Categories

Info

  • About
  • Donate
  • Contact

Legal

  • Terms of Use
  • Privacy Policy

© 2026 JustPaste.app. All rights reserved.

Made with ♥ by JustPaste

Untitled Page | JustPaste.app
2 months ago1 views
👨‍💻Programming
import os import nltk
from nltk.corpus import stopwords
from nltk.tokenize import word_tokenize from nltk.stem import PorterStemmer from collections import defaultdict import json
nltk.download('punkt') nltk.download('stopwords') def preprocess(text):
text = text.lower()
tokens = word_tokenize(text)
stop_words = set(stopwords.words('english')) stemmer = PorterStemmer()
words = [stemmer.stem(word) for word in tokens if word.isalnum() and word not in stop_words]
return words documents = {}

for filename in os.listdir():
if filename.endswith(".txt"):
with open(filename, 'r', encoding='utf-8', errors='ignore') as f: text = f.read()
documents[filename] = preprocess(text) print(f"Total documents loaded: {len(documents)}") inverted_index = defaultdict(set)
for doc_id, words in documents.items():
for word in set(words): # avoid duplicates per document inverted_index[word].add(doc_id)
vocab_size = len(inverted_index) print(f"\nVocabulary Size: {vocab_size} words") print("\nSample inverted index terms:")
for term in list(inverted_index)[:10]: print(f"{term}: {sorted(inverted_index[term])}")
← Back to timeline