ysn-rfd/text-dataset-tiny-code-script-py-format
USED of tahamajs/medicine_ds_persian for .parquet file USED of Alijafarixcs2/persian-it-llama2-2k for .parquet file USED of Abirate/english_quotes for .jsonl file NEW FILES (05/12/2025) NEW FILES (12/26/2025) NEW FILES (02/15/2026)
31.7k
1import torch
2import torch.nn as nn
3import torch.optim as optim
4from torch.utils.data import Dataset, DataLoader
5import nltk
6from nltk.util import ngrams
7from collections import Counter
8import numpy as np
9
10# Download NLTK resources if you haven't already
11nltk.download('punkt')
12
13nltk.download('stopwords')
14
15
16class TextDataset(Dataset):
17 def __init__(self, filepath, n=3, min_freq=1): # n-gram size, minimum frequency
18 self.n = n
19 self.data = self.load_and_preprocess(filepath, min_freq)
20
21 def load_and_preprocess(self, filepath, min_freq):
22 with open(filepath, 'r', encoding='utf-8') as f: # Handle encoding
23 text = f.read()
24
25 # Tokenization and lowercasing
26 tokens = nltk.word_tokenize(text.lower())
27
28 # N-gram creation and frequency counting
29 n_grams = ngrams(tokens, self.n)
30 ngram_counts = Counter(n_grams)
31
32 # Filtering based on minimum frequency
33 filtered_ngrams = [ngram for ngram, count in ngram_counts.items() if count >= min_freq]
34
35 # Vocabulary creation
36 self.vocabulary = sorted(set(token for ngram in filtered_ngrams for token in ngram))
37 self.word_to_index = {word: index for index, word in enumerate(self.vocabulary)}
38 self.index_to_word = {index: word for word, index in self.word_to_index.items()}
39
40 # Data preparation for PyTorch
41 data = []
42 for ngram in filtered_ngrams:
43 context = [self.word_to_index[token] for token in ngram[:-1]]
44 target = self.word_to_index[ngram[-1]]
45 data.append((context, target))
46 return data
47
48 def __len__(self):
49 return len(self.data)
50
51 def __getitem__(self, idx):
52 context, target = self.data[idx]
53 return torch.tensor(context), torch.tensor(target)
54
55
56class LanguageModel(nn.Module):
57 def __init__(self, vocab_size, embedding_dim, hidden_dim):
58 super(LanguageModel, self).__init__()
59 self.embedding = nn.Embedding(vocab_size, embedding_dim)
60 self.lstm = nn.LSTM(embedding_dim, hidden_dim, batch_first=True) # Use LSTM
61 self.linear = nn.Linear(hidden_dim, vocab_size)
62
63 def forward(self, context):
64 embedded = self.embedding(context)
65 output, _ = self.lstm(embedded) # LSTM output
66 output = self.linear(output[:, -1, :]) # Get the last timestep's output
67 return output
68
69
70# Training parameters
71filepath = 'dataset.txt' # Replace with your dataset file
72n_gram_size = 3
73min_frequency = 2 # Adjust as needed
74embedding_dimension = 32
75hidden_dimension = 64
76learning_rate = 0.01
77batch_size = 32
78epochs = 10
79
80# Data loading and preprocessing
81dataset = TextDataset(filepath, n_gram_size, min_frequency)
82dataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True)
83
84# Model initialization
85vocab_size = len(dataset.vocabulary)
86model = LanguageModel(vocab_size, embedding_dimension, hidden_dimension)
87
88# Loss function and optimizer
89criterion = nn.CrossEntropyLoss()
90optimizer = optim.Adam(model.parameters(), lr=learning_rate)
91
92# Training loop
93for epoch in range(epochs):
94 for contexts, targets in dataloader:
95 optimizer.zero_grad()
96 outputs = model(contexts)
97 loss = criterion(outputs, targets)
98 loss.backward()
99 optimizer.step()
100
101 print(f"Epoch [{epoch+1}/{epochs}], Loss: {loss.item():.4f}")
102
103# Save the trained model
104torch.save(model.state_dict(), 'language_model.pth')
105
106print("Training complete. Model saved as language_model.pth")
107
108
109
110# Example of text generation (after training and loading)
111def generate_text(model, dataset, start_sequence="the", max_length=50):
112 model.eval() # Set to evaluation mode
113 tokens = start_sequence.split() # start sequence as list of tokens
114 context = [dataset.word_to_index[token] for token in tokens]
115 context_tensor = torch.tensor([context]) # wrap the context list to a tensor and add one dimension
116
117 generated_text = tokens[:] # start with the start sequence
118
119 for _ in range(max_length):
120 with torch.no_grad():
121 output = model(context_tensor)
122 predicted_index = torch.argmax(output).item()
123 predicted_word = dataset.index_to_word[predicted_index]
124 generated_text.append(predicted_word)
125 context.append(predicted_index) # update context with the new predicted word
126 context = context[-n_gram_size+1:] # keep the context of n-gram size
127 context_tensor = torch.tensor([context]) # update the context tensor
128
129 if predicted_word == ".": # stop if the predicted word is end of sentence
130 break
131
132 return " ".join(generated_text)
133
134# Example usage (after training and saving)
135# Load the model
136model = LanguageModel(vocab_size, embedding_dimension, hidden_dimension)
137model.load_state_dict(torch.load('language_model.pth'))
138model.eval()
139
140generated_text = generate_text(model, dataset, start_sequence="the quick brown")
141print(generated_text)