Team Ai
Datasetpublic

ysn-rfd/text-dataset-tiny-code-script-py-format

USED of tahamajs/medicine_ds_persian for .parquet file USED of Alijafarixcs2/persian-it-llama2-2k for .parquet file USED of Abirate/english_quotes for .jsonl file NEW FILES (05/12/2025) NEW FILES (12/26/2025) NEW FILES (02/15/2026)

sourceHugging Faceapache-2.0updated 4mo agoView on Hugging Face
3likes1.7kdownloads
slm_py_train.py141 linesDownload Raw Back to pytorch
1import torch
2import torch.nn as nn
3import torch.optim as optim
4from torch.utils.data import Dataset, DataLoader
5import nltk
6from nltk.util import ngrams
7from collections import Counter
8import numpy as np
9
10# Download NLTK resources if you haven't already
11nltk.download('punkt')
12
13nltk.download('stopwords')
14
15
16class TextDataset(Dataset):
17    def __init__(self, filepath, n=3, min_freq=1):  # n-gram size, minimum frequency
18        self.n = n
19        self.data = self.load_and_preprocess(filepath, min_freq)
20
21    def load_and_preprocess(self, filepath, min_freq):
22        with open(filepath, 'r', encoding='utf-8') as f:  # Handle encoding
23            text = f.read()
24
25        # Tokenization and lowercasing
26        tokens = nltk.word_tokenize(text.lower())
27
28        # N-gram creation and frequency counting
29        n_grams = ngrams(tokens, self.n)
30        ngram_counts = Counter(n_grams)
31
32        # Filtering based on minimum frequency
33        filtered_ngrams = [ngram for ngram, count in ngram_counts.items() if count >= min_freq]
34
35        # Vocabulary creation
36        self.vocabulary = sorted(set(token for ngram in filtered_ngrams for token in ngram))
37        self.word_to_index = {word: index for index, word in enumerate(self.vocabulary)}
38        self.index_to_word = {index: word for word, index in self.word_to_index.items()}
39
40        # Data preparation for PyTorch
41        data = []
42        for ngram in filtered_ngrams:
43            context = [self.word_to_index[token] for token in ngram[:-1]]
44            target = self.word_to_index[ngram[-1]]
45            data.append((context, target))
46        return data
47
48    def __len__(self):
49        return len(self.data)
50
51    def __getitem__(self, idx):
52        context, target = self.data[idx]
53        return torch.tensor(context), torch.tensor(target)
54
55
56class LanguageModel(nn.Module):
57    def __init__(self, vocab_size, embedding_dim, hidden_dim):
58        super(LanguageModel, self).__init__()
59        self.embedding = nn.Embedding(vocab_size, embedding_dim)
60        self.lstm = nn.LSTM(embedding_dim, hidden_dim, batch_first=True)  # Use LSTM
61        self.linear = nn.Linear(hidden_dim, vocab_size)
62
63    def forward(self, context):
64        embedded = self.embedding(context)
65        output, _ = self.lstm(embedded)  # LSTM output
66        output = self.linear(output[:, -1, :])  # Get the last timestep's output
67        return output
68
69
70# Training parameters
71filepath = 'dataset.txt'  # Replace with your dataset file
72n_gram_size = 3
73min_frequency = 2  # Adjust as needed
74embedding_dimension = 32
75hidden_dimension = 64
76learning_rate = 0.01
77batch_size = 32
78epochs = 10
79
80# Data loading and preprocessing
81dataset = TextDataset(filepath, n_gram_size, min_frequency)
82dataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True)
83
84# Model initialization
85vocab_size = len(dataset.vocabulary)
86model = LanguageModel(vocab_size, embedding_dimension, hidden_dimension)
87
88# Loss function and optimizer
89criterion = nn.CrossEntropyLoss()
90optimizer = optim.Adam(model.parameters(), lr=learning_rate)
91
92# Training loop
93for epoch in range(epochs):
94    for contexts, targets in dataloader:
95        optimizer.zero_grad()
96        outputs = model(contexts)
97        loss = criterion(outputs, targets)
98        loss.backward()
99        optimizer.step()
100
101    print(f"Epoch [{epoch+1}/{epochs}], Loss: {loss.item():.4f}")
102
103# Save the trained model
104torch.save(model.state_dict(), 'language_model.pth')
105
106print("Training complete. Model saved as language_model.pth")
107
108
109
110# Example of text generation (after training and loading)
111def generate_text(model, dataset, start_sequence="the", max_length=50):
112    model.eval()  # Set to evaluation mode
113    tokens = start_sequence.split() # start sequence as list of tokens
114    context = [dataset.word_to_index[token] for token in tokens]
115    context_tensor = torch.tensor([context]) # wrap the context list to a tensor and add one dimension
116
117    generated_text = tokens[:] # start with the start sequence
118
119    for _ in range(max_length):
120        with torch.no_grad():
121            output = model(context_tensor)
122            predicted_index = torch.argmax(output).item()
123            predicted_word = dataset.index_to_word[predicted_index]
124            generated_text.append(predicted_word)
125            context.append(predicted_index) # update context with the new predicted word
126            context = context[-n_gram_size+1:] # keep the context of n-gram size
127            context_tensor = torch.tensor([context]) # update the context tensor
128            
129            if predicted_word == ".": # stop if the predicted word is end of sentence
130                break
131
132    return " ".join(generated_text)
133
134# Example usage (after training and saving)
135# Load the model
136model = LanguageModel(vocab_size, embedding_dimension, hidden_dimension)
137model.load_state_dict(torch.load('language_model.pth'))
138model.eval()
139
140generated_text = generate_text(model, dataset, start_sequence="the quick brown")
141print(generated_text)