ysn-rfd/text-dataset-tiny-code-script-py-format
USED of tahamajs/medicine_ds_persian for .parquet file USED of Alijafarixcs2/persian-it-llama2-2k for .parquet file USED of Abirate/english_quotes for .jsonl file NEW FILES (05/12/2025) NEW FILES (12/26/2025) NEW FILES (02/15/2026)
31.7k
1import torch
2import torch.nn as nn
3import torch.optim as optim
4
5# 1. Prepare the Dataset
6
7def prepare_dataset(filepath, seq_length):
8 """
9 Prepares the dataset for training from a text file.
10
11 Args:
12 filepath (str): Path to the text file (e.g., 'dataset.txt').
13 seq_length (int): The length of input sequences.
14
15 Returns:
16 tuple: vocab (set), char_to_index (dict), index_to_char (dict),
17 input_sequences (list), target_sequences (list)
18 """
19 try:
20 with open(filepath, 'r', encoding='utf-8') as file:
21 text = file.read()
22 except FileNotFoundError:
23 print(f"Error: File '{filepath}' not found. Make sure the file exists in the correct directory.")
24 return None, None, None, None, None
25
26 vocab = sorted(list(set(text)))
27 char_to_index = {char: index for index, char in enumerate(vocab)}
28 index_to_char = {index: char for index, char in enumerate(vocab)}
29
30 input_sequences = []
31 target_sequences = []
32
33 for i in range(0, len(text) - seq_length):
34 input_seq = text[i:i + seq_length]
35 target_seq = text[i + seq_length]
36 input_sequences.append([char_to_index[char] for char in input_seq])
37 target_sequences.append(char_to_index[target_seq])
38
39 return vocab, char_to_index, index_to_char, input_sequences, target_sequences
40
41
42# 2. Define the Language Model (Simple RNN)
43
44class SimpleRNNLM(nn.Module):
45 def __init__(self, vocab_size, embedding_dim, hidden_dim, num_layers):
46 super(SimpleRNNLM, self).__init__()
47 self.embedding = nn.Embedding(vocab_size, embedding_dim)
48 self.rnn = nn.RNN(embedding_dim, hidden_dim, num_layers, batch_first=True)
49 self.fc = nn.Linear(hidden_dim, vocab_size)
50
51 def forward(self, input_seq, hidden):
52 embedded = self.embedding(input_seq)
53 output, hidden = self.rnn(embedded, hidden)
54 output = self.fc(output[:, -1, :])
55 return output, hidden
56
57 def init_hidden(self, batch_size, num_layers, hidden_dim):
58 return torch.zeros(num_layers, batch_size, hidden_dim)
59
60
61# Example Usage
62dataset_filepath = 'dataset.txt' # Path to your dataset text file
63seq_length = 64
64
65vocab, char_to_index, index_to_char, input_seqs, target_seqs = prepare_dataset(dataset_filepath, seq_length)
66
67if vocab is None:
68 exit()
69
70print(f"Vocabulary Size: {len(vocab)}")
71print(f"Number of Input Sequences: {len(input_seqs)}")
72
73
74# 3. Instantiate Model, Loss Function, and Optimizer
75
76vocab_size = len(vocab)
77embedding_dim = 32
78hidden_dim = 64
79num_layers = 1
80learning_rate = 0.01
81num_epochs = 10
82
83model = SimpleRNNLM(vocab_size, embedding_dim, hidden_dim, num_layers)
84criterion = nn.CrossEntropyLoss()
85optimizer = optim.Adam(model.parameters(), lr=learning_rate)
86
87device = torch.device("cpu")
88model.to(device)
89criterion.to(device)
90
91
92# 4. Training Loop
93
94batch_size = 256
95
96for epoch in range(num_epochs):
97 model.train()
98 total_loss = 0
99
100 for i in range(0, len(input_seqs), batch_size):
101 input_batch = input_seqs[i:i+batch_size]
102 target_batch = target_seqs[i:i+batch_size]
103
104 input_batch_tensor = torch.LongTensor(input_batch).to(device)
105 target_batch_tensor = torch.LongTensor(target_batch).to(device)
106
107 hidden = model.init_hidden(len(input_batch), num_layers, hidden_dim).to(device)
108
109 optimizer.zero_grad()
110
111 output, hidden = model(input_batch_tensor, hidden)
112 loss = criterion(output, target_batch_tensor)
113
114 loss.backward()
115 optimizer.step()
116
117 total_loss += loss.item()
118
119 average_loss = total_loss / (len(input_seqs) // batch_size + (len(input_seqs) % batch_size != 0))
120 print(f"Epoch [{epoch+1}/{num_epochs}], Loss: {average_loss:.4f}")
121
122
123# 5. Text Generation
124
125def generate_text(model, start_text, predict_len, char_to_index, index_to_char, vocab, device):
126 model.eval()
127 generated_text = start_text
128
129 input_sequence = [char_to_index[char] for char in start_text]
130 input_tensor = torch.LongTensor([input_sequence]).to(device)
131
132 hidden = model.init_hidden(1, num_layers, hidden_dim).to(device)
133
134 with torch.no_grad():
135 for _ in range(predict_len):
136 output, hidden = model(input_tensor, hidden)
137
138 probabilities = torch.softmax(output, dim=1)
139 predicted_index = torch.multinomial(probabilities, 1).item()
140 predicted_char = index_to_char[predicted_index]
141
142 generated_text += predicted_char
143
144 input_sequence = input_sequence[1:] + [predicted_index]
145 input_tensor = torch.LongTensor([input_sequence]).to(device)
146
147 return generated_text
148
149
150# Example Generation
151start_text = "The "
152predict_length = 500
153
154generated_output = generate_text(model, start_text, predict_length, char_to_index, index_to_char, vocab, device)
155print("\nGenerated Text:")
156print(generated_output)