aks1981/S20-tokenization
0
1 2 3 4from urllib.request import urlopen5from bs4 import BeautifulSoup6 7url = "https://raw.githubusercontent.com/cltk/hindi_text_ltrc/master/tulasidaas/Raamacharita_maanasa/1/main.txt"8html = urlopen(url).read()9soup = BeautifulSoup(html, features="html.parser")10 11# kill all script and style elements12for script in soup(["script", "style"]):13 script.extract() # rip it out14 15# get text16text = soup.get_text()17ramayana_text = text18print(type(text))19 20 21 