OpenTransformer/web-crawl-2026
Web Crawl 2026 A large-scale web crawl dataset for language model pretraining, collected by the OpenTransformer project. Dataset Description This dataset contains text extracted from web pages crawled directly from the internet using custom high-throughput crawlers. All data is freshly scraped. Data Format Each record is a JSON line (gzipped) with fields: text: extracted text content (200-200,000 chars) url: source URL domain: source domain… See the full description on the dataset page: https://huggingface.co/datasets/OpenTransformer/web-crawl-2026.
17.8k
1import os, time, glob, datetime2from huggingface_hub import HfApi3 4TOKEN = "HF_TOKEN_REDACTED"5REPO = "OpenTransformer/web-crawl-2026"6STALE_SEC = 300 # 5 min no writes = done7 8api = HfApi(token=TOKEN)9 10def log(msg):11 ts = datetime.datetime.utcnow().isoformat()12 print(f"{ts} {msg}", flush=True)13 14def find_ready_files():15 ready = []16 now = time.time()17 for d in ["/workspace/staging", "/workspace/scraped_data_go", "/workspace/scraped_data_rust", "/workspace/scraped_data"]:18 for f in glob.glob(os.path.join(d, "*.gz")):19 age = now - os.path.getmtime(f)20 sz = os.path.getsize(f)21 if age > STALE_SEC and sz > 1024*1024: # stale and >1MB22 ready.append((f, sz, age))23 log(f" Ready: {f} ({sz/(1024*1024):.0f}MB, {age/3600:.1f}h old)")24 return ready25 26def upload_file(filepath, size):27 ts = datetime.datetime.utcnow().strftime("%Y%m%d_%H%M%S")28 basename = os.path.basename(filepath)29 remote = f"crawl/combined/{basename.replace(chr(46)+chr(106),chr(95)+ts+chr(46)+chr(106))}"30 log(f"Uploading {basename} ({size/(1024*1024):.0f}MB) -> {remote}")31 try:32 api.upload_file(33 path_or_fileobj=filepath,34 path_in_repo=remote,35 repo_id=REPO,36 repo_type="dataset",37 commit_message=f"Crawl data: {basename} ({size/(1024*1024):.0f}MB)"38 )39 log(f"Uploaded! Removing {filepath}")40 os.remove(filepath)41 return True42 except Exception as e:43 log(f"Upload failed: {e}")44 return False45 46def combine_and_upload(files):47 ts = datetime.datetime.utcnow().strftime("%Y%m%d_%H%M%S")48 combined = f"/workspace/crawl_batch_{ts}.jsonl.gz"49 total = sum(s for _, s, _ in files)50 log(f"Combining {len(files)} files ({total/(1024*1024):.0f}MB)")51 with open(combined, "wb") as out:52 for f, _, _ in files:53 with open(f, "rb") as inp:54 while True:55 chunk = inp.read(8*1024*1024)56 if not chunk:57 break58 out.write(chunk)59 remote = f"crawl/combined/crawl_batch_{ts}.jsonl.gz"60 final = os.path.getsize(combined)61 log(f"Uploading combined {final/(1024*1024):.0f}MB -> {remote}")62 try:63 api.upload_file(64 path_or_fileobj=combined,65 path_in_repo=remote,66 repo_id=REPO,67 repo_type="dataset",68 commit_message=f"Crawl batch {ts} ({final/(1024*1024):.0f}MB, {len(files)} files)"69 )70 log(f"Uploaded! Cleaning up...")71 for f, _, _ in files:72 os.remove(f)73 os.remove(combined)74 return True75 except Exception as e:76 log(f"Upload failed: {e}")77 if os.path.exists(combined):78 os.remove(combined)79 return False80 81def main():82 log("Upload daemon v2 starting")83 while True:84 log("Scanning...")85 ready = find_ready_files()86 if not ready:87 log("No files ready, sleeping 30min")88 time.sleep(1800)89 continue90 91 # If any single file >= 100MB, upload individually92 big = [(f, s, a) for f, s, a in ready if s >= 100*1024*1024]93 small = [(f, s, a) for f, s, a in ready if s < 100*1024*1024]94 95 for f, s, a in big:96 upload_file(f, s)97 98 # Combine small files if total >= 100MB, or if any are >12h old (avoid data loss)99 if small:100 total_small = sum(s for _, s, _ in small)101 max_age = max(a for _, _, a in small)102 if total_small >= 100*1024*1024 or max_age > 43200: # 12 hours103 if len(small) == 1:104 upload_file(small[0][0], small[0][1])105 else:106 combine_and_upload(small)107 else:108 log(f"Small files total {total_small/(1024*1024):.0f}MB, newest {max_age/3600:.1f}h old, waiting")109 110 time.sleep(1800)111 112if __name__ == "__main__":113 main()114 