thefinalboss commited on
Commit
815ba07
·
verified ·
1 Parent(s): c711fa2

Upload scripts/build_quality_corpus.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. scripts/build_quality_corpus.py +91 -0
scripts/build_quality_corpus.py ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python
2
+ """Build a quality corpus that includes Fractus's own source code.
3
+
4
+ Combines:
5
+ - The existing communication_corpus (18.6M tokens of dialogue/text)
6
+ - ALL Python source files from fractus/ and fractus1B/
7
+ - All markdown docs
8
+ - The white paper (extracted text)
9
+
10
+ The model learns its own architecture — this is the palimpseste principle:
11
+ Fractus contains its own description.
12
+ """
13
+ import os, sys, glob, time
14
+ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
15
+ import torch
16
+ from fractus.tokenizer import FractusTokenizer
17
+
18
+ HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
19
+
20
+
21
+ def collect_source_files():
22
+ """Collect all .py and .md files from the fractus packages + docs."""
23
+ patterns = [
24
+ os.path.join(HERE, "fractus", "**", "*.py"),
25
+ os.path.join(HERE, "fractus1B", "**", "*.py"),
26
+ os.path.join(HERE, "docs", "**", "*.md"),
27
+ os.path.join(HERE, "experiments", "**", "*.py"),
28
+ os.path.join(HERE, "scripts", "*.py"),
29
+ os.path.join(HERE, "*.py"),
30
+ os.path.join(HERE, "README.md"),
31
+ ]
32
+ files = set()
33
+ for pattern in patterns:
34
+ files.update(glob.glob(pattern, recursive=True))
35
+ # Filter out __pycache__.
36
+ files = sorted(f for f in files if "__pycache__" not in f and ".pyc" not in f)
37
+ return files
38
+
39
+
40
+ def main():
41
+ tok = FractusTokenizer.gpt2_compatible()
42
+ print("=== Building Quality Corpus (with Fractus source code) ===", flush=True)
43
+
44
+ # 1. Collect source files.
45
+ source_files = collect_source_files()
46
+ print(f"Source files: {len(source_files)}", flush=True)
47
+
48
+ # Tokenize all source files.
49
+ all_source_tokens = []
50
+ total_chars = 0
51
+ for fpath in source_files:
52
+ try:
53
+ with open(fpath, "r", encoding="utf-8", errors="ignore") as f:
54
+ text = f.read()
55
+ total_chars += len(text)
56
+ # Add file separator token.
57
+ text = "\n\n# === FILE: " + os.path.relpath(fpath, HERE) + " ===\n\n" + text
58
+ ids = tok.encode(text)
59
+ all_source_tokens.extend(ids)
60
+ except Exception as e:
61
+ print(f" skip {fpath}: {e}", flush=True)
62
+
63
+ source_tensor = torch.tensor(all_source_tokens, dtype=torch.int32)
64
+ print(f"Source code: {total_chars:,} chars → {len(source_tensor):,} tokens", flush=True)
65
+
66
+ # 2. Load existing corpus.
67
+ existing = torch.load(os.path.join(HERE, "data", "communication_corpus.pt"),
68
+ weights_only=False)
69
+ print(f"Existing corpus: {len(existing):,} tokens", flush=True)
70
+
71
+ # 3. Combine: existing + source code repeated 3x (so the model really learns its code).
72
+ combined = torch.cat([
73
+ existing,
74
+ source_tensor, source_tensor, source_tensor, # 3x for emphasis
75
+ ])
76
+ print(f"Combined corpus: {len(combined):,} tokens", flush=True)
77
+
78
+ # 4. Shuffle (fixed seed).
79
+ g = torch.Generator().manual_seed(42)
80
+ perm = torch.randperm(len(combined), generator=g)
81
+ combined = combined[perm]
82
+ print(f"Shuffled (seed=42)", flush=True)
83
+
84
+ # 5. Save.
85
+ out_path = os.path.join(HERE, "data", "quality_corpus.pt")
86
+ torch.save(combined, out_path)
87
+ print(f"Saved: {out_path} ({os.path.getsize(out_path)/1e6:.0f}MB)", flush=True)
88
+
89
+
90
+ if __name__ == "__main__":
91
+ main()