File size: 13,303 Bytes
6c223ff
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
#!/usr/bin/env python
"""Convert GitHub repos to JSONL datasets for Fractus training.

Shallow-clones each repo, extracts every text-based file (code, markdown,
docs, config), and writes one JSONL per repo. Uploads each to the HF
dataset repo so build_corpus.py picks it up automatically.

Usage:
    python scripts/repo_to_jsonl.py AFKmoney/kortex AFKmoney/CogNet
    python scripts/repo_to_jsonl.py --batch all        # ~40 high-value repos
    python scripts/repo_to_jsonl.py --batch all --no-upload   # local only
"""
import argparse, os, sys, json, subprocess, tempfile, shutil, time, fnmatch
import urllib.request, io, tarfile
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))

HF_REPO = "thefinalboss/fractus-datasets"

_TOKEN_CACHE = None
def gh_token():
    """GitHub token from `gh auth token` (cached). Enables private-repo access."""
    global _TOKEN_CACHE
    if _TOKEN_CACHE is None:
        try:
            _TOKEN_CACHE = subprocess.check_output(
                ["gh", "auth", "token"], text=True).strip()
        except Exception:
            _TOKEN_CACHE = ""
    return _TOKEN_CACHE

# Text-bearing extensions (code + prose + config). All feed the LM.
TEXT_EXT = {
    ".py", ".rs", ".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".json", ".json5",
    ".md", ".mdx", ".markdown", ".rst", ".txt", ".org", ".tex", ".toml", ".yaml",
    ".yml", ".sh", ".bash", ".zsh", ".ps1", ".go", ".c", ".h", ".cpp", ".hpp",
    ".cc", ".cxx", ".java", ".rb", ".lua", ".sql", ".html", ".htm", ".css", ".scss",
    ".sass", ".less", ".ipynb", ".csv", ".tsv", ".ini", ".cfg", ".conf", ".swift",
    ".kt", ".kts", ".php", ".r", ".jl", ".clj", ".cljs", ".ex", ".exs", ".erl",
    ".hs", ".ml", ".fs", ".nim", ".v", ".sv", ".zig", ".odin", ".dart", ".scala",
    ".groovy", ".gradle", ".makefile", ".mk", ".dockerfile", ".env.example",
    ".gitignore", ".properties", ".tf", ".proto", ".thrift", ".graphql", ".vue",
    ".svelte", ".el", ".lisp", ".scm", ".rkt", ".f90", ".f95", ".asm", ".wat",
}
# Also include extensionless files with these names.
EXTENSIONLESS = {"Makefile", "Dockerfile", "LICENSE", "README", "Rakefile",
                 "CMakeLists", "Cargo", "Gemfile", "Procfile", "Vagrantfile",
                 "Brewfile", "Justfile", "BUILD", "WORKSPACE", "go.mod", "go.sum"}

# Skip these dirs entirely (junk / generated / deps).
SKIP_DIRS = {".git", "node_modules", "__pycache__", "dist", "build", ".venv",
             "venv", "env", ".env", "target", ".next", ".cache", "vendor",
             "deps", "_build", ".idea", ".vscode", ".pytest_cache", ".mypy_cache",
             ".tox", ".eggs", "site-packages", "coverage", ".nuxt", "out",
             ".gradle", ".terraform", "bower_components", "Pods", "DerivedData",
             ".gitlab", ".circleci"}

MAX_FILE_BYTES = 2_000_000  # skip files bigger than 2MB (generated bloat)

# High-value AFKmoney repos with dense technical content (AI architectures,
# OS, papers, neural nets, tooling). Excludes repos already converted to .pt.
HIGH_VALUE_REPOS = [
    "radical-cognitive-architectures",
    "Fractal-Neural-Network",
    "kortex",
    "CogNet",
    "CogNet-MoE-1B",
    "oscillon-architecture",
    "Modele-Variance-Topologique",
    "kahnn",
    "prism",
    "prism-kb",
    "Alpha-N",
    "aether-ai",
    "aether-engine",
    "nexusOS",
    "nxs",
    "kuramoto-controller",
    "synergion",
    "omega-1",
    "omega2",
    "gguf-knowledge-extractor",
    "AICL",
    "NFRC",
    "lea2",
    "nova-spike-hybrid",
    "z-agent-desktop",
    "morphos",
    "datasetfoundry",
    "omega-hedge-fund",
    "ghost-packet-symbiotic",
    "ghost-packet-runtime",
    "mixstudio",
    "Crowd-Adaptive-Alpha-Swarm",
    "pumpfun-agent",
    "nexusOS-dev-",
    "omega-trader",
]


def should_include(path: str, name: str) -> bool:
    if name in EXTENSIONLESS:
        return True
    ext = os.path.splitext(name)[1].lower()
    return ext in TEXT_EXT


# ── Secret filtering ─────────────────────────────────────────────────────
# Files matching these filename patterns are skipped outright (high-density
# secret carriers).
import re as _re
FILENAME_SECRET_RE = _re.compile(
    r"(^\.env$|\.env\.|^secrets?\.[a-z]+$|^credentials?\.|^\.npmrc$|"
    r"\.pem$|\.key$|\.p12$|\.pfx$|\.keystore$|\.jks$|\.wif$|"
    r"^id_rsa|^id_ed25519|^id_ecdsa|^wallet\.|^wallets?/|"
    r"mnemonic|seed_?phrase|private_?key\.|service_account)",
    _re.IGNORECASE)

# High-precision content patterns. If ANY match, the whole file is skipped β€”
# better to drop a legit file than leak a key.
SECRET_CONTENT_RES = [_re.compile(p) for p in [
    r"-----BEGIN (?:RSA |EC |OPENSSH |DSA |ENCRYPTED )?PRIVATE KEY-----",
    # key = "value" / key: value  with a cred-like name and 16+ char value
    r"""(?i)(?:api[_-]?key|api[_-]?secret|secret[_-]?key|access[_-]?token|"""
    r"""auth[_-]?token|bearer|client[_-]?secret|private[_-]?key|wallet|"""
    r"""mnemonic|seed[_-]?phrase|passphrase|password|passwd|consumer[_-]?secret)"""
    r"""\s*[:=]\s*['"]?[A-Za-z0-9+/=_\-]{16,}""",
    r"(?<![A-Za-z0-9])sk-[A-Za-z0-9]{20,}",                # OpenAI
    r"(?<![A-Za-z0-9])gh[pousr]_[A-Za-z0-9]{36,}",         # GitHub PAT
    r"(?<![A-Za-z0-9])hf_[A-Za-z0-9]{30,}",                # HuggingFace
    r"(?<![A-Za-z0-9])AKIA[0-9A-Z]{16}",                   # AWS access key
    r"(?<![A-Za-z0-9])xox[baprs]-[A-Za-z0-9-]{10,}",       # Slack
    r"(?<![A-Za-z0-9])sk_(?:live|test)_[A-Za-z0-9]{20,}",  # Stripe secret
    r"\beyJ[A-Za-z0-9_-]{10,}\.eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{6,}",  # JWT
    r"[a-z][a-z0-9+\-.]*://[^\s/:@]+:[^\s/:@]+@[^\s/]+",   # user:pass@host conn string
    r"(?<![1-9A-HJ-NP-Za-km-z])[1-9A-HJ-NP-Za-km-z]{87,88}(?![1-9A-HJ-NP-Za-km-z])",  # Solana base58 privkey
]]


def looks_like_secret(text: str) -> bool:
    """True if the text contains a high-confidence secret pattern."""
    for rx in SECRET_CONTENT_RES:
        if rx.search(text):
            return True
    return False


def clone(repo: str, dest: str) -> bool:
    """Download a repo tarball and extract into dest. Returns True on success.

    Uses the GitHub API tarball endpoint with the gh token β€” works for BOTH
    public and private repos (codeload is unauthenticated, private-only).
    Tries main then master, with one retry per branch.
    """
    owner = repo.split("/")[0] if "/" in repo else "AFKmoney"
    name = repo.split("/")[-1]
    token = gh_token()
    last_err = ""
    for branch in ("main", "master"):
        url = f"https://api.github.com/repos/{owner}/{name}/tarball/{branch}"
        for attempt in (1, 2):
            try:
                headers = {"User-Agent": "fractus-dataset"}
                if token:
                    headers["Authorization"] = f"token {token}"
                req = urllib.request.Request(url, headers=headers)
                with urllib.request.urlopen(req, timeout=180) as r:
                    data = r.read()
                with tarfile.open(fileobj=io.BytesIO(data), mode="r:gz") as tar:
                    tar.extractall(dest)
                tops = [d for d in os.listdir(dest) if os.path.isdir(os.path.join(dest, d))]
                if tops:
                    stable = os.path.join(dest, "_repo")
                    if os.path.exists(stable):
                        shutil.rmtree(stable)
                    os.rename(os.path.join(dest, tops[0]), stable)
                return True
            except Exception as e:
                last_err = f"{type(e).__name__}: {str(e)[:100]}"
                if attempt == 1:
                    time.sleep(3)
    print(f"  DOWNLOAD FAILED ({last_err})", flush=True)
    return False


def convert(repo: str, out_path: str) -> int:
    """Download repo, extract text to JSONL. Returns byte count written."""
    with tempfile.TemporaryDirectory() as td:
        local = os.path.join(td, "_repo")
        if not clone(repo, td):
            return 0
        n_files, n_bytes, n_skipped_secret = 0, 0, 0
        with open(out_path, "w", encoding="utf-8") as out:
            for root, dirs, files in os.walk(local):
                dirs[:] = [d for d in dirs if d not in SKIP_DIRS
                           and not d.startswith(".")
                           and "egg-info" not in d]
                for fn in files:
                    if not should_include(root, fn):
                        continue
                    # Skip secret-dense filenames outright.
                    if FILENAME_SECRET_RE.search(fn) or FILENAME_SECRET_RE.search(
                            os.path.relpath(root, local)):
                        n_skipped_secret += 1
                        continue
                    fp = os.path.join(root, fn)
                    try:
                        if os.path.getsize(fp) > MAX_FILE_BYTES:
                            continue
                        with open(fp, "r", encoding="utf-8", errors="ignore") as f:
                            text = f.read()
                    except Exception:
                        continue
                    if not text.strip():
                        continue
                    # Scan content for secrets β€” skip the whole file on any match.
                    if looks_like_secret(text):
                        n_skipped_secret += 1
                        continue
                    rel = os.path.relpath(fp, local)
                    out.write(json.dumps({
                        "text": text,
                        "source": f"{repo}/{rel}",
                        "file": fn,
                    }, ensure_ascii=False) + "\n")
                    n_files += 1
                    n_bytes += len(text)
        if n_skipped_secret:
            print(f"  [secret-filter] {n_skipped_secret} files skipped", flush=True)
        print(f"  {n_files} files, {n_bytes:,} chars (~{n_bytes//4:,} tokens)",
              flush=True)
        return n_bytes


def upload(local_path: str, repo_name: str):
    """Upload the JSONL to the HF dataset repo under <repo_name>/."""
    from huggingface_hub import HfApi
    api = HfApi()
    safe = repo_name.split("/")[-1]
    api.upload_file(
        path_or_fileobj=local_path,
        path_in_repo=f"repos/{safe}/data.jsonl",
        repo_id=HF_REPO,
        repo_type="dataset",
    )
    print(f"  β†’ uploaded to {HF_REPO}/repos/{safe}/data.jsonl", flush=True)


def all_repos_from_gh(owner="AFKmoney"):
    """All repo names (public + private) via `gh repo list`."""
    out = subprocess.check_output(
        ["gh", "repo", "list", owner, "--limit", "200", "--json", "name"],
        text=True)
    return [r["name"] for r in json.loads(out)]


def already_uploaded():
    """Set of repo names already present as repos/<name>/data.jsonl on HF."""
    from huggingface_hub import HfApi
    files = HfApi().list_repo_files(HF_REPO, repo_type="dataset")
    done = set()
    for f in files:
        parts = f.split("/")
        if len(parts) == 3 and parts[0] == "repos" and parts[2] == "data.jsonl":
            done.add(parts[1])
    return done


def main():
    ap = argparse.ArgumentParser(description="Convert GitHub repos to JSONL")
    ap.add_argument("repos", nargs="*", help="repo names (e.g. AFKmoney/kortex)")
    ap.add_argument("--batch", choices=["all"], help="use the predefined high-value list")
    ap.add_argument("--all-from-gh", action="store_true",
                    help="fetch ALL repos (public+private) via gh and skip already-uploaded")
    ap.add_argument("--no-upload", action="store_true", help="skip HF upload")
    ap.add_argument("--out-dir", default="data/_repos_jsonl")
    args = ap.parse_args()

    repos = args.repos
    if args.batch == "all":
        repos = HIGH_VALUE_REPOS
    if args.all_from_gh:
        repos = all_repos_from_gh()
    if not repos:
        ap.error("provide repos, --batch all, or --all-from-gh")

    # Skip repos already on HF (only meaningful when uploading).
    skip = already_uploaded() if not args.no_upload else set()
    todo = [r for r in repos if r.split("/")[-1] not in skip]
    if skip:
        print(f"  {len(skip)} already on HF β€” skipping. {len(todo)} to convert.",
              flush=True)

    os.makedirs(args.out_dir, exist_ok=True)
    print(f"=== Converting {len(todo)} repos ===", flush=True)
    if not todo:
        print("Nothing to do β€” all repos already uploaded.", flush=True)
        return
    t0 = time.time()
    grand_bytes = 0
    done = 0
    for i, repo in enumerate(todo, 1):
        name = repo.split("/")[-1]
        print(f"\n[{i}/{len(todo)}] {repo}", flush=True)
        out_path = os.path.join(args.out_dir, f"{name}.jsonl")
        try:
            nb = convert(repo, out_path)
            if nb == 0:
                continue
            grand_bytes += nb
            done += 1
            if not args.no_upload:
                upload(out_path, repo)
            os.remove(out_path)
        except Exception as e:
            print(f"  ERROR: {e}", flush=True)

    print(f"\n=== DONE: {done}/{len(todo)} repos, {grand_bytes:,} chars "
          f"(~{grand_bytes//4:,} tokens) in {time.time()-t0:.0f}s ===", flush=True)


if __name__ == "__main__":
    main()