AGofficial commited on
Commit
727b410
·
verified ·
1 Parent(s): 59cdc4d

Delete prepare_data.py

Browse files
Files changed (1) hide show
  1. prepare_data.py +0 -152
prepare_data.py DELETED
@@ -1,152 +0,0 @@
1
- import csv
2
- import json
3
- import os
4
- import sys
5
-
6
- # Increase field size limit for large CSV fields
7
- csv.field_size_limit(sys.maxsize)
8
-
9
- def clean_text(text):
10
- if not text:
11
- return ""
12
- return text.strip()
13
-
14
- def process_item(system, conversation, output_file):
15
- # conversion to text format:
16
- # <system>...
17
- # <user>...
18
- # <ai>...
19
-
20
- # Check if we have valid content
21
- if not conversation:
22
- return
23
-
24
- text_parts = []
25
-
26
- # Add system if present
27
- if system:
28
- text_parts.append(f"<system>{clean_text(system)}")
29
-
30
- # Process conversation turns
31
- # conversation is a list of (role, content)
32
- # roles: 'user', 'ai' (we map 'assistant'->'ai')
33
-
34
- for role, content in conversation:
35
- content = clean_text(content)
36
- if not content:
37
- continue
38
-
39
- if role == 'system':
40
- # Handle system in message list if somehow present/overriding
41
- text_parts.append(f"<system>{content}")
42
- elif role == 'user':
43
- text_parts.append(f"<user>{content}")
44
- elif role == 'assistant' or role == 'ai':
45
- text_parts.append(f"<ai>{content}")
46
- else:
47
- # Fallback for unknown roles
48
- text_parts.append(f"<{role}>{content}")
49
-
50
- if text_parts:
51
- final_str = "\n".join(text_parts)
52
- output_file.write(final_str + "\n\n")
53
-
54
- def process_csv(filepath, output_path):
55
- print(f"Processing CSV: {filepath}")
56
- try:
57
- with open(filepath, 'r', encoding='utf-8', errors='replace') as f:
58
- reader = csv.DictReader(f)
59
- with open(output_path, 'a', encoding='utf-8') as out:
60
- count = 0
61
- for row in reader:
62
- # Mapping logic for this specific CSV structure:
63
- # thread_title -> User context/prompt
64
- # instruction -> System prompt
65
- # message -> AI response
66
-
67
- sys_prompt = row.get('instruction', '')
68
- title = row.get('thread_title', '')
69
- msg = row.get('message', '')
70
-
71
- # If we have mainly 'text' and it looks like it contains everything, we might prefer it?
72
- # But analysis suggested 'text' was just instruction/duplicate in some rows.
73
- # We'll stick to constructing from parts which is safer for structured training.
74
-
75
- conversation = []
76
- if title:
77
- conversation.append(('user', title))
78
- if msg:
79
- conversation.append(('ai', msg))
80
-
81
- if conversation:
82
- process_item(sys_prompt, conversation, out)
83
- count += 1
84
- if count % 10000 == 0:
85
- print(f"CSV Processed {count}...", flush=True)
86
- print(f"Finished CSV. Processed {count} rows.")
87
- except Exception as e:
88
- print(f"Error processing CSV: {e}")
89
-
90
- def process_jsonl(filepath, output_path):
91
- print(f"Processing JSONL: {filepath}")
92
- try:
93
- with open(filepath, 'r', encoding='utf-8') as f, open(output_path, 'a', encoding='utf-8') as out:
94
- count = 0
95
- for line in f:
96
- if not line.strip(): continue
97
- try:
98
- data = json.loads(line)
99
- messages = data.get('messages', [])
100
-
101
- # Extract system prompt if it exists as a separate field or role
102
- system_prompt = data.get('system', '')
103
-
104
- conversation = []
105
-
106
- for m in messages:
107
- role = m.get('role', '')
108
- content = m.get('content', '')
109
-
110
- if role == 'system':
111
- # If we hit a system role, treat it as global system or part of flow
112
- # User asked to "add system as <system>"
113
- # I'll just map it directly.
114
- conversation.append(('system', content))
115
- else:
116
- conversation.append((role, content))
117
-
118
- if conversation:
119
- # Pass None for separate system arg since we handle it in loop
120
- process_item(None, conversation, out)
121
- count += 1
122
-
123
- if count % 10000 == 0:
124
- print(f"JSONL Processed {count}...", flush=True)
125
-
126
- except json.JSONDecodeError:
127
- continue
128
- print(f"Finished JSONL. Processed {count} items.")
129
- except Exception as e:
130
- print(f"Error processing JSONL: {e}")
131
-
132
- def main():
133
- data_dir = "data"
134
- output_filename = "processed_corpus.txt"
135
- output_path = os.path.join(data_dir, output_filename)
136
-
137
- # Overwrite/Create new
138
- with open(output_path, 'w', encoding='utf-8') as f:
139
- f.write("")
140
-
141
- # Process JSONL
142
- jsonl_path = os.path.join(data_dir, "dataset.jsonl")
143
- if os.path.exists(jsonl_path):
144
- process_jsonl(jsonl_path, output_path)
145
-
146
- # Process CSV
147
- csv_path = os.path.join(data_dir, "train.csv")
148
- if os.path.exists(csv_path):
149
- process_csv(csv_path, output_path)
150
-
151
- if __name__ == "__main__":
152
- main()