-
Notifications
You must be signed in to change notification settings - Fork 44
Expand file tree
/
Copy pathupload_to_hf.py
More file actions
500 lines (414 loc) · 17.8 KB
/
Copy pathupload_to_hf.py
File metadata and controls
500 lines (414 loc) · 17.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
import os
import json
from pathlib import Path
from datetime import datetime
from typing import List, Dict, Any, Optional, Union, Tuple
import time
import backoff
from huggingface_hub.utils import HfHubHTTPError
import requests
# CAMEL imports
from camel.datasets import DataPoint, StaticDataset
from camel.datahubs.huggingface import HuggingFaceDatasetManager
from camel.datahubs.models import Record
# Import Hugging Face Dataset
from datasets import Dataset as HFDataset
def split_train_test(data, train_ratio, random_seed=42):
"""
Split dataset into training and testing sets.
Args:
data (list): List of data entries to split
train_ratio (float): Ratio of training data
random_seed (int): Random seed for reproducibility, defaults to 42
Returns:
tuple: (training_data, testing_data)
"""
import random
random.seed(random_seed)
# Copy data to avoid modifying original
data_copy = data.copy()
# Shuffle data randomly
random.shuffle(data_copy)
# Calculate training set size
train_size = int(len(data_copy) * train_ratio)
# Split data
train_data = data_copy[:train_size]
test_data = data_copy[train_size:]
return train_data, test_data
def load_dataset_files(file_paths):
"""
Load and parse multiple JSON dataset files.
Args:
file_paths (list): List of paths to the JSON dataset files.
Returns:
list: Combined list of data from all files.
"""
combined_data = []
for file_path in file_paths:
print(f"Loading data from {file_path}")
with open(file_path, 'r', encoding='utf-8') as f:
data = json.load(f)
if isinstance(data, list):
combined_data.extend(data)
else:
# If the data is not a list, it might be a dictionary with a key containing the list
for key, value in data.items():
if isinstance(value, list):
combined_data.extend(value)
break
print(f"Loaded {len(combined_data)} total entries")
return combined_data
def transform_data_to_datapoints(data_entries: List[Dict[str, Any]]) -> List[DataPoint]:
"""
Transform loong data into DataPoint objects for local use.
Args:
data_entries (List[Dict[str, Any]]): List of data dictionaries.
Returns:
List[DataPoint]: List of DataPoint objects.
"""
datapoints = []
for entry in data_entries:
# Extract fields
question = entry.get('question', '')
rationale = entry.get('rationale', '')
final_answer = entry.get('final_answer', '')
metadata = entry.get('metadata', {})
# stringify metadata from a dict to a string
meta_data_str = json.dumps(metadata)
# Create DataPoint
datapoint = dict(
question=question,
rationale=rationale,
final_answer=final_answer,
meta_data=meta_data_str,
)
datapoints.append(datapoint)
return datapoints
def convert_datapoints_to_hf_dataset(datapoints: List[DataPoint]) -> HFDataset:
"""Convert a list of DataPoint objects to a Hugging Face Dataset."""
# Convert DataPoints to dictionaries
data_dicts = [dp for dp in datapoints]
# Create a Hugging Face Dataset
hf_dataset = HFDataset.from_list(data_dicts)
return hf_dataset
def transform_data_to_records(data_entries: List[Dict[str, Any]]) -> List[Record]:
"""
Transform loong data into Record objects for Hugging Face upload.
Args:
data_entries (List[Dict[str, Any]]): List of data dictionaries.
Returns:
List[Record]: List of Record objects.
"""
records = []
for id, entry in enumerate(data_entries):
# Extract fields
question = entry.get('question', '')
final_answer = entry.get('final_answer', '')
rationale = entry.get('rationale', '')
meta_data = entry.get('metadata', '')
# read back from metadata string
domain = meta_data.get('domain', '')
record = Record(
source_type=domain,
question=question,
final_answer=final_answer,
rationale=rationale,
meta_data=json.dumps(meta_data),
)
records.append(record)
return records
def generate_or_validate_dataset_name(username, dataset_name=None):
"""
Generates a default dataset name or validates and formats a user-provided name.
Args:
username (str): Hugging Face username.
dataset_name (str, optional): User-provided custom dataset name.
Returns:
str: Formatted dataset name.
"""
if not dataset_name:
dataset_name = "loong"
# Format the dataset name to include the username
return f"{username}/{dataset_name}"
def create_dataset(manager, dataset_name):
"""
Creates a new dataset on Hugging Face and returns the dataset URL.
Args:
manager (HuggingFaceDatasetManager): Instance of HuggingFaceDatasetManager.
dataset_name (str): Name of the dataset.
Returns:
str: URL of the created dataset.
"""
dataset_url = manager.create_dataset(dataset_name)
return dataset_url
def create_dataset_card(manager, dataset_name, username):
"""
Creates a dataset card to add metadata.
Args:
manager (HuggingFaceDatasetManager): Instance of HuggingFaceDatasetManager.
dataset_name (str): Name of the dataset.
username (str): Hugging Face username.
"""
manager.create_dataset_card(
dataset_name=dataset_name,
description="A comprehensive collection of 3,551 high-quality problems across 8 diverse domains, curated for Project Loong. Each problem includes a detailed executable rationale and solution, designed for training and evaluating reasoning models.",
license="mit", # Using lowercase 'mit' as required by HuggingFace
tags=["reasoning", "problem-solving", "project-loong", "multi-domain", "mathematics", "physics", "finance", "optimization"],
authors=[username],
language=["en"],
task_categories=["question-answering"],
content="# Project Loong Seed Dataset\n\n"
"This dataset is part of Project Loong, a collaborative effort to explore whether reasoning-capable models can bootstrap themselves from small, high-quality seed datasets by generating synthetic data and verifying LLM agent responses.\n\n"
"## Dataset Description\n\n"
"This comprehensive collection contains 3,551 human-vetted problems across 11 diverse domains:\n\n"
"- 🧮 **Advanced Math:** 1,611 questions\n"
"- ⚛️ **Advanced Physics:** 429 questions\n"
"- 🧬 **Computational Biology:** 51 questions\n"
"- 💹 **Finance:** 235 questions\n"
"- 🎮 **Game:** 926 questions\n"
"- 📈 **Graph & Discrete Math:** 178 questions\n"
"- 🧠 **Logic:** 130 questions\n"
"- 📐 **Mathematical Programming:** 76 questions\n"
"- 💊 **Medicine:** 916 questions\n"
"- 🔒 **Security & Safety:** 516 questions\n"
"- 🧑💻 **Programming:** 585 questions\n\n"
"## Data Structure\n\n"
"Each entry includes:\n\n"
"- A problem statement\n"
"- A detailed rationale explaining the solution approach\n"
"- The final answer or solution\n"
"- Metadata including problem ID, domain information, and other relevant attributes\n\n"
"## Dataset Purpose\n\n"
"Each dataset is designed to allow automatic evaluation via verifiers, usually by executing the rationale code and comparing the output to the known answer. This collection serves as seed data for exploring whether reasoning-capable models can bootstrap themselves by generating and verifying synthetic data."
)
def add_records_to_dataset(manager, dataset_name, records):
"""
Adds a list of Record objects to the dataset.
Args:
manager (HuggingFaceDatasetManager): Instance of HuggingFaceDatasetManager.
dataset_name (str): Name of the dataset.
records (list): List of Record objects.
"""
manager.add_records(dataset_name, records)
# Add retry decorator for API calls
@backoff.on_exception(
backoff.expo,
(HfHubHTTPError, requests.exceptions.HTTPError),
max_tries=5,
max_time=300,
giveup=lambda e: e.response.status_code not in [429, 500, 502, 503, 504] if hasattr(e, 'response') else False
)
def push_to_hub_with_retry(dataset, dataset_name, config_name, split):
"""
Push dataset to hub with retry mechanism.
"""
dataset.push_to_hub(
dataset_name,
config_name=config_name,
split=split,
private=False,
token=os.environ.get("HF_TOKEN")
)
# Add delay after successful push
time.sleep(5)
def upload_domain_dataset(data_entries: List[Dict[str, Any]],
username: str,
base_dataset_name: str,
domain: str,
split: str):
"""
Upload dataset for a specific domain and split.
Args:
data_entries (List[Dict[str, Any]]): Data entries for the domain.
username (str): Hugging Face username.
base_dataset_name (str): Base name for the dataset.
domain (str): Domain name.
split (str): Split name ('train' or 'test').
"""
try:
# Generate dataset name in format: username/base_dataset_name
dataset_name = f"{username}/{base_dataset_name}"
# Transform data while keeping original structure
formatted_data = []
for entry in data_entries:
# Ensure metadata is in string format
metadata = entry.get('metadata', {})
if isinstance(metadata, str):
try:
metadata = json.loads(metadata)
except json.JSONDecodeError:
metadata = {}
# Add domain to metadata
metadata['domain'] = domain
# Convert metadata to string
metadata_str = json.dumps(metadata)
formatted_data.append({
'question': str(entry.get('question', '')),
'rationale': str(entry.get('rationale', '')),
'final_answer': str(entry.get('final_answer', '')),
'metadata': metadata_str,
'domain': domain
})
# Create HuggingFace Dataset
hf_dataset = HFDataset.from_list(formatted_data)
# Push to hub with retry mechanism
push_to_hub_with_retry(hf_dataset, dataset_name, domain, split)
print(f"Successfully uploaded {domain} {split} split")
except Exception as e:
print(f"Error uploading {domain} {split} split: {e}")
raise
def create_domain_dataset_card(manager, dataset_name, username):
"""
Creates a dataset card for the entire dataset.
Args:
manager (HuggingFaceDatasetManager): Dataset manager instance.
dataset_name (str): Name of the dataset.
username (str): Hugging Face username.
"""
domain_descriptions = {
"advanced_math": "Advanced mathematics problems including calculus, algebra, and number theory",
"advanced_physics": "Physics problems covering mechanics, thermodynamics, and quantum physics",
"computational_biology": "Biological computation and analysis problems",
"finance": "Financial analysis and modeling problems",
"graph_discrete_math": "Graph theory and discrete mathematics problems",
"logic": "Logical reasoning and proof problems",
"mathematical_programming": "Optimization and mathematical programming problems",
"security_and_safety": "Security and safety analysis problems",
"medicine": "Medicine and biology problems",
"programming": "Programming problems"
}
content = """# Project Loong Dataset
This dataset is part of Project Loong, a collaborative effort to explore whether reasoning-capable models can bootstrap themselves from small, high-quality seed datasets.
## Dataset Description
This comprehensive collection contains problems across multiple domains, each split is determined by the domain.
### Available Domains:
"""
# Add domain descriptions
for domain, desc in domain_descriptions.items():
content += f"### {domain.replace('_', ' ').title()}\n{desc}\n\n"
content += """
## Data Structure
Each entry includes:
- A problem statement
- A detailed rationale explaining the solution approach
- The final answer or solution
- Metadata including problem ID and domain information
- Domain label
## Usage
```python
from datasets import load_dataset
# Load a specific domain's data
domain = "advanced_math" # or any other domain
dataset = load_dataset("camel-ai/loong", domain)
# Access specific splits
train_data = dataset["train"]
test_data = dataset["test"]
```
"""
# Create dataset card
manager.create_dataset_card(
dataset_name=dataset_name,
description="A comprehensive collection of high-quality problems across diverse domains, curated for Project Loong. Each problem includes a detailed executable rationale and solution.",
license="mit",
tags=["reasoning", "problem-solving", "project-loong", "multi-domain", "mathematics", "physics", "finance", "optimization"],
authors=[username],
language=["en"],
task_categories=["question-answering"],
content=content
)
def main():
# Get the parent directory of the current script (i.e., loong directory)
current_dir = Path(__file__).parent.parent
data_dir = current_dir / "data"
print(f"Looking for data in: {data_dir}")
# Define domain paths and their corresponding train/test split ratios
# Higher ratio means more training data, lower ratio means more test data
# For example:
# - 0.7: 70% training, 30% testing (standard split)
# - 0.8: 80% training, 20% testing (for domains with less data)
# - 0.6: 60% training, 40% testing (for domains needing more testing)
domain_configs = {
"advanced_math": {
"path": data_dir / "advanced_math" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"advanced_physics": {
"path": data_dir / "advanced_physics" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"computational_biology": {
"path": data_dir / "computational_biology" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"finance": {
"path": data_dir / "finance" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"games": {
"path": data_dir / "games" / "blackjack" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"graph_discrete_math": {
"path": data_dir / "graph_discrete_math" / "seed_dataset.json",
"train_ratio": 0.5 # Standard split
},
"logic": {
"path": data_dir / "logic" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"mathematical_programming": {
"path": data_dir / "mathematical_programming" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"security_and_safety": {
"path": data_dir / "security_and_safety" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"medicine": {
"path": data_dir / "medicine" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
},
"programming": {
"path": data_dir / "programming" / "seed_dataset.json",
"train_ratio": 0.3 # Standard split
}
}
# When using the configurations:
domain_paths = {domain: config["path"] for domain, config in domain_configs.items()}
username = "camel-ai"
base_dataset_name = "loong"
# First create the main dataset (if it doesn't exist)
manager = HuggingFaceDatasetManager()
dataset_name = f"{username}/{base_dataset_name}"
try:
create_dataset(manager, dataset_name)
create_domain_dataset_card(manager, dataset_name, username)
except Exception as e:
print(f"Dataset might already exist: {e}")
# Process data for each domain
for domain, file_path in domain_paths.items():
if not file_path.exists():
print(f"Skipping {domain} due to missing file")
continue
print(f"\nProcessing domain: {domain}")
# Load data for this domain
domain_data = load_dataset_files([str(file_path)])
# Ensure metadata contains the correct domain
for entry in domain_data:
if 'metadata' not in entry:
entry['metadata'] = {}
entry['metadata']['domain'] = domain
# Split into train/test
train_data, test_data = split_train_test(domain_data, domain_configs[domain]["train_ratio"])
print(f"Uploading {len(train_data)} training examples and {len(test_data)} test examples")
# Upload train split
upload_domain_dataset(train_data, username, base_dataset_name, domain, 'train')
# Add delay between train and test uploads
time.sleep(5)
# Upload test split
upload_domain_dataset(test_data, username, base_dataset_name, domain, 'test')
print(f"Completed uploading {domain} dataset")
# Add delay between domains
time.sleep(10)
if __name__ == "__main__":
main()