import React, { useState } from 'react'
import {
Database,
Cpu,
Zap,
Filter,
ArrowRight,
Code,
Play,
CheckCircle,
AlertCircle,
BookOpen,
Terminal,
Layers,
GitBranch,
Settings,
ChevronDown,
ChevronUp,
Copy,
ExternalLink,
Sparkles,
Rocket,
Server,
Globe,
Shield,
Star,
Menu,
X
} from 'lucide-react'
function App() {
const [activePhase, setActivePhase] = useState(0)
const [expandedSections, setExpandedSections] = useState({})
const [copiedCode, setCopiedCode] = useState(null)
const [mobileMenuOpen, setMobileMenuOpen] = useState(false)
const toggleSection = (section) => {
setExpandedSections(prev => ({
...prev,
[section]: !prev[section]
}))
}
const copyToClipboard = (code, id) => {
navigator.clipboard.writeText(code)
setCopiedCode(id)
setTimeout(() => setCopiedCode(null), 2000)
}
const phases = [
{
id: 0,
name: 'Seed Generator',
shortName: 'Phase 0',
icon: ,
space: 'mindchain/hf-space-nemo-seed-generator-huggingface-inference',
description: 'Generates initial seed Q&A pairs from a topic using LLM inference.',
color: 'amber',
gradient: 'from-amber-500 to-orange-600',
bgGradient: 'from-amber-500/10 to-orange-500/10',
endpoint: '/generate',
method: 'POST',
features: ['Topic-based generation', 'Configurable seed count', 'Quality pre-filtering'],
requestParams: [
{ name: 'topic', type: 'string', desc: 'Topic for Q&A generation' },
{ name: 'num_seeds', type: 'int', desc: 'Number of seeds to generate (default: 10)' },
{ name: 'model', type: 'string', desc: 'Qwen/Qwen2.5-7B-Instruct' },
{ name: 'provider', type: 'string', desc: 'together (recommended)' },
],
requestExample: `{
"topic": "Python programming basics",
"num_seeds": 10,
"model": "Qwen/Qwen2.5-7B-Instruct",
"provider": "together"
}`,
responseExample: `{
"success": true,
"seeds": [
{
"instruction": "What is a variable in Python?",
"output": "A variable in Python is a named container...",
"quality_score": 8
}
],
"count": 10
}`
},
{
id: 1,
name: 'Distilabel Generator',
shortName: 'Phase 1',
icon: ,
space: 'mindchain/hf-space-distilabel-generator-huggingface-inference',
description: 'Generates synthetic data with LLM Judge scoring using UltraFeedback criteria.',
color: 'blue',
gradient: 'from-blue-500 to-cyan-500',
bgGradient: 'from-blue-500/10 to-cyan-500/10',
endpoint: '/generate',
method: 'POST',
features: ['UltraFeedback scoring', 'Multi-model support', 'Configurable thresholds'],
requestParams: [
{ name: 'seed_dataset', type: 'string', desc: 'HF dataset ID from Phase 0' },
{ name: 'num_records', type: 'int', desc: 'Records to generate (default: 100)' },
{ name: 'model', type: 'string', desc: 'Generator model' },
{ name: 'judge_model', type: 'string', desc: 'Judge model for scoring' },
{ name: 'use_judge', type: 'bool', desc: 'Enable LLM Judge (default: true)' },
{ name: 'min_score', type: 'int', desc: 'Minimum score filter (1-10)' },
],
requestExample: `{
"seed_dataset": "mindchain/synthetic-seeds",
"num_records": 100,
"model": "Qwen/Qwen2.5-7B-Instruct",
"provider": "together",
"judge_model": "Qwen/Qwen2.5-7B-Instruct",
"judge_provider": "together",
"use_judge": true,
"min_score": 5
}`,
responseExample: `{
"success": true,
"records": [
{
"instruction": "Explain Python loops...",
"output": "Python loops allow you to...",
"quality_score": 8,
"distilabel_metadata": {...}
}
],
"stats": {
"total": 100,
"avg_score": 7.2,
"filtered": 15
}
}`
},
{
id: 2,
name: 'Argilla Curator',
shortName: 'Phase 2',
icon: ,
space: 'mindchain/hf-space-argilla-curator-huggingface-inference',
description: 'Final curation with quality scoring and filtering for high-quality dataset.',
color: 'purple',
gradient: 'from-purple-500 to-pink-500',
bgGradient: 'from-purple-500/10 to-pink-500/10',
endpoint: '/curate',
method: 'POST',
features: ['Quality filtering', 'Score distribution', 'Hub push'],
requestParams: [
{ name: 'raw_dataset', type: 'string', desc: 'Raw dataset ID from Phase 1' },
{ name: 'model', type: 'string', desc: 'Judge model for re-scoring' },
{ name: 'provider', type: 'string', desc: 'Inference provider' },
{ name: 'min_score', type: 'int', desc: 'Minimum score (default: 7)' },
{ name: 'target_dataset', type: 'string', desc: 'Output dataset ID (optional)' },
],
requestExample: `{
"raw_dataset": "mindchain/synthetic-distilabel-raw",
"model": "Qwen/Qwen2.5-7B-Instruct",
"provider": "together",
"min_score": 7,
"target_dataset": "mindchain/synthetic-curated"
}`,
responseExample: `{
"success": true,
"curated_count": 78,
"total_count": 100,
"filtered_count": 22,
"score_distribution": {
"7": 25, "8": 30, "9": 15, "10": 8
}
}`
}
]
const curlExample = (phase) => `curl -X POST https://${phase.space}.hf.space${phase.endpoint} \\
-H "Content-Type: application/json" \\
-d '${phase.requestExample.replace(/\n/g, ' ')}'`
const scrollToSection = (id) => {
document.getElementById(id)?.scrollIntoView({ behavior: 'smooth' })
setMobileMenuOpen(false)
}
return (
{/* Navigation */}
Synthetic Pipeline
HF Spaces Documentation
{/* Desktop Nav */}
scrollToSection('overview')} className="text-sm text-slate-400 hover:text-white transition-colors">Overview
scrollToSection('pipeline')} className="text-sm text-slate-400 hover:text-white transition-colors">Pipeline
scrollToSection('api')} className="text-sm text-slate-400 hover:text-white transition-colors">API
scrollToSection('quickstart')} className="text-sm text-slate-400 hover:text-white transition-colors">Quick Start
HF Collection
{/* Mobile menu button */}
setMobileMenuOpen(!mobileMenuOpen)}
className="md:hidden p-2 text-slate-400 hover:text-white"
>
{mobileMenuOpen ? : }
{/* Mobile Nav */}
{mobileMenuOpen && (
scrollToSection('overview')} className="block w-full text-left px-3 py-2 text-slate-400 hover:text-white hover:bg-slate-800 rounded-lg">Overview
scrollToSection('pipeline')} className="block w-full text-left px-3 py-2 text-slate-400 hover:text-white hover:bg-slate-800 rounded-lg">Pipeline
scrollToSection('api')} className="block w-full text-left px-3 py-2 text-slate-400 hover:text-white hover:bg-slate-800 rounded-lg">API
scrollToSection('quickstart')} className="block w-full text-left px-3 py-2 text-slate-400 hover:text-white hover:bg-slate-800 rounded-lg">Quick Start
)}
{/* Hero Section */}
{/* Background Effects */}
HuggingFace Spaces + HF Inference API
Synthetic Data Pipeline
Generate high-quality synthetic Q&A datasets with a 3-phase pipeline.
CPU-only, no GPU required.
scrollToSection('quickstart')}
className="btn btn-primary flex items-center gap-2"
>
Get Started
View Spaces
{/* Features Grid */}
{[
{ icon:
, label: 'CPU Only', desc: 'No GPU required' },
{ icon:
, label: 'Always On', desc: 'Docker Spaces' },
{ icon:
, label: 'Quality Scored', desc: 'LLM Judge' },
{ icon:
, label: 'HF Hub', desc: 'Dataset push' },
].map((feature, i) => (
{feature.icon}
{feature.label}
{feature.desc}
))}
{/* Overview Section */}
This pipeline generates high-quality synthetic Q&A datasets using HuggingFace Spaces
and the HF Inference API. It follows a three-phase approach with LLM Judge scoring
based on UltraFeedback criteria.
{[
{ title: 'Seed Generation', desc: 'Create initial Q&A pairs from any topic', icon:
, color: 'amber' },
{ title: 'Distilabel Augmentation', desc: 'Generate with LLM Judge scoring', icon:
, color: 'blue' },
{ title: 'Quality Curation', desc: 'Filter for high-quality records only', icon:
, color: 'purple' },
].map((item, i) => (
{item.icon}
{item.title}
{item.desc}
))}
{/* Pipeline Architecture */}
{/* Phase Selector */}
{phases.map((phase) => (
setActivePhase(phase.id)}
className={`w-full text-left p-4 rounded-xl border-2 transition-all ${
activePhase === phase.id
? `border-${phase.color}-500 bg-gradient-to-r ${phase.bgGradient}`
: 'border-slate-700 hover:border-slate-600 bg-slate-800/30'
}`}
>
{phase.icon}
{phase.shortName}
{phase.name}
))}
{/* Phase Details */}
{phases[activePhase].icon}
{phases[activePhase].name}
{phases[activePhase].description}
{/* Features */}
{phases[activePhase].features.map((f, i) => (
{f}
))}
{/* Space Link */}
{/* Endpoint */}
Endpoint
{phases[activePhase].method}
https://{phases[activePhase].space}.hf.space{phases[activePhase].endpoint}
{/* Parameters */}
Parameters
{phases[activePhase].requestParams.map((param, i) => (
{param.name}
{param.type}
{param.desc}
))}
{/* API Examples Section */}
{/* Phase Tabs */}
{phases.map((phase) => (
setActivePhase(phase.id)}
className={`flex items-center gap-2 px-4 py-2 rounded-lg font-medium text-sm whitespace-nowrap transition-colors ${
activePhase === phase.id
? `bg-gradient-to-r ${phase.gradient} text-white`
: 'bg-slate-800 text-slate-400 hover:text-white'
}`}
>
{phase.icon}
{phase.shortName}
))}
{/* Request */}
Request Body
copyToClipboard(phases[activePhase].requestExample, 'req')}
className="text-slate-400 hover:text-white transition-colors"
>
{copiedCode === 'req' ? : }
{phases[activePhase].requestExample}
{/* Response */}
Response
copyToClipboard(phases[activePhase].responseExample, 'res')}
className="text-slate-400 hover:text-white transition-colors"
>
{copiedCode === 'res' ? : }
{phases[activePhase].responseExample}
{/* cURL */}
cURL Command
copyToClipboard(curlExample(phases[activePhase]), 'curl')}
className="text-slate-400 hover:text-white transition-colors"
>
{copiedCode === 'curl' ? : }
{curlExample(phases[activePhase])}
{/* Quick Start */}
{[
{ step: 1, title: 'Generate Seeds', desc: 'Create initial Q&A pairs from your topic', color: 'amber', phase: 'Phase 0' },
{ step: 2, title: 'Generate Data', desc: 'Augment with LLM Judge quality scoring', color: 'blue', phase: 'Phase 1' },
{ step: 3, title: 'Curate Dataset', desc: 'Filter high-quality records (score >= 7)', color: 'purple', phase: 'Phase 2' },
].map((item) => (
{item.step}
{item.phase}
{item.title}
{item.desc}
))}
{/* Full Script */}
run_pipeline.py
copyToClipboard(fullPipelineScript, 'script')}
className="text-slate-400 hover:text-white transition-colors"
>
{copiedCode === 'script' ? : }
{fullPipelineScript}
{/* LLM Judge Criteria */}
LLM Judge Criteria (UltraFeedback)
{[
{ title: 'Instruction-Following', desc: 'Does the answer directly address the question?', color: 'blue' },
{ title: 'Truthfulness', desc: 'Is the information accurate and factually correct?', color: 'green' },
{ title: 'Honesty', desc: 'Does it avoid hallucination and acknowledge uncertainty?', color: 'yellow' },
{ title: 'Helpfulness', desc: 'Is the answer useful for learning?', color: 'purple' },
].map((item, i) => (
))}
Score Range
1-10 scale, recommended minimum: 7 for curated datasets
{[1, 2, 3, 4, 5, 6, 7, 8, 9, 10].map((n) => (
= 7 ? 'bg-emerald-500/20 text-emerald-400 border border-emerald-500/30' : 'bg-slate-800 text-slate-500'
}`}
>
{n}
))}
{/* Configuration */}
{/* Models */}
toggleSection('models')}
className="w-full flex items-center justify-between p-5 hover:bg-slate-800/30 transition-colors"
>
Supported Models
{expandedSections['models'] ? : }
{expandedSections['models'] && (
Generator Models
{[
{ name: 'Qwen/Qwen2.5-7B-Instruct', default: true },
{ name: 'Qwen/Qwen2.5-72B-Instruct', default: false },
{ name: 'openai/gpt-4o-mini', default: false },
].map((m, i) => (
{m.name}
{m.default && (default) }
))}
Judge Models
Qwen/Qwen2.5-7B-Instruct
(default)
)}
{/* Providers */}
toggleSection('providers')}
className="w-full flex items-center justify-between p-5 hover:bg-slate-800/30 transition-colors"
>
Providers
{expandedSections['providers'] ? : }
{expandedSections['providers'] && (
{[
{ name: 'together', desc: '(default, recommended)' },
{ name: 'cerebras', desc: '' },
{ name: 'hf', desc: '(serverless)' },
].map((p, i) => (
{p.name}
{p.desc && {p.desc} }
))}
)}
{/* Environment */}
toggleSection('env')}
className="w-full flex items-center justify-between p-5 hover:bg-slate-800/30 transition-colors"
>
Environment Variables
{expandedSections['env'] ? : }
{expandedSections['env'] && (
{`# Required for all spaces
HF_TOKEN=your_huggingface_token_here
# Optional
MAX_RECORDS=1000
DEFAULT_MODEL=Qwen/Qwen2.5-7B-Instruct
DEFAULT_PROVIDER=together`}
)}
{/* Footer */}
)
}
const fullPipelineScript = `#!/usr/bin/env python3
"""Full Synthetic Data Pipeline - Run all three phases"""
import requests
import json
# HF Spaces URLs
PHASE0_URL = "https://mindchain-hf-space-nemo-seed-generator-huggingface-inference.hf.space"
PHASE1_URL = "https://mindchain-hf-space-distilabel-generator-huggingface-inference.hf.space"
PHASE2_URL = "https://mindchain-hf-space-argilla-curator-huggingface-inference.hf.space"
def phase0_generate_seeds(topic: str, num_seeds: int = 10):
"""Phase 0: Generate seed Q&A pairs"""
print(f"\\n=== Phase 0: Generating {num_seeds} seeds for '{topic}' ===")
response = requests.post(
f"{PHASE0_URL}/generate",
json={
"topic": topic,
"num_seeds": num_seeds,
"model": "Qwen/Qwen2.5-7B-Instruct",
"provider": "together"
}
)
result = response.json()
print(f"Generated {result.get('count', 0)} seeds")
return result
def phase1_generate_synthetic(seed_dataset: str, num_records: int = 100):
"""Phase 1: Generate synthetic data with Judge scoring"""
print(f"\\n=== Phase 1: Generating {num_records} records ===")
response = requests.post(
f"{PHASE1_URL}/generate",
json={
"seed_dataset": seed_dataset,
"num_records": num_records,
"model": "Qwen/Qwen2.5-7B-Instruct",
"provider": "together",
"judge_model": "Qwen/Qwen2.5-7B-Instruct",
"judge_provider": "together",
"use_judge": True,
"min_score": 5
}
)
result = response.json()
stats = result.get('stats', {})
print(f"Generated {stats.get('total', 0)} records, avg score: {stats.get('avg_score', 0):.1f}")
return result
def phase2_curate(raw_dataset: str, min_score: int = 7):
"""Phase 2: Curate final high-quality dataset"""
print(f"\\n=== Phase 2: Curating with min_score={min_score} ===")
response = requests.post(
f"{PHASE2_URL}/curate",
json={
"raw_dataset": raw_dataset,
"model": "Qwen/Qwen2.5-7B-Instruct",
"provider": "together",
"min_score": min_score
}
)
result = response.json()
print(f"Curated {result.get('curated_count', 0)}/{result.get('total_count', 0)} records")
return result
if __name__ == "__main__":
# Run complete pipeline
seeds = phase0_generate_seeds("Python programming basics", num_seeds=10)
synthetic = phase1_generate_synthetic("mindchain/synthetic-seeds", num_records=100)
curated = phase2_curate("mindchain/synthetic-distilabel-raw", min_score=7)
print("\\n=== Pipeline Complete! ===")
print(f"Final curated dataset: {curated.get('curated_count', 0)} high-quality records")`
export default App