import React, { useState } from 'react' import { Database, Cpu, Zap, Filter, ArrowRight, Code, Play, CheckCircle, AlertCircle, BookOpen, Terminal, Layers, GitBranch, Settings, ChevronDown, ChevronUp, Copy, ExternalLink, Sparkles, Rocket, Server, Globe, Shield, Star, Menu, X } from 'lucide-react' function App() { const [activePhase, setActivePhase] = useState(0) const [expandedSections, setExpandedSections] = useState({}) const [copiedCode, setCopiedCode] = useState(null) const [mobileMenuOpen, setMobileMenuOpen] = useState(false) const toggleSection = (section) => { setExpandedSections(prev => ({ ...prev, [section]: !prev[section] })) } const copyToClipboard = (code, id) => { navigator.clipboard.writeText(code) setCopiedCode(id) setTimeout(() => setCopiedCode(null), 2000) } const phases = [ { id: 0, name: 'Seed Generator', shortName: 'Phase 0', icon: , space: 'mindchain/hf-space-nemo-seed-generator-huggingface-inference', description: 'Generates initial seed Q&A pairs from a topic using LLM inference.', color: 'amber', gradient: 'from-amber-500 to-orange-600', bgGradient: 'from-amber-500/10 to-orange-500/10', endpoint: '/generate', method: 'POST', features: ['Topic-based generation', 'Configurable seed count', 'Quality pre-filtering'], requestParams: [ { name: 'topic', type: 'string', desc: 'Topic for Q&A generation' }, { name: 'num_seeds', type: 'int', desc: 'Number of seeds to generate (default: 10)' }, { name: 'model', type: 'string', desc: 'Qwen/Qwen2.5-7B-Instruct' }, { name: 'provider', type: 'string', desc: 'together (recommended)' }, ], requestExample: `{ "topic": "Python programming basics", "num_seeds": 10, "model": "Qwen/Qwen2.5-7B-Instruct", "provider": "together" }`, responseExample: `{ "success": true, "seeds": [ { "instruction": "What is a variable in Python?", "output": "A variable in Python is a named container...", "quality_score": 8 } ], "count": 10 }` }, { id: 1, name: 'Distilabel Generator', shortName: 'Phase 1', icon: , space: 'mindchain/hf-space-distilabel-generator-huggingface-inference', description: 'Generates synthetic data with LLM Judge scoring using UltraFeedback criteria.', color: 'blue', gradient: 'from-blue-500 to-cyan-500', bgGradient: 'from-blue-500/10 to-cyan-500/10', endpoint: '/generate', method: 'POST', features: ['UltraFeedback scoring', 'Multi-model support', 'Configurable thresholds'], requestParams: [ { name: 'seed_dataset', type: 'string', desc: 'HF dataset ID from Phase 0' }, { name: 'num_records', type: 'int', desc: 'Records to generate (default: 100)' }, { name: 'model', type: 'string', desc: 'Generator model' }, { name: 'judge_model', type: 'string', desc: 'Judge model for scoring' }, { name: 'use_judge', type: 'bool', desc: 'Enable LLM Judge (default: true)' }, { name: 'min_score', type: 'int', desc: 'Minimum score filter (1-10)' }, ], requestExample: `{ "seed_dataset": "mindchain/synthetic-seeds", "num_records": 100, "model": "Qwen/Qwen2.5-7B-Instruct", "provider": "together", "judge_model": "Qwen/Qwen2.5-7B-Instruct", "judge_provider": "together", "use_judge": true, "min_score": 5 }`, responseExample: `{ "success": true, "records": [ { "instruction": "Explain Python loops...", "output": "Python loops allow you to...", "quality_score": 8, "distilabel_metadata": {...} } ], "stats": { "total": 100, "avg_score": 7.2, "filtered": 15 } }` }, { id: 2, name: 'Argilla Curator', shortName: 'Phase 2', icon: , space: 'mindchain/hf-space-argilla-curator-huggingface-inference', description: 'Final curation with quality scoring and filtering for high-quality dataset.', color: 'purple', gradient: 'from-purple-500 to-pink-500', bgGradient: 'from-purple-500/10 to-pink-500/10', endpoint: '/curate', method: 'POST', features: ['Quality filtering', 'Score distribution', 'Hub push'], requestParams: [ { name: 'raw_dataset', type: 'string', desc: 'Raw dataset ID from Phase 1' }, { name: 'model', type: 'string', desc: 'Judge model for re-scoring' }, { name: 'provider', type: 'string', desc: 'Inference provider' }, { name: 'min_score', type: 'int', desc: 'Minimum score (default: 7)' }, { name: 'target_dataset', type: 'string', desc: 'Output dataset ID (optional)' }, ], requestExample: `{ "raw_dataset": "mindchain/synthetic-distilabel-raw", "model": "Qwen/Qwen2.5-7B-Instruct", "provider": "together", "min_score": 7, "target_dataset": "mindchain/synthetic-curated" }`, responseExample: `{ "success": true, "curated_count": 78, "total_count": 100, "filtered_count": 22, "score_distribution": { "7": 25, "8": 30, "9": 15, "10": 8 } }` } ] const curlExample = (phase) => `curl -X POST https://${phase.space}.hf.space${phase.endpoint} \\ -H "Content-Type: application/json" \\ -d '${phase.requestExample.replace(/\n/g, ' ')}'` const scrollToSection = (id) => { document.getElementById(id)?.scrollIntoView({ behavior: 'smooth' }) setMobileMenuOpen(false) } return (
{/* Navigation */} {/* Hero Section */}
{/* Background Effects */}
HuggingFace Spaces + HF Inference API

Synthetic Data Pipeline

Generate high-quality synthetic Q&A datasets with a 3-phase pipeline. CPU-only, no GPU required.

View Spaces
{/* Features Grid */}
{[ { icon: , label: 'CPU Only', desc: 'No GPU required' }, { icon: , label: 'Always On', desc: 'Docker Spaces' }, { icon: , label: 'Quality Scored', desc: 'LLM Judge' }, { icon: , label: 'HF Hub', desc: 'Dataset push' }, ].map((feature, i) => (
{feature.icon}

{feature.label}

{feature.desc}

))}
{/* Overview Section */}

Overview

This pipeline generates high-quality synthetic Q&A datasets using HuggingFace Spaces and the HF Inference API. It follows a three-phase approach with LLM Judge scoring based on UltraFeedback criteria.

{[ { title: 'Seed Generation', desc: 'Create initial Q&A pairs from any topic', icon: , color: 'amber' }, { title: 'Distilabel Augmentation', desc: 'Generate with LLM Judge scoring', icon: , color: 'blue' }, { title: 'Quality Curation', desc: 'Filter for high-quality records only', icon: , color: 'purple' }, ].map((item, i) => (
{item.icon}

{item.title}

{item.desc}

))}
{/* Pipeline Architecture */}

Pipeline Architecture

{/* Phase Selector */}
{phases.map((phase) => ( ))}
{/* Phase Details */}
{phases[activePhase].icon}

{phases[activePhase].name}

{phases[activePhase].description}

{/* Features */}
{phases[activePhase].features.map((f, i) => ( {f} ))}
{/* Space Link */} {/* Endpoint */}
Endpoint {phases[activePhase].method}
https://{phases[activePhase].space}.hf.space{phases[activePhase].endpoint}
{/* Parameters */}

Parameters

{phases[activePhase].requestParams.map((param, i) => (
{param.name} {param.type} {param.desc}
))}
{/* API Examples Section */}

API Examples

{/* Phase Tabs */}
{phases.map((phase) => ( ))}
{/* Request */}
Request Body
                {phases[activePhase].requestExample}
              
{/* Response */}
Response
                {phases[activePhase].responseExample}
              
{/* cURL */}
cURL Command
              {curlExample(phases[activePhase])}
            
{/* Quick Start */}

Quick Start

{[ { step: 1, title: 'Generate Seeds', desc: 'Create initial Q&A pairs from your topic', color: 'amber', phase: 'Phase 0' }, { step: 2, title: 'Generate Data', desc: 'Augment with LLM Judge quality scoring', color: 'blue', phase: 'Phase 1' }, { step: 3, title: 'Curate Dataset', desc: 'Filter high-quality records (score >= 7)', color: 'purple', phase: 'Phase 2' }, ].map((item) => (
{item.step}
{item.phase}

{item.title}

{item.desc}

))}
{/* Full Script */}

Full Pipeline Script

run_pipeline.py
{fullPipelineScript}
            
{/* LLM Judge Criteria */}

LLM Judge Criteria (UltraFeedback)

{[ { title: 'Instruction-Following', desc: 'Does the answer directly address the question?', color: 'blue' }, { title: 'Truthfulness', desc: 'Is the information accurate and factually correct?', color: 'green' }, { title: 'Honesty', desc: 'Does it avoid hallucination and acknowledge uncertainty?', color: 'yellow' }, { title: 'Helpfulness', desc: 'Is the answer useful for learning?', color: 'purple' }, ].map((item, i) => (

{item.title}

{item.desc}

))}

Score Range

1-10 scale, recommended minimum: 7 for curated datasets

{[1, 2, 3, 4, 5, 6, 7, 8, 9, 10].map((n) => (
= 7 ? 'bg-emerald-500/20 text-emerald-400 border border-emerald-500/30' : 'bg-slate-800 text-slate-500' }`} > {n}
))}
{/* Configuration */}

Configuration

{/* Models */}
{expandedSections['models'] && (

Generator Models

{[ { name: 'Qwen/Qwen2.5-7B-Instruct', default: true }, { name: 'Qwen/Qwen2.5-72B-Instruct', default: false }, { name: 'openai/gpt-4o-mini', default: false }, ].map((m, i) => (
{m.name} {m.default && (default)}
))}

Judge Models

Qwen/Qwen2.5-7B-Instruct (default)
)}
{/* Providers */}
{expandedSections['providers'] && (
{[ { name: 'together', desc: '(default, recommended)' }, { name: 'cerebras', desc: '' }, { name: 'hf', desc: '(serverless)' }, ].map((p, i) => (
{p.name} {p.desc && {p.desc}}
))}
)}
{/* Environment */}
{expandedSections['env'] && (
{`# Required for all spaces
HF_TOKEN=your_huggingface_token_here

# Optional
MAX_RECORDS=1000
DEFAULT_MODEL=Qwen/Qwen2.5-7B-Instruct
DEFAULT_PROVIDER=together`}
                  
)}
{/* Footer */}
) } const fullPipelineScript = `#!/usr/bin/env python3 """Full Synthetic Data Pipeline - Run all three phases""" import requests import json # HF Spaces URLs PHASE0_URL = "https://mindchain-hf-space-nemo-seed-generator-huggingface-inference.hf.space" PHASE1_URL = "https://mindchain-hf-space-distilabel-generator-huggingface-inference.hf.space" PHASE2_URL = "https://mindchain-hf-space-argilla-curator-huggingface-inference.hf.space" def phase0_generate_seeds(topic: str, num_seeds: int = 10): """Phase 0: Generate seed Q&A pairs""" print(f"\\n=== Phase 0: Generating {num_seeds} seeds for '{topic}' ===") response = requests.post( f"{PHASE0_URL}/generate", json={ "topic": topic, "num_seeds": num_seeds, "model": "Qwen/Qwen2.5-7B-Instruct", "provider": "together" } ) result = response.json() print(f"Generated {result.get('count', 0)} seeds") return result def phase1_generate_synthetic(seed_dataset: str, num_records: int = 100): """Phase 1: Generate synthetic data with Judge scoring""" print(f"\\n=== Phase 1: Generating {num_records} records ===") response = requests.post( f"{PHASE1_URL}/generate", json={ "seed_dataset": seed_dataset, "num_records": num_records, "model": "Qwen/Qwen2.5-7B-Instruct", "provider": "together", "judge_model": "Qwen/Qwen2.5-7B-Instruct", "judge_provider": "together", "use_judge": True, "min_score": 5 } ) result = response.json() stats = result.get('stats', {}) print(f"Generated {stats.get('total', 0)} records, avg score: {stats.get('avg_score', 0):.1f}") return result def phase2_curate(raw_dataset: str, min_score: int = 7): """Phase 2: Curate final high-quality dataset""" print(f"\\n=== Phase 2: Curating with min_score={min_score} ===") response = requests.post( f"{PHASE2_URL}/curate", json={ "raw_dataset": raw_dataset, "model": "Qwen/Qwen2.5-7B-Instruct", "provider": "together", "min_score": min_score } ) result = response.json() print(f"Curated {result.get('curated_count', 0)}/{result.get('total_count', 0)} records") return result if __name__ == "__main__": # Run complete pipeline seeds = phase0_generate_seeds("Python programming basics", num_seeds=10) synthetic = phase1_generate_synthetic("mindchain/synthetic-seeds", num_records=100) curated = phase2_curate("mindchain/synthetic-distilabel-raw", min_score=7) print("\\n=== Pipeline Complete! ===") print(f"Final curated dataset: {curated.get('curated_count', 0)} high-quality records")` export default App