diff --git a/docs/gns3-copilot/template-based-configuration-roadmap.md b/docs/gns3-copilot/template-based-configuration-roadmap.md index 7d9564e00..92fff7625 100644 --- a/docs/gns3-copilot/template-based-configuration-roadmap.md +++ b/docs/gns3-copilot/template-based-configuration-roadmap.md @@ -119,6 +119,19 @@ User Request: "Configure OSPF on all routers" | **Template Method** | **~400 tokens** | Template: 100 + Parameters: 300 | | **Savings** | **75%** | 1200 tokens saved | +#### 🔥 Scenario: Large-Scale Topology - 500+ Routers + +**This is where the template-based approach truly shines for rapid environment provisioning.** + +| Approach | Token Usage | Execution Time | Breakdown | +|----------|-------------|----------------|-----------| +| **Current Method (AI)** | **~75,000 tokens** | ~25 minutes | 150 tokens/device × 500 devices, serial execution | +| **Template + AI** | **~5,000 tokens** | ~10 minutes | Template once + AI generates params, but slow | +| **Template + Rules (Direct)** | **~400 tokens** | **~3 minutes** | Template once + rule engine (0 tokens) + parallel execution | +| **Savings** | **99.5%** | **88%** | **Game-changing for large deployments** | + +**Key Insight:** For environments with **hundreds or thousands of nodes**, the direct execution mode (skipping AI) becomes critical for rapid topology preparation. + --- ## Core Components @@ -520,6 +533,449 @@ workflow.add_edge("execute", END) --- +## 🔥 Large-Scale Topology Support (1000+ Nodes) + +### Overview + +One of the most powerful use cases for the template-based configuration system is **rapid provisioning of large-scale network topologies**. This section details optimizations for environments with **hundreds to thousands of nodes**. + +### Challenge: Traditional AI Approach at Scale + +``` +Problem: Configure 1000 routers with OSPF + +Traditional AI Approach: +- AI generates config for each router: 150 tokens × 1000 = 150,000 tokens +- Serial or limited parallel execution: ~30-50 minutes +- High cost, slow execution, poor scalability +``` + +### Solution: Direct Execution Mode + +The key innovation is allowing users to **modify and directly execute** templates without requiring AI re-analysis: + +``` +Template-Based Direct Execution: +1. AI generates template once: ~150 tokens +2. User reviews and modifies if needed +3. User clicks "⚡ Confirm & Execute" +4. Rule engine generates params for 1000 devices: 0 tokens +5. Parallel execution (50-100 concurrent): ~5 minutes +6. Total: 150 tokens, 5 minutes +``` + +### Enhanced HITL Workflow for Scale + +``` +┌─────────────────────────────────────────────────────────────┐ +│ Step 1: AI Generates Template (Once) │ +│ │ +│ User: "Configure OSPF on all 1000 routers" │ +│ │ +│ AI generates template: ~150 tokens │ +│ router ospf {{ process_id }} │ +│ {% for network in networks %} │ +│ network {{ network.ip }} {{ network.mask }} area {{ area }} │ +│ {% endfor %} │ +└─────────────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────────────┐ +│ 🔵 HITL Checkpoint 1: Template Review │ +│ │ +│ User can: │ +│ - Review template syntax │ +│ - Modify template directly │ +│ - See preview with sample data │ +│ │ +│ Actions: [✓ Confirm & Continue] [⚡ Confirm & Execute*] │ +│ [✏️ Modify] [❌ Cancel] │ +│ │ +│ * "Confirm & Execute" = Skip AI, go to rule engine │ +└─────────────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────────────┐ +│ Step 2A: Rule Engine (0 tokens) OR Step 2B: AI (5000 tokens)│ +│ │ +│ If user chose "⚡ Confirm & Execute": │ +│ → Rule engine analyzes template │ +│ → Extracts device names from topology │ +│ → Auto-assigns IPs and parameters │ +│ → Generates 1000 device param sets: 0 tokens │ +│ │ +│ If user chose "✓ Confirm & Continue": │ +│ → AI analyzes template │ +│ → Generates parameters: ~5000 tokens │ +└─────────────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────────────┐ +│ 🔵 HITL Checkpoint 2: Parameter Review │ +│ │ +│ For 1000 devices, show SUMMARY: │ +│ - Total devices: 1000 │ +│ - Configuration patterns: 3 unique patterns │ +│ - Sample configs (first 3 devices) │ +│ - IP addressing scheme used │ +│ │ +│ Actions: [⚡ Execute All*] [✓ Review & Modify] [❌ Cancel] │ +│ │ +│ * "Execute All" = Start parallel execution │ +└─────────────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────────────┐ +│ Step 3: Parallel Batch Execution │ +│ │ +│ Configuration execution: │ +│ - Batch size: 50 devices (configurable) │ +│ - Batches: 20 total (1000 / 50) │ +│ - Parallel execution within each batch │ +│ - Real-time progress updates via SSE │ +│ - Estimated time: 3-5 minutes │ +│ │ +│ Progress updates: │ +│ Batch 1/20: Configuring devices 1-50... │ +│ Batch 2/20: Configuring devices 51-100... │ +│ ... │ +│ Complete: 998 success, 2 failed │ +└─────────────────────────────────────────────────────────────┘ +``` + +### Rule Engine: Intelligent Parameter Generation + +```python +# gns3server/agent/gns3_copilot/config_templates/param_generator.py + +def generate_params_for_large_topology( + template: str, + topology_info: dict, + addressing_scheme: str = "sequential" +) -> dict: + """ + Generate parameters for 1000+ devices using rule-based logic. + + Key features: + - Extract device numbering from names (R1, R2, ... R1000) + - Auto-assign IP addresses sequentially + - Group devices by type and apply patterns + - Zero AI token consumption + """ + + nodes = topology_info.get("nodes", []) + + # Group by device type + devices_by_type = group_by_device_type(nodes) + # Result: {"router": [R1, R2, ..., R500], "switch": [SW1, ..., SW500]} + + device_params = [] + + for device_type, type_nodes in devices_by_type.items(): + for idx, node in enumerate(type_nodes, start=1): + device_name = node.get("name") + + # Extract device number from name + device_num = extract_device_number(device_name, idx) + # R1 → 1, Router-100 → 100, DeviceX → fallback to idx + + # Generate parameters using rules + params = { + "device_name": device_name, + "process_id": 1, + "area": "0", + "router_id": f"1.1.1.{device_num}", + "networks": [ + { + "ip": f"192.168.{device_num}.0", + "mask": "0.0.0.255" + } + ], + "loopback": { + "ip": f"10.{device_num}.1.1", + "mask": "255.255.255.255" + } + } + + device_params.append(params) + + return { + "device_params": device_params, + "total_devices": len(device_params), + "generation_method": "rule_engine", + "addressing_scheme": addressing_scheme + } + + +# Example: 1000 devices configured in < 1 second +# Token cost: 0 (pure rule-based logic) +``` + +### Batch Parallel Execution + +```python +# gns3server/agent/gns3_copilot/tools_v2/config_tools_nornir.py + +class ExecuteTemplateBasedConfig(BaseTool): + """Optimized for large-scale parallel execution.""" + + def _run(self, tool_input: str | dict) -> dict: + """Execute configuration with dynamic batching.""" + + device_params = data.get("device_params", []) + total_devices = len(device_params) + + # 🔥 Dynamic batch sizing based on device count + if total_devices <= 10: + batch_size = 10 + elif total_devices <= 50: + batch_size = 20 + elif total_devices <= 100: + batch_size = 30 + elif total_devices <= 500: + batch_size = 50 + else: # 500+ devices + batch_size = 100 # High concurrency for large topologies + + results = { + "total_devices": total_devices, + "batch_size": batch_size, + "total_batches": (total_devices + batch_size - 1) // batch_size, + "batches": [] + } + + # Process in batches with progress tracking + for batch_num in range(0, total_devices, batch_size): + batch_end = min(batch_num + batch_size, total_devices) + batch_params = device_params[batch_num:batch_end] + + # Render configs for this batch + batch_configs = [ + { + "device_name": p["device_name"], + "config_commands": renderer.render(template, p) + } + for p in batch_params + ] + + # Execute batch in parallel using Nornir + batch_result = self._execute_batch_parallel( + project_id, + batch_configs, + batch_num // batch_size + 1 + ) + + results["batches"].append(batch_result) + + # Yield progress for SSE streaming + yield_progress({ + "type": "batch_complete", + "batch": batch_num // batch_size + 1, + "progress": int((batch_end / total_devices) * 100) + }) + + return results +``` + +### Real-Time Progress Streaming + +```typescript +// Frontend: Large-scale configuration progress UI + +class LargeScaleConfigProgress { + displayProgress() { + // Show progress bar for 1000 devices + return ` +