Advanced Debugging and Logging for OpenClaw at Scale

Clawpedia · For Humans

Enterprise-level debugging and logging strategies for large-scale OpenClaw deployments.

Advanced Debugging and Logging for OpenClaw at Scale

When running OpenClaw in production with many users, basic logging is not enough. This guide covers structured logging, distributed tracing, log aggregation, and debugging complex multi-agent issues.

Logging Architecture at Scale


┌───────────┐  ┌───────────┐  ┌───────────┐
│ Instance 1 │  │ Instance 2 │  │ Instance 3 │
│  stdout    │  │  stdout    │  │  stdout    │
└─────┬──────┘  └─────┬──────┘  └─────┬──────┘
      │               │               │
      └───────────────┬┘───────────────┘
                      ▼
              ┌───────────────┐
              │ Log Aggregator │
              │ (Loki/ELK/    │
              │  Datadog)      │
              └───────┬───────┘
                      ▼
              ┌───────────────┐
              │  Dashboard     │
              │  (Grafana)     │
              └───────────────┘

Structured Logging

Replace unstructured console.log with structured JSON logs:


// lib/logger.js
const pino = require("pino");

const logger = pino({
  level: process.env.LOG_LEVEL || "info",
  formatters: {
    level(label) { return { level: label }; }
  },
  base: {
    instance: process.env.INSTANCE_ID,
    version: process.env.APP_VERSION,
    environment: process.env.NODE_ENV,
  },
  timestamp: pino.stdTimeFunctions.isoTime,
});

// Usage
logger.info({ userId: "abc", messageId: "123", model: "gpt-4" }, 
  "Processing message");

logger.warn({ userId: "abc", tokensUsed: 15000, limit: 16000 }, 
  "Approaching token limit");

logger.error({ userId: "abc", error: err.message, stack: err.stack }, 
  "LLM API call failed");

Request Tracing

Track a request through multiple agents and services:


// lib/tracing.js
const { v4: uuid } = require("uuid");

class RequestTracer {
  static startTrace(userId, messageId) {
    const traceId = uuid();
    return {
      traceId,
      userId,
      messageId,
      spans: [],
      startTime: Date.now()
    };
  }
  
  static startSpan(trace, name) {
    const span = {
      spanId: uuid(),
      name,
      startTime: Date.now(),
      metadata: {}
    };
    trace.spans.push(span);
    return span;
  }
  
  static endSpan(span, metadata = {}) {
    span.endTime = Date.now();
    span.duration = span.endTime - span.startTime;
    span.metadata = { ...span.metadata, ...metadata };
  }
}

// Usage in message handling
async function handleMessage(message) {
  const trace = RequestTracer.startTrace(message.userId, message.id);
  
  // Memory retrieval
  const memSpan = RequestTracer.startSpan(trace, "memory_retrieval");
  const memories = await memory.search(message.text);
  RequestTracer.endSpan(memSpan, { resultsCount: memories.length });
  
  // LLM call
  const llmSpan = RequestTracer.startSpan(trace, "llm_call");
  const response = await llm.complete(prompt);
  RequestTracer.endSpan(llmSpan, { 
    model: "gpt-4", 
    tokens: response.usage.total_tokens 
  });
  
  // Log complete trace
  logger.info({ trace }, "Request completed");
}

Log Levels Strategy

LevelWhen to UseExamples
traceExtremely verbose, development onlyVariable values, loop iterations
debugDetailed operational infoMemory search results, prompt content
infoNormal operationsMessage received, response sent
warnPotential issuesHigh token usage, slow response, retries
errorFailures requiring attentionAPI errors, tool failures

# config.yaml
logging:
  level: "info"           # Production
  # level: "debug"        # Staging
  # level: "trace"        # Development
  
  overrides:
    "llm-adapter": "debug"    # More detail for LLM calls
    "memory": "warn"          # Less noise from memory system
    "platform": "info"        # Standard for platform adapters

Log Aggregation Setup

With Grafana Loki


# docker-compose.yml (additions)
services:
  loki:
    image: grafana/loki:latest
    ports:
      - "3100:3100"
    volumes:
      - loki_data:/loki

  promtail:
    image: grafana/promtail:latest
    volumes:
      - /var/log:/var/log
      - ./promtail-config.yaml:/etc/promtail/config.yaml

  grafana:
    image: grafana/grafana:latest
    ports:
      - "3000:3000"
    environment:
      - GF_AUTH_ANONYMOUS_ENABLED=true

Debugging Patterns

The Replay Pattern

fatalApplication cannot continueDatabase connection lost, OOM

Record and replay conversations for debugging:


// Record mode
async function handleWithRecording(message) {
  const recording = {
    input: message,
    timestamp: Date.now(),
    systemPrompt: getCurrentSystemPrompt(),
    memories: await memory.search(message.text),
    toolCalls: []
  };
  
  const response = await processWithRecording(message, recording);
  
  // Save recording for later replay
  await debugStore.save(`debug_${message.id}`, recording);
  return response;
}

// Replay mode
async function replayMessage(recordingId) {
  const recording = await debugStore.get(recordingId);
  
  // Replay with exact same context
  const result = await llm.complete({
    systemPrompt: recording.systemPrompt,
    memories: recording.memories,
    message: recording.input
  });
  
  // Compare with original
  return {
    original: recording.response,
    replayed: result,
    differences: diff(recording.response, result)
  };
}

The Canary Pattern

Test changes on a subset of traffic:


// Route 10% of users to the new agent configuration
function getAgentConfig(userId) {
  const hash = hashUserId(userId);
  if (hash % 10 === 0) {
    logger.info({ userId, variant: "canary" }, "Using canary config");
    return canaryConfig;
  }
  return productionConfig;
}

Alert Configuration


# alerts.yaml
alerts:
  - name: "High Error Rate"
    condition: "error_rate > 5% over 5 minutes"
    severity: critical
    notify: [slack, pagerduty]
    
  - name: "Slow Response Time"
    condition: "p95_latency > 10s over 10 minutes"
    severity: warning
    notify: [slack]
    
  - name: "Token Budget Exceeded"
    condition: "daily_token_cost > $100"
    severity: warning
    notify: [email]
    
  - name: "Memory Storage Growing"
    condition: "memory_entries_growth > 1000/hour"
    severity: info
    notify: [slack]

Debugging CLI Commands


# Real-time log streaming
openclaw logs --follow --level error

# Search logs by user
openclaw logs --user "user_123" --last 1h

# Inspect a specific conversation
openclaw debug conversation --id "conv_abc"

# Replay a failed request
openclaw debug replay --message-id "msg_xyz"

# Check system health
openclaw health --verbose

# Profile memory usage
openclaw debug memory --stats

Production Debugging Checklist

When investigating production issues:

Best Practices

Related Articles