add compute profiler: CPU bottleneck detection per SemiWiki article

This commit is contained in:
artale 2026-06-13 13:32:41 +02:00
parent 94064666e1
commit be7354ee38
1 changed files with 82 additions and 0 deletions

View File

@ -5,6 +5,13 @@
* Sakana AI RSI Lab (rsi-lab.sakana.ai)
* Anthropic "When AI builds itself" (anthropic.com/institute/recursive-self-improvement)
* Stephen Wolfram "Games between Programs" (writings.stephenwolfram.com)
* SemiWiki "Agentic AI Demands More Than GPUs" (semiwiki.com)
*
* Key insight from SemiWiki: CPU orchestration, not GPU inference, is the
* bottleneck for agentic AI. Safety checks, tool orchestration, verification,
* and sub-agent coordination all run on CPU. Our system's safety stack
* (content gate, decomp guard, prompt adapter) runs before any model call
* this is efficient: we block unsafe tasks without wasting GPU cycles.
*
* Three RSI bottlenecks this module addresses:
*
@ -285,3 +292,78 @@ export class GoalPersistenceGuard {
return lines.join("\n");
}
}
// ── Compute Profiler (SemiWiki insight) ──────────────
export interface ComputeSample {
timestamp: string;
phase: "safety_check" | "model_inference" | "tool_execution" | "verification" | "orchestration";
durationMs: number;
isCpu: boolean;
}
export interface ComputeProfile {
totalCpuMs: number;
totalGpuMs: number;
cpuRatio: number;
bottleneckPhase: string;
recommendation: string;
}
export class ComputeProfiler {
private store: StateStore;
constructor(store?: StateStore) {
this.store = store ?? new StateStore();
this.store.ensureSubDir("compute");
}
/** Record a compute sample */
record(sample: ComputeSample): void {
this.store.append("compute", "samples.jsonl", sample);
}
/** Profile recent compute usage */
profile(samples: number = 50): ComputeProfile {
const recent = this.store.readLines<ComputeSample>("compute", "samples.jsonl")
.slice(-samples);
if (recent.length === 0) {
return { totalCpuMs: 0, totalGpuMs: 0, cpuRatio: 0, bottleneckPhase: "unknown", recommendation: "No data yet" };
}
const cpuSamples = recent.filter((s) => s.isCpu);
const gpuSamples = recent.filter((s) => !s.isCpu);
const totalCpuMs = cpuSamples.reduce((s, c) => s + c.durationMs, 0);
const totalGpuMs = gpuSamples.reduce((s, c) => s + c.durationMs, 0);
const totalMs = totalCpuMs + totalGpuMs;
// Find the phase that consumes the most time
const phaseTimes = new Map<string, number>();
for (const s of recent) {
phaseTimes.set(s.phase, (phaseTimes.get(s.phase) ?? 0) + s.durationMs);
}
const bottleneckPhase = [...phaseTimes.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? "unknown";
const cpuRatio = totalMs > 0 ? totalCpuMs / totalMs : 0;
const recommendation = cpuRatio > 0.6
? `CPU bottleneck (${(cpuRatio * 100).toFixed(0)}% of time). Consider: parallel safety checks, vectorized tool execution, async verification.`
: `GPU dominates (${((1 - cpuRatio) * 100).toFixed(0)}% of time). Consider: larger batch sizes, model quantization.`;
return { totalCpuMs, totalGpuMs, cpuRatio, bottleneckPhase, recommendation };
}
/** Get compute report */
getReport(): string {
const p = this.profile();
return [
"Compute Profile (SemiWiki bottleneck analysis)",
"",
` CPU time: ${(p.totalCpuMs / 1000).toFixed(1)}s (${(p.cpuRatio * 100).toFixed(0)}%)`,
` GPU time: ${(p.totalGpuMs / 1000).toFixed(1)}s (${((1 - p.cpuRatio) * 100).toFixed(0)}%)`,
` Bottleneck: ${p.bottleneckPhase}`,
"",
` ${p.recommendation}`,
].join("\n");
}
}