Import AITURK IDE 1.0.0-beta.1 from Hermes 63279301; preserve MIT license

This commit is contained in:
2026-09-05 13:26:46 +03:00
commit 03634b1ca3
11340 changed files with 3442369 additions and 0 deletions
@@ -0,0 +1,426 @@
// Multi-tab working sessions: N session tiles stacked as tabs in the main
// zone, EVERY tab mounted (keep-alive), all streaming concurrently — the
// "5 tabs doing PR review" workload. Measures frame pacing + longtasks while
// the whole stack streams, which is where multitab renderers crawl.
//
// --zones M splits the tiles across M VISIBLE split zones (a 2×2 grid for 4)
// instead of one tab stack — the "4 tiles with 4 sessions each" workload,
// where M transcripts stream on screen at once and the rest are mounted
// keep-alive tabs behind them. --streaming S caps how many sessions are
// actually mid-turn (zone leaders first, so S=zones means "every visible
// transcript streams, every hidden tab idles"); the rest sit settled.
// --sessions N seeds a populated recents list (a lived-in sessions DB).
// --turns N sets transcript depth per tile (long sessions), and --tools makes
// every transcript an AGENT session: seeded turns carry settled tool rounds,
// and the live stream opens/completes tool calls between text chunks.
//
// Drives the real pipeline synthetically (no backend, no credits): each tick
// routes one delta per streaming session through `hook.update` — the same
// wiring-cache write (journal + publish + view sync) the gateway's delta
// flush performs — via the __HERMES_SESSION_TILES__ hook.
//
// node scripts/perf/run.mjs multitab --spawn [--tiles 5] [--tokens 240]
// node scripts/perf/run.mjs multitab --spawn --tiles 16 --zones 4 --sessions 300
import { sleep } from '../lib/cdp.mjs'
import { frameHistogram, percentile } from '../lib/stats.mjs'
// Same recorder pattern as stream.mjs (generation-guarded rAF + longtasks).
const RECORDERS = `
(() => {
window.__FT_GEN__ = (window.__FT_GEN__ || 0) + 1
const ftGen = window.__FT_GEN__
window.__FT__ = { times: [], stop: false }
let last = performance.now()
const tick = () => {
if (window.__FT_GEN__ !== ftGen || window.__FT__.stop) return
const now = performance.now()
window.__FT__.times.push(now - last)
last = now
requestAnimationFrame(tick)
}
requestAnimationFrame(tick)
window.__LT__ = { entries: [], stop: false }
try {
const po = new PerformanceObserver((list) => {
if (window.__LT__.stop) return
for (const e of list.getEntries()) window.__LT__.entries.push({ duration: e.duration, startTime: e.startTime })
})
po.observe({ entryTypes: ['longtask'] })
window.__LT__.po = po
} catch {}
return 'armed'
})()
`
const COLLECT = `
(() => {
window.__FT__.stop = true
window.__LT__.stop = true
try { window.__LT__.po && window.__LT__.po.disconnect() } catch {}
return JSON.stringify({ frames: window.__FT__.times, longtasks: window.__LT__.entries })
})()
`
/** Page-side setup: open `tiles` session tiles — one tab stack in the main
* zone (zones=1), or spread across `zones` visible splits (a 2×2 grid for 4)
* — bind fake runtime ids, and seed each with a realistic transcript.
*
* States are written through `hook.update` — the REAL gateway write path
* (wiring cache + in-flight journal + publish + view sync). Driving
* `hook.publish` alone under-models a stream: it skips the journal and the
* cache, which is exactly where multi-session cost used to hide. */
const setup = (tiles, seedTurns, streamSeed, zones, seedSessions, streaming, dead, tools) => `
(() => {
const hook = window.__HERMES_SESSION_TILES__
if (!hook) return 'no-hook'
if (!hook.update) return 'no-update-hook'
// A settled tool call the way the gateway stores one: streamed args (kept
// as argsText too) and a result blob. Real agent transcripts are MOSTLY
// these — a long session is hundreds of terminal/read_file/patch rounds.
const toolPart = (sid, i, k) => {
const args = { command: 'rg -n "handler" src/module-' + i + ' | head -40', background: false }
return {
type: 'tool-call', toolCallId: sid + '-t' + i + '-' + k, toolName: k % 2 ? 'read_file' : 'terminal',
args, argsText: JSON.stringify(args),
result: JSON.stringify({ success: true, output: Array.from({ length: 18 },
(_, l) => 'src/module-' + i + '.ts:' + (l * 7 + 3) + ': const handler = wrap(ctx, retry)').join('\\n') })
}
}
const turn = (sid, i) => {
const answer = { id: sid + '-a' + i, role: 'assistant', timestamp: Date.now(), pending: false,
parts: [{ type: 'text', text: [
'## Finding ' + i, '',
'The handler swallows the rejection. Key points for hunk \\\`' + i + '\\\`:', '',
'- The catch block drops the original error.',
'- Retries are unbounded — see [the loop](https://example.com/loop).', '',
'\\\`\\\`\\\`ts',
'async function retry' + i + '(fn: () => Promise<void>) {',
' for (;;) { try { return await fn() } catch {} }',
'}',
'\\\`\\\`\\\`', '',
'| path | covered |', '|---|---|', '| happy | yes |', '| error | no |', ''
].join('\\n') }] }
const rows = [
{ id: sid + '-u' + i, role: 'user', timestamp: Date.now(),
parts: [{ type: 'text', text: 'Review question ' + i + ': does the diff in module ' + i + ' handle the error path?' }] }
]
// Agent work turn (--tools): two tool rounds before the answer, the
// shape run_conversation actually produces.
if (${tools}) {
rows.push({ id: sid + '-w' + i, role: 'assistant', timestamp: Date.now(), pending: false,
parts: [{ type: 'text', text: 'Checking module ' + i + '.' }, toolPart(sid, i, 0), toolPart(sid, i, 1)] })
}
rows.push(answer)
return rows
}
const state = (sid, rid, isStreaming) => {
const messages = []
for (let i = 0; i < ${seedTurns}; i++) messages.push(...turn(sid, i))
// Streaming tail the driver grows (--code seeds an open fence); a
// non-streaming session sits settled — open, mounted, mid-nothing.
if (isStreaming) {
messages.push({ id: sid + '-stream', role: 'assistant', timestamp: Date.now(), pending: true,
parts: [{ type: 'text', text: ${JSON.stringify(streamSeed)} }] })
}
return {
storedSessionId: sid, messages, branch: '', cwd: '', model: '', provider: '',
reasoningEffort: '', serviceTier: '', fast: false, yolo: false, personality: '',
busy: isStreaming, awaitingResponse: false,
streamId: isStreaming ? sid + '-stream' : null, sawAssistantPayload: true,
pendingBranchGroup: null, interrupted: false, interimBoundaryPending: false,
needsInput: false, turnStartedAt: isStreaming ? Date.now() : null, usage: null
}
}
// A populated recents list (--sessions): every store publish re-runs the
// busy/attention/draft projections against it, so an empty list hides
// that scaling. Restored by CLEANUP.
if (${seedSessions} > 0) {
window.__MT_SAVED_SESSIONS__ = hook.sessions()
const rows = []
for (let i = 0; i < ${seedSessions}; i++) {
rows.push({
id: 'perf-row-' + i, title: 'Seeded session ' + i, ended_at: null,
input_tokens: 1200, output_tokens: 800, is_active: false,
last_active: Date.now() - i * 60000, message_count: 12,
model: 'hermes-4', preview: 'seeded row', cwd: '/tmp/proj-' + (i % 7)
})
}
hook.seedSessions(rows)
}
// Leaked residue (--dead): sessions that ran with no surface referencing
// them and then settled — what a day of opening and closing tiles
// accumulates. Modeled on the real path (insert while busy, then the
// settle publish) so publish-time eviction, where present, engages.
// CLEANUP drops whatever survives, for builds without eviction.
window.__MT_DEAD__ = []
for (let d = 0; d < ${dead}; d++) {
const sid = 'perf-dead-' + d
const rid = 'perf-dead-rt-' + d
window.__MT_DEAD__.push(rid)
const settled = state(sid, rid, false)
hook.publish(rid, { ...settled, busy: true })
hook.publish(rid, settled)
}
// Zone leaders open as visible splits (right of the workspace, then
// subdividing that column into a grid); followers stack as tabs into
// their zone. zones=1 keeps the classic one-stack workload.
const perZone = Math.ceil(${tiles} / ${zones})
const leaders = []
// Streaming slots go to zone LEADERS first (rank orders round-robin across
// zones), so --streaming ${'$'}{zones} means "every VISIBLE transcript streams,
// every hidden tab idles" — the split the all-vs-visible snapshots diff.
window.__MT__ = { ids: [], leaders, streaming: [], timer: null }
for (let n = 1; n <= ${tiles}; n++) {
const sid = 'perf-tile-' + n
const rid = 'perf-rt-' + n
window.__MT__.ids.push({ sid, rid })
const zone = ${zones} > 1 ? Math.floor((n - 1) / perZone) : 0
const posInZone = ${zones} > 1 ? (n - 1) % perZone : n - 1
const rank = posInZone * ${zones} + zone
const isStreaming = rank < ${streaming}
if (isStreaming) window.__MT__.streaming.push(rid)
const leader = leaders[zone]
if (leader) {
hook.open(sid, 'center', 'session-tile:' + leader)
} else if (${zones} === 1) {
hook.open(sid, 'center')
} else {
leaders[zone] = sid
if (zone === 0) hook.open(sid, 'right')
else if (zone === 1) hook.open(sid, 'bottom', 'session-tile:' + leaders[0])
else hook.open(sid, 'right', 'session-tile:' + leaders[zone - 2])
}
hook.patch(sid, { runtimeId: rid })
hook.update(rid, () => state(sid, rid, isStreaming))
}
return 'ok'
})()
`
// Activate every tab once so keep-alive mounts the full stack (lazy mount:
// a never-activated tab stays unmounted, which would understate the cost).
const reveal = sid => `window.__HERMES_LAYOUT_TREE__.reveal(${JSON.stringify(`session-tile:${sid}`)})`
/** Page-side driver: grow every tile's streaming tail by `chunk` each
* `intervalMs`, through the same write path the gateway flush uses.
*
* With `tools`, the stream is a working AGENT turn, not a monologue: every
* 12th tick opens a live tool call on the streaming message (args, no
* result — the running spinner), every 12th+6 completes it with a result
* blob, and text keeps flowing between rounds. That exercises the tool-part
* update path (find + replace inside the parts array) and the ToolCall
* renderer's pending→complete transitions, which text-only streaming never
* touches. */
const drive = (chunk, intervalMs, totalTokens, tools) => `
(() => {
const hook = window.__HERMES_SESSION_TILES__
let pushed = 0
const tick = () => {
for (const rid of window.__MT__.streaming) {
hook.update(rid, prev => {
if (!prev.streamId) return prev
const messages = prev.messages.map(m => {
if (m.id !== prev.streamId) return m
const parts = m.parts.slice()
if (${tools} && pushed % 12 === 0) {
const args = { command: 'npm test -- --run suite-' + pushed, background: false }
parts.push({ type: 'tool-call', toolCallId: rid + '-live-' + pushed, toolName: 'terminal',
args, argsText: JSON.stringify(args) })
} else if (${tools} && pushed % 12 === 6) {
for (let p = parts.length - 1; p >= 0; p--) {
const part = parts[p]
if (part.type === 'tool-call' && part.result === undefined) {
parts[p] = { ...part, result: JSON.stringify({ success: true,
output: 'suite-' + pushed + ': 214 passed, 0 failed\\n'.repeat(12) }) }
break
}
}
parts.push({ type: 'text', text: '' })
} else {
const last = parts[parts.length - 1]
if (last && last.type === 'text') {
parts[parts.length - 1] = { type: 'text', text: last.text + ${JSON.stringify(chunk)} }
} else {
parts.push({ type: 'text', text: ${JSON.stringify(chunk)} })
}
}
return { ...m, parts }
})
return { ...prev, messages }
})
}
pushed += 1
if (pushed < ${totalTokens}) window.__MT__.timer = setTimeout(tick, ${intervalMs})
else window.__MT__.done = true
}
window.__MT__.timer = setTimeout(tick, ${intervalMs})
return 'driving'
})()
`
const CLEANUP = `
(() => {
const hook = window.__HERMES_SESSION_TILES__
if (window.__MT_DEAD__) {
for (const rid of window.__MT_DEAD__) hook.drop?.(rid)
window.__MT_DEAD__ = null
}
if (window.__MT__) {
clearTimeout(window.__MT__.timer)
for (const { sid, rid } of window.__MT__.ids) {
// Settle through the real path so the in-flight journal entry clears.
hook.update(rid, prev => ({ ...prev, busy: false, streamId: null }))
hook.close(sid)
}
window.__MT__ = null
}
if (window.__MT_SAVED_SESSIONS__) {
hook.seedSessions(window.__MT_SAVED_SESSIONS__)
window.__MT_SAVED_SESSIONS__ = null
}
return 'cleaned'
})()
`
export default {
name: 'multitab',
tier: 'ci',
description: 'N mounted session-tile tabs all streaming: frame pacing + longtasks.',
async run(cdp, opts = {}) {
const tiles = Number(opts.tiles ?? 5)
const zones = Number(opts.zones ?? 1)
const seedTurns = Number(opts.turns ?? 20)
const seedSessions = Number(opts.sessions ?? 0)
const streaming = Math.min(Number(opts.streaming ?? tiles), tiles)
const dead = Number(opts.dead ?? 0)
// --tools: seeded turns carry settled tool rounds and the live stream
// opens/completes tool calls between text — an agent working, not talking.
const tools = Boolean(opts.tools)
const tokens = Number(opts.tokens ?? 240)
// Matches STREAM_DELTA_FLUSH_MS — one publish per session per real flush.
const intervalMs = Number(opts.intervalMs ?? 33)
// --code: every tile grows ONE giant fenced code block with no settle
// boundaries — what a coding agent streams. The block re-parses and
// re-renders fully every flush (block memoization can't settle it), the
// documented worst case and the "5 tabs all coding" crawl.
const chunk = opts.code
? ' const value = await resolve(ctx, { retry: true }) // step\n'
: (opts.chunk ?? 'A streamed review sentence with **bold**, `code`, and ordinary prose.\n\n')
const streamSeed = opts.code ? '```ts\n' : ''
await cdp.send('Runtime.enable')
const ok = await cdp.eval(setup(tiles, seedTurns, streamSeed, zones, seedSessions, streaming, dead, tools))
if (ok !== 'ok') {
throw new Error(`multitab setup failed (${ok}) — dev hooks missing? (needs a dev/probe renderer)`)
}
// Mount every tab (keep-alive mounts on first activation), then settle.
// Each reveal is timed to the next paint — with deep transcripts the
// first mount is the "why does switching tabs hang" number.
const revealMs = []
for (let n = 1; n <= tiles; n++) {
const ms = Number(
await cdp.eval(`
new Promise(resolve => {
const t0 = performance.now()
${reveal(`perf-tile-${n}`)}
requestAnimationFrame(() => requestAnimationFrame(() => resolve(performance.now() - t0)))
})
`)
)
revealMs.push(ms)
await sleep(350)
}
// Front each zone's leader so the visible set is one transcript per zone
// (the reveal loop above leaves each zone on its LAST tab).
if (zones > 1) {
const leaders = JSON.parse(await cdp.eval('JSON.stringify(window.__MT__.leaders)'))
for (const sid of leaders) {
await cdp.eval(reveal(sid))
await sleep(150)
}
}
await sleep(1000)
await cdp.eval(RECORDERS)
await cdp.eval(drive(chunk, intervalMs, tokens, tools))
await sleep(tokens * intervalMs + 1500)
const data = JSON.parse(await cdp.eval(COLLECT))
await cdp.eval(CLEANUP)
// Drop the first 500ms (recorder install + settle).
const frames = []
let acc = 0
for (const f of data.frames) {
acc += f
if (acc >= 500) {
frames.push(f)
}
}
const ltDurations = data.longtasks.map(e => e.duration)
const windowS = frames.reduce((a, b) => a + b, 0) / 1000
// The felt numbers: sustained fps over the window, and the fps of the
// worst 1-second slice (a 333ms frame IS "3fps" even if the average looks
// fine). Worst slice = max summed frame time in any sliding 1s window.
const avgFps = windowS ? frames.length / windowS : 0
let worstFps = avgFps
for (let i = 0, j = 0, sum = 0; j < frames.length; j++) {
sum += frames[j]
while (sum > 1000) {
sum -= frames[i++]
}
// Only a window that actually spans ~1s counts; short prefixes don't.
if (sum >= 900) {
worstFps = Math.min(worstFps, ((j - i + 1) / sum) * 1000)
}
}
return {
metrics: {
longtasks_n: data.longtasks.length,
longtask_max_ms: Math.round((ltDurations.length ? Math.max(...ltDurations) : 0) * 10) / 10,
frame_p95_ms: Math.round(percentile(frames, 0.95) * 10) / 10,
frame_p99_ms: Math.round(percentile(frames, 0.99) * 10) / 10,
slow_frames_33: frames.filter(f => f > 33).length,
reveal_max_ms: Math.round(Math.max(...revealMs) * 10) / 10
},
detail: {
tiles,
zones,
streaming,
dead,
sessions: seedSessions,
tools,
turns: seedTurns,
code: Boolean(opts.code),
windowS: Math.round(windowS * 10) / 10,
avgFps: Math.round(avgFps * 10) / 10,
worstSecondFps: Math.round(worstFps * 10) / 10,
revealMs: revealMs.map(v => Math.round(v)),
frameHistogram: frameHistogram(frames)
}
}
}
}