RunMainTask drives one MAIN_TASK conversation loop to its end. It sends messages with the configured tool definitions, executes any tool calls returned by the model, and collects review comments until task_done is called or limits are reached. Token usage and warnings are aggregated on the Runner ac
(ctx context.Context, messages []llm.Message, taskKey string)
| 372 | // Review calls this once per review round, so one review subtask can drive |
| 373 | // several of these conversations in sequence. |
| 374 | func (r *Runner) RunMainTask(ctx context.Context, messages []llm.Message, taskKey string) (bool, MainLoopStop, error) { |
| 375 | // Every round of this loop re-sends the growing conversation, so each |
| 376 | // request is a prefix extension of the previous one — exactly what |
| 377 | // provider prompt caches reuse. Scope the affinity key to this subtask's |
| 378 | // main-task conversation so every round routes to the same cache node. |
| 379 | ctx = llm.ContextWithSessionKey(ctx, |
| 380 | llm.SessionTaskKey(r.deps.Session.SessionID, string(session.MainTask), taskKey)) |
| 381 | |
| 382 | toolReqCount := r.deps.Template.MaxToolRequestTimes |
| 383 | const maxConsecutiveEmptyRounds = 3 |
| 384 | consecutiveEmptyRounds := 0 |
| 385 | sessionID := uuid.NewString() |
| 386 | |
| 387 | // Async compression is owned by this conversation alone; the deferred |
| 388 | // cancel aborts any job still in flight when the conversation ends. |
| 389 | st := &compressionState{} |
| 390 | defer r.cancelPendingCompression(st) |
| 391 | |
| 392 | // stop defaults to StopMaxRounds: if the for-loop exits because toolReqCount |
| 393 | // reached zero, the run stopped on the round budget. The empty-round, |
| 394 | // compression and token-budget breaks overwrite it at their trigger points. |
| 395 | stop := StopMaxRounds |
| 396 | for toolReqCount > 0 { |
| 397 | select { |
| 398 | case <-ctx.Done(): |
| 399 | return false, StopNone, ctx.Err() |
| 400 | default: |
| 401 | } |
| 402 | |
| 403 | // The aggregate budget is checked here, before the request that would |
| 404 | // spend past it, rather than only at subtask dispatch: a conversation |
| 405 | // grows with every round and re-sends its whole history, so the one |
| 406 | // long group is exactly the spender a between-subtasks gate never sees. |
| 407 | if r.tokenBudgetExceeded() { |
| 408 | stop = StopTokenBudget |
| 409 | break |
| 410 | } |
| 411 | |
| 412 | toolReqCount-- |
| 413 | |
| 414 | fs := r.deps.Session.GetOrCreateFileSession(taskKey) |
| 415 | rec := fs.AppendTaskRecord(session.MainTask, append([]llm.Message(nil), messages...)) |
| 416 | startTime := time.Now() |
| 417 | |
| 418 | // Scoped to this round: ctx itself must stay identity-free so each |
| 419 | // iteration's meta replaces the previous one instead of nesting. |
| 420 | reqCtx := r.requestCtx(ctx, taskKey, session.MainTask, rec.RequestNo) |
| 421 | |
| 422 | _, llmSpan := telemetry.StartLLMSpan(ctx, r.deps.Model) |
| 423 | resp, err := r.deps.LLMClient.CompletionsWithCtx(reqCtx, llm.ChatRequest{ |
| 424 | Model: r.deps.Model, |
| 425 | Messages: messages, |
| 426 | Tools: r.deps.MainToolDefs, |
| 427 | MaxTokens: r.deps.Template.CompletionTokenLimit(), |
| 428 | SessionID: sessionID, |
| 429 | }) |
| 430 | duration := time.Since(startTime) |
| 431 | if err != nil { |