diff --git a/.mcp.json b/.mcp.json index ddd0b14..feab372 100644 --- a/.mcp.json +++ b/.mcp.json @@ -4,7 +4,7 @@ "command": "maf-doctor", "args": [], "env": { - "MAF_DOCTOR_INIT_VERSION": "1.15.0", + "MAF_DOCTOR_INIT_VERSION": "1.18.0", "MAF_DOCTOR_WORKSPACE_ROOTS": "C:\\Users\\jesseliberty\\ai\\.net\\blogWriter;E:\\ai\\.net\\blog\\blogMigration---public;E:\\ai\\.net\\blog\\BlogWriter" } } diff --git a/.vscode/mcp.json b/.vscode/mcp.json index 79fca00..38d6c5c 100644 --- a/.vscode/mcp.json +++ b/.vscode/mcp.json @@ -5,7 +5,7 @@ "command": "maf-doctor", "args": [], "env": { - "MAF_DOCTOR_INIT_VERSION": "1.15.0", + "MAF_DOCTOR_INIT_VERSION": "1.18.0", "MAF_DOCTOR_WORKSPACE_ROOTS": "C:\\Users\\jesseliberty\\ai\\.net\\blogWriter;E:\\ai\\.net\\blog\\blogMigration---public;E:\\ai\\.net\\blog\\BlogWriter" } }, diff --git a/BlogWriter.Tests/TestChatClients.cs b/BlogWriter.Tests/TestChatClients.cs index 09deeaa..0c821ee 100644 --- a/BlogWriter.Tests/TestChatClients.cs +++ b/BlogWriter.Tests/TestChatClients.cs @@ -7,6 +7,8 @@ internal sealed class FakeChatClient : IChatClient { private readonly Func _usageFactory; + public ChatOptions? LastOptions { get; private set; } + public FakeChatClient(Func usageFactory) => _usageFactory = usageFactory; public FakeChatClient(long totalTokens) : this(() => new UsageDetails { TotalTokenCount = totalTokens }) @@ -16,6 +18,7 @@ internal sealed class FakeChatClient : IChatClient public Task GetResponseAsync( IEnumerable messages, ChatOptions? options = null, CancellationToken cancellationToken = default) { + LastOptions = options; var response = new ChatResponse(new ChatMessage(ChatRole.Assistant, "ok")) { Usage = _usageFactory(), diff --git a/BlogWriter.Tests/TokenCapChatClientTests.cs b/BlogWriter.Tests/TokenCapChatClientTests.cs index 3d70df4..43356a7 100644 --- a/BlogWriter.Tests/TokenCapChatClientTests.cs +++ b/BlogWriter.Tests/TokenCapChatClientTests.cs @@ -51,4 +51,34 @@ public async Task SharedFactory_EnforcesOneBudgetAcrossClients() await Assert.ThrowsAsync( () => secondClient.GetResponseAsync([new ChatMessage(ChatRole.User, "second agent")])); } + + [Fact] + public async Task SharedFactory_AppliesDefaultOutputCapWithoutMutatingCallerOptions() + { + using var inner = new FakeChatClient(totalTokens: 10); + using IChatClient client = TokenCapChatClient.CreateSharedFactory( + maxTotalTokens: 100, + maxOutputTokens: 256)(inner); + var options = new ChatOptions { Temperature = 0.25f }; + + await client.GetResponseAsync([new ChatMessage(ChatRole.User, "draft")], options); + + Assert.Equal(256, inner.LastOptions?.MaxOutputTokens); + Assert.Equal(0.25f, inner.LastOptions?.Temperature); + Assert.Null(options.MaxOutputTokens); + } + + [Fact] + public async Task SharedFactory_PreservesExplicitOutputCap() + { + using var inner = new FakeChatClient(totalTokens: 10); + using IChatClient client = TokenCapChatClient.CreateSharedFactory( + maxTotalTokens: 100, + maxOutputTokens: 256)(inner); + var options = new ChatOptions { MaxOutputTokens = 128 }; + + await client.GetResponseAsync([new ChatMessage(ChatRole.User, "short answer")], options); + + Assert.Equal(128, inner.LastOptions?.MaxOutputTokens); + } } diff --git a/BlogWriterServiceCollectionExtensions.cs b/BlogWriterServiceCollectionExtensions.cs index b2c285d..46ff67a 100644 --- a/BlogWriterServiceCollectionExtensions.cs +++ b/BlogWriterServiceCollectionExtensions.cs @@ -29,6 +29,12 @@ public static IServiceCollection AddBlogWriterApplication( out long configuredMaxTokens) ? configuredMaxTokens : 40000; + int maxOutputTokens = int.TryParse( + configuration["Foundry:MaxOutputTokens"] ?? configuration["MAX_OUTPUT_TOKENS"], + out int configuredMaxOutputTokens) + ? configuredMaxOutputTokens + : TokenCapChatClient.DefaultMaxOutputTokens; + services.AddSingleton(TokenCapChatClient.CreateSharedFactory(maxTokens, maxOutputTokens)); services.AddSingleton(credential); services.AddSingleton(new CosmosClient(cosmosEndpoint.ToString(), credential)); diff --git a/HostedAgents/Author/Program.cs b/HostedAgents/Author/Program.cs index fbc00d8..8c71cda 100644 --- a/HostedAgents/Author/Program.cs +++ b/HostedAgents/Author/Program.cs @@ -15,7 +15,7 @@ string modelDeployment = Environment.GetEnvironmentVariable("AZURE_AI_MODEL_DEPLOYMENT_NAME") ?? "gpt-5-mini"; // Entra ID only — no API keys, per repository constraint. -AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential()) +AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned)) .AsAIAgent( model: modelDeployment, instructions: PromptCatalog.AuthorInstructions, diff --git a/HostedAgents/Blogger/Program.cs b/HostedAgents/Blogger/Program.cs index 0888d5a..5c620e5 100644 --- a/HostedAgents/Blogger/Program.cs +++ b/HostedAgents/Blogger/Program.cs @@ -16,7 +16,7 @@ string modelDeployment = Environment.GetEnvironmentVariable("AZURE_AI_MODEL_DEPLOYMENT_NAME") ?? "gpt-5-mini"; // Entra ID only — no API keys, per repository constraint. -AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential()) +AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned)) .AsAIAgent( model: modelDeployment, instructions: PromptCatalog.BloggerInstructions, diff --git a/HostedAgents/Researcher/Program.cs b/HostedAgents/Researcher/Program.cs index 20317f2..4df1c48 100644 --- a/HostedAgents/Researcher/Program.cs +++ b/HostedAgents/Researcher/Program.cs @@ -22,7 +22,7 @@ // Foundry model and hosted web-search authentication). // The agent owns a Foundry-native HostedWebSearchTool(). Web searches, pagination, and document retrievals execute entirely inside the remote hosted process in Azure, not locally on the client machine. -AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential()) +AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned)) .AsAIAgent( model: modelDeployment, instructions: PromptCatalog.ResearcherInstructions, diff --git a/HostedAgents/Reviewer/Program.cs b/HostedAgents/Reviewer/Program.cs index 517c03e..1ae94f9 100644 --- a/HostedAgents/Reviewer/Program.cs +++ b/HostedAgents/Reviewer/Program.cs @@ -15,7 +15,7 @@ string modelDeployment = Environment.GetEnvironmentVariable("AZURE_AI_MODEL_DEPLOYMENT_NAME") ?? "gpt-5-mini"; // Entra ID only — no API keys, per repository constraint. -AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential()) +AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned)) .AsAIAgent( model: modelDeployment, instructions: PromptCatalog.ReviewerInstructions, diff --git a/Program.cs b/Program.cs index fc8dd54..85be412 100644 --- a/Program.cs +++ b/Program.cs @@ -34,6 +34,9 @@ string GetRequired(string key) => // Cumulative process-wide budget shared by all four MAF-hosted agent clients. long maxTotalTokens = long.TryParse(config["MAX_TOTAL_TOKENS"], out long configuredMaxTotalTokens) ? configuredMaxTotalTokens : 40000; +int maxOutputTokens = int.TryParse(config["MAX_OUTPUT_TOKENS"], out int configuredMaxOutputTokens) + ? configuredMaxOutputTokens + : TokenCapChatClient.DefaultMaxOutputTokens; // Entra ID only — no API keys, per repository constraint. Agent Framework owns // the Foundry transport and Responses protocol details. @@ -44,7 +47,7 @@ string GetRequired(string key) => AIProjectClient projectClient = new(foundryProjectEndpoint, azureCredential); -Func tokenCapFactory = TokenCapChatClient.CreateSharedFactory(maxTotalTokens); +Func tokenCapFactory = TokenCapChatClient.CreateSharedFactory(maxTotalTokens, maxOutputTokens); // the console app uses AIProjectClient.AsAIAgent(...) to construct a remote AIAgent (researcherLlm) targeting the hosted agent endpoint AIAgent BuildFoundryAgent(string hostedAgentName) diff --git a/TokenCapChatClient.cs b/TokenCapChatClient.cs index 61cf363..d52d6ac 100644 --- a/TokenCapChatClient.cs +++ b/TokenCapChatClient.cs @@ -12,6 +12,8 @@ namespace BlogWriter; /// public sealed class TokenCapChatClient : DelegatingChatClient { + public const int DefaultMaxOutputTokens = 8192; + // Key used by the OpenAI connector to report reasoning tokens inside // UsageDetails.AdditionalCounts (there is no dedicated top-level property). private const string ReasoningTokenCountKey = "OutputTokenDetails.ReasoningTokenCount"; @@ -30,10 +32,21 @@ private TokenCapChatClient(IChatClient innerClient, TokenBudget budget) : base(i /// Creates a MAF chat-client middleware factory whose clients share one /// cumulative process-wide token budget. /// - public static Func CreateSharedFactory(long maxTotalTokens) + public static Func CreateSharedFactory( + long maxTotalTokens, + int maxOutputTokens = DefaultMaxOutputTokens) { + if (maxOutputTokens <= 0) + { + throw new ArgumentOutOfRangeException(nameof(maxOutputTokens), maxOutputTokens, "Output token cap must be positive."); + } + var budget = new TokenBudget(maxTotalTokens); - return innerClient => new TokenCapChatClient(innerClient, budget); + return innerClient => new TokenCapChatClient( + innerClient.AsBuilder() + .ConfigureOptions(options => options.MaxOutputTokens ??= maxOutputTokens) + .Build(), + budget); } /// Cumulative token usage observed across every model round-trip so far. diff --git a/docs/maf-doctor-assessment.md b/docs/maf-doctor-assessment.md new file mode 100644 index 0000000..a95f6f6 --- /dev/null +++ b/docs/maf-doctor-assessment.md @@ -0,0 +1,82 @@ +# MAF Doctor Assessment + +**Assessment date:** 2026-10-02 +**Repository:** BlogWriter +**MAF Doctor:** 1.18.0 + +## Summary + +MAF Doctor assigned the repository a **B** health grade. It found no anti-pattern errors, prompt-lint errors, or workflow fan-out starvation risks. It reported 7 warnings and 6 agent call sites without a detected `MaxOutputTokens` cap, for 13 findings total. Four findings are marked auto-fixable; the other nine are heuristic and should be checked against the actual runtime configuration before changing code. + +The report inspected four `[MessageHandler]` methods and six `RunAsync` / `RunStreamingAsync` sites. It did not report an incomplete scan. + +## Findings and suggested actions + +### 1. Production credentials: `DefaultAzureCredential` (4 findings) + +MAF Doctor flags `DefaultAzureCredential` in these hosted-agent programs: + +- [HostedAgents/Author/Program.cs](../HostedAgents/Author/Program.cs#L18) +- [HostedAgents/Blogger/Program.cs](../HostedAgents/Blogger/Program.cs#L19) +- [HostedAgents/Researcher/Program.cs](../HostedAgents/Researcher/Program.cs#L25) +- [HostedAgents/Reviewer/Program.cs](../HostedAgents/Reviewer/Program.cs#L18) + +**Suggested action:** Confirm how each hosted agent authenticates in its deployed environment. Where production uses managed identity, use `ManagedIdentityCredential` explicitly, as required by this repository's MAF guidance. Preserve a deliberate local-development credential path if developers need one; do not let a production deployment fall back to credential-chain discovery. MAF Doctor marks these findings as mechanically auto-fixable, but review the preview and resulting diff before applying it, then build and test the affected hosts. + +### 2. Agent calls without a detected output-token cap (6 heuristic findings) + +MAF Doctor recommends setting `MaxOutputTokens` on the relevant `ChatOptions`. These findings are heuristic; in particular, workflow `RunAsync` calls may inherit limits from the underlying agents or chat-client configuration. + +- [AuthorAgent.cs](../AuthorAgent.cs#L68) +- [BlogWriterSessionService.cs](../BlogWriterSessionService.cs#L27) +- [BlogWriterSessionService.cs](../BlogWriterSessionService.cs#L57) +- [BloggerAgent.cs](../BloggerAgent.cs#L115) +- [ResearcherAgent.cs](../ResearcherAgent.cs#L51) +- [ReviewerAgent.cs](../ReviewerAgent.cs#L61) + +**Suggested action:** Trace each listed call to the actual agent/chat-client options and verify whether an output-token limit is already applied. If not, choose limits based on expected output size and product needs, and apply them at the owning configuration point. Check that normal long-form drafts and structured responses still complete acceptably; avoid adding duplicate or ineffective caps at workflow call sites. + +### 3. Agent chat pipelines without detected OpenTelemetry setup (3 heuristic findings) + +MAF Doctor did not detect `UseOpenTelemetry` in the files that construct these agents: + +- [AuthorAgent.cs](../AuthorAgent.cs#L34) +- [BloggerAgent.cs](../BloggerAgent.cs#L43) +- [ReviewerAgent.cs](../ReviewerAgent.cs#L34) + +**Suggested action:** Check whether telemetry is configured in a shared chat-client factory, host startup, or another layer the file-level heuristic cannot see. If traces and metrics for agent calls are not emitted, configure OpenTelemetry on the `IChatClient` pipeline and ensure the host has an exporter and appropriate filtering. Verify spans are emitted, and avoid duplicate instrumentation if telemetry is already wired elsewhere. + +## Recommended order + +1. Review the four credential findings first because they affect deployed identity behavior. Preview the deterministic autofixes, inspect the diff, and apply only if the result matches the deployment design. +2. Verify whether the six call sites truly lack effective output-token limits. Add limits only where the underlying agent configuration has none. +3. Verify whether the three agent pipelines already emit telemetry. Wire OpenTelemetry only for pipelines that are actually uninstrumented. +4. Build and run relevant tests after changes, then rerun `maf-doctor doctor` and compare the grade and findings. + +## Remediation Progress + +After the credential and output-cap changes on `fix/maf-doctor-findings`: + +- All four hosted agents use an explicit system-assigned managed identity. Each hosted-agent project builds without warnings. +- The shared chat-client factory now supplies an 8192-token per-response default while preserving explicit per-call values. Override it with `MAX_OUTPUT_TOKENS` in the console app, or `Foundry:MaxOutputTokens` / `MAX_OUTPUT_TOKENS` in the web app. +- The latest maf-doctor scan remains grade **B** and still reports the six `RunAsync` cost heuristics because the scanner does not follow the shared chat-client wrapper. The focused tests verify the cap is applied before the underlying chat client and that caller options are not mutated. +- Three `UseOpenTelemetry` heuristics remain. No telemetry changes were made; the Foundry tracing guidance requires opening its trace viewer first, and the available VS Code command tool is restricted to workspace-creation flows. +- Validation: all 92 console tests, all 89 web tests, and builds of all four hosted-agent projects pass. + +## Commands + +Preview supported mechanical fixes without writing files by calling `MafAutoFixAll` in an MCP client with `repoPath` set to the repository root and `dryRun: true` (the default). + +Apply the CLI autofixes only after reviewing the preview: + +```powershell +maf-doctor autofix-all "e:\AI\.NET\blog\BlogWriter" --apply +``` + +Reassess after remediation: + +```powershell +maf-doctor doctor "e:\AI\.NET\blog\BlogWriter" +``` + +Treat this assessment as advisory: verify heuristic findings against runtime and host configuration before changing code.