Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .mcp.json
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
"command": "maf-doctor",
"args": [],
"env": {
"MAF_DOCTOR_INIT_VERSION": "1.15.0",
"MAF_DOCTOR_INIT_VERSION": "1.18.0",
"MAF_DOCTOR_WORKSPACE_ROOTS": "C:\\Users\\jesseliberty\\ai\\.net\\blogWriter;E:\\ai\\.net\\blog\\blogMigration---public;E:\\ai\\.net\\blog\\BlogWriter"
}
}
Expand Down
2 changes: 1 addition & 1 deletion .vscode/mcp.json
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
"command": "maf-doctor",
"args": [],
"env": {
"MAF_DOCTOR_INIT_VERSION": "1.15.0",
"MAF_DOCTOR_INIT_VERSION": "1.18.0",
"MAF_DOCTOR_WORKSPACE_ROOTS": "C:\\Users\\jesseliberty\\ai\\.net\\blogWriter;E:\\ai\\.net\\blog\\blogMigration---public;E:\\ai\\.net\\blog\\BlogWriter"
}
},
Expand Down
3 changes: 3 additions & 0 deletions BlogWriter.Tests/TestChatClients.cs
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,8 @@ internal sealed class FakeChatClient : IChatClient
{
private readonly Func<UsageDetails?> _usageFactory;

public ChatOptions? LastOptions { get; private set; }

public FakeChatClient(Func<UsageDetails?> usageFactory) => _usageFactory = usageFactory;

public FakeChatClient(long totalTokens) : this(() => new UsageDetails { TotalTokenCount = totalTokens })
Expand All @@ -16,6 +18,7 @@ internal sealed class FakeChatClient : IChatClient
public Task<ChatResponse> GetResponseAsync(
IEnumerable<ChatMessage> messages, ChatOptions? options = null, CancellationToken cancellationToken = default)
{
LastOptions = options;
var response = new ChatResponse(new ChatMessage(ChatRole.Assistant, "ok"))
{
Usage = _usageFactory(),
Expand Down
30 changes: 30 additions & 0 deletions BlogWriter.Tests/TokenCapChatClientTests.cs
Original file line number Diff line number Diff line change
Expand Up @@ -51,4 +51,34 @@ public async Task SharedFactory_EnforcesOneBudgetAcrossClients()
await Assert.ThrowsAsync<TokenCapExceededException>(
() => secondClient.GetResponseAsync([new ChatMessage(ChatRole.User, "second agent")]));
}

[Fact]
public async Task SharedFactory_AppliesDefaultOutputCapWithoutMutatingCallerOptions()
{
using var inner = new FakeChatClient(totalTokens: 10);
using IChatClient client = TokenCapChatClient.CreateSharedFactory(
maxTotalTokens: 100,
maxOutputTokens: 256)(inner);
var options = new ChatOptions { Temperature = 0.25f };

await client.GetResponseAsync([new ChatMessage(ChatRole.User, "draft")], options);

Assert.Equal(256, inner.LastOptions?.MaxOutputTokens);
Assert.Equal(0.25f, inner.LastOptions?.Temperature);
Assert.Null(options.MaxOutputTokens);
}

[Fact]
public async Task SharedFactory_PreservesExplicitOutputCap()
{
using var inner = new FakeChatClient(totalTokens: 10);
using IChatClient client = TokenCapChatClient.CreateSharedFactory(
maxTotalTokens: 100,
maxOutputTokens: 256)(inner);
var options = new ChatOptions { MaxOutputTokens = 128 };

await client.GetResponseAsync([new ChatMessage(ChatRole.User, "short answer")], options);

Assert.Equal(128, inner.LastOptions?.MaxOutputTokens);
}
}
6 changes: 6 additions & 0 deletions BlogWriterServiceCollectionExtensions.cs
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,12 @@ public static IServiceCollection AddBlogWriterApplication(
out long configuredMaxTokens)
? configuredMaxTokens
: 40000;
int maxOutputTokens = int.TryParse(
configuration["Foundry:MaxOutputTokens"] ?? configuration["MAX_OUTPUT_TOKENS"],
out int configuredMaxOutputTokens)
? configuredMaxOutputTokens
: TokenCapChatClient.DefaultMaxOutputTokens;
services.AddSingleton(TokenCapChatClient.CreateSharedFactory(maxTokens, maxOutputTokens));

services.AddSingleton(credential);
services.AddSingleton(new CosmosClient(cosmosEndpoint.ToString(), credential));
Expand Down
2 changes: 1 addition & 1 deletion HostedAgents/Author/Program.cs
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@
string modelDeployment = Environment.GetEnvironmentVariable("AZURE_AI_MODEL_DEPLOYMENT_NAME") ?? "gpt-5-mini";

// Entra ID only — no API keys, per repository constraint.
AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential())
AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned))
.AsAIAgent(
model: modelDeployment,
instructions: PromptCatalog.AuthorInstructions,
Expand Down
2 changes: 1 addition & 1 deletion HostedAgents/Blogger/Program.cs
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@
string modelDeployment = Environment.GetEnvironmentVariable("AZURE_AI_MODEL_DEPLOYMENT_NAME") ?? "gpt-5-mini";

// Entra ID only — no API keys, per repository constraint.
AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential())
AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned))
.AsAIAgent(
model: modelDeployment,
instructions: PromptCatalog.BloggerInstructions,
Expand Down
2 changes: 1 addition & 1 deletion HostedAgents/Researcher/Program.cs
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@
// Foundry model and hosted web-search authentication).

// The agent owns a Foundry-native HostedWebSearchTool(). Web searches, pagination, and document retrievals execute entirely inside the remote hosted process in Azure, not locally on the client machine.
AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential())
AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned))
.AsAIAgent(
model: modelDeployment,
instructions: PromptCatalog.ResearcherInstructions,
Expand Down
2 changes: 1 addition & 1 deletion HostedAgents/Reviewer/Program.cs
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@
string modelDeployment = Environment.GetEnvironmentVariable("AZURE_AI_MODEL_DEPLOYMENT_NAME") ?? "gpt-5-mini";

// Entra ID only — no API keys, per repository constraint.
AIAgent agent = new AIProjectClient(projectEndpoint, new DefaultAzureCredential())
AIAgent agent = new AIProjectClient(projectEndpoint, new ManagedIdentityCredential(ManagedIdentityId.SystemAssigned))
.AsAIAgent(
model: modelDeployment,
instructions: PromptCatalog.ReviewerInstructions,
Expand Down
5 changes: 4 additions & 1 deletion Program.cs
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,9 @@ string GetRequired(string key) =>

// Cumulative process-wide budget shared by all four MAF-hosted agent clients.
long maxTotalTokens = long.TryParse(config["MAX_TOTAL_TOKENS"], out long configuredMaxTotalTokens) ? configuredMaxTotalTokens : 40000;
int maxOutputTokens = int.TryParse(config["MAX_OUTPUT_TOKENS"], out int configuredMaxOutputTokens)
? configuredMaxOutputTokens
: TokenCapChatClient.DefaultMaxOutputTokens;

// Entra ID only — no API keys, per repository constraint. Agent Framework owns
// the Foundry transport and Responses protocol details.
Expand All @@ -44,7 +47,7 @@ string GetRequired(string key) =>
AIProjectClient projectClient = new(foundryProjectEndpoint, azureCredential);


Func<IChatClient, IChatClient> tokenCapFactory = TokenCapChatClient.CreateSharedFactory(maxTotalTokens);
Func<IChatClient, IChatClient> tokenCapFactory = TokenCapChatClient.CreateSharedFactory(maxTotalTokens, maxOutputTokens);

// the console app uses AIProjectClient.AsAIAgent(...) to construct a remote AIAgent (researcherLlm) targeting the hosted agent endpoint
AIAgent BuildFoundryAgent(string hostedAgentName)
Expand Down
17 changes: 15 additions & 2 deletions TokenCapChatClient.cs
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,8 @@ namespace BlogWriter;
/// </summary>
public sealed class TokenCapChatClient : DelegatingChatClient
{
public const int DefaultMaxOutputTokens = 8192;

// Key used by the OpenAI connector to report reasoning tokens inside
// UsageDetails.AdditionalCounts (there is no dedicated top-level property).
private const string ReasoningTokenCountKey = "OutputTokenDetails.ReasoningTokenCount";
Expand All @@ -30,10 +32,21 @@ private TokenCapChatClient(IChatClient innerClient, TokenBudget budget) : base(i
/// Creates a MAF chat-client middleware factory whose clients share one
/// cumulative process-wide token budget.
/// </summary>
public static Func<IChatClient, IChatClient> CreateSharedFactory(long maxTotalTokens)
public static Func<IChatClient, IChatClient> CreateSharedFactory(
long maxTotalTokens,
int maxOutputTokens = DefaultMaxOutputTokens)
{
if (maxOutputTokens <= 0)
{
throw new ArgumentOutOfRangeException(nameof(maxOutputTokens), maxOutputTokens, "Output token cap must be positive.");
}

var budget = new TokenBudget(maxTotalTokens);
return innerClient => new TokenCapChatClient(innerClient, budget);
return innerClient => new TokenCapChatClient(
innerClient.AsBuilder()
.ConfigureOptions(options => options.MaxOutputTokens ??= maxOutputTokens)
.Build(),
budget);
}

/// <summary>Cumulative token usage observed across every model round-trip so far.</summary>
Expand Down
82 changes: 82 additions & 0 deletions docs/maf-doctor-assessment.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,82 @@
# MAF Doctor Assessment

**Assessment date:** 2026-10-02
**Repository:** BlogWriter
**MAF Doctor:** 1.18.0

## Summary

MAF Doctor assigned the repository a **B** health grade. It found no anti-pattern errors, prompt-lint errors, or workflow fan-out starvation risks. It reported 7 warnings and 6 agent call sites without a detected `MaxOutputTokens` cap, for 13 findings total. Four findings are marked auto-fixable; the other nine are heuristic and should be checked against the actual runtime configuration before changing code.

The report inspected four `[MessageHandler]` methods and six `RunAsync` / `RunStreamingAsync` sites. It did not report an incomplete scan.

## Findings and suggested actions

### 1. Production credentials: `DefaultAzureCredential` (4 findings)

MAF Doctor flags `DefaultAzureCredential` in these hosted-agent programs:

- [HostedAgents/Author/Program.cs](../HostedAgents/Author/Program.cs#L18)
- [HostedAgents/Blogger/Program.cs](../HostedAgents/Blogger/Program.cs#L19)
- [HostedAgents/Researcher/Program.cs](../HostedAgents/Researcher/Program.cs#L25)
- [HostedAgents/Reviewer/Program.cs](../HostedAgents/Reviewer/Program.cs#L18)

**Suggested action:** Confirm how each hosted agent authenticates in its deployed environment. Where production uses managed identity, use `ManagedIdentityCredential` explicitly, as required by this repository's MAF guidance. Preserve a deliberate local-development credential path if developers need one; do not let a production deployment fall back to credential-chain discovery. MAF Doctor marks these findings as mechanically auto-fixable, but review the preview and resulting diff before applying it, then build and test the affected hosts.

### 2. Agent calls without a detected output-token cap (6 heuristic findings)

MAF Doctor recommends setting `MaxOutputTokens` on the relevant `ChatOptions`. These findings are heuristic; in particular, workflow `RunAsync` calls may inherit limits from the underlying agents or chat-client configuration.

- [AuthorAgent.cs](../AuthorAgent.cs#L68)
- [BlogWriterSessionService.cs](../BlogWriterSessionService.cs#L27)
- [BlogWriterSessionService.cs](../BlogWriterSessionService.cs#L57)
- [BloggerAgent.cs](../BloggerAgent.cs#L115)
- [ResearcherAgent.cs](../ResearcherAgent.cs#L51)
- [ReviewerAgent.cs](../ReviewerAgent.cs#L61)

**Suggested action:** Trace each listed call to the actual agent/chat-client options and verify whether an output-token limit is already applied. If not, choose limits based on expected output size and product needs, and apply them at the owning configuration point. Check that normal long-form drafts and structured responses still complete acceptably; avoid adding duplicate or ineffective caps at workflow call sites.

### 3. Agent chat pipelines without detected OpenTelemetry setup (3 heuristic findings)

MAF Doctor did not detect `UseOpenTelemetry` in the files that construct these agents:

- [AuthorAgent.cs](../AuthorAgent.cs#L34)
- [BloggerAgent.cs](../BloggerAgent.cs#L43)
- [ReviewerAgent.cs](../ReviewerAgent.cs#L34)

**Suggested action:** Check whether telemetry is configured in a shared chat-client factory, host startup, or another layer the file-level heuristic cannot see. If traces and metrics for agent calls are not emitted, configure OpenTelemetry on the `IChatClient` pipeline and ensure the host has an exporter and appropriate filtering. Verify spans are emitted, and avoid duplicate instrumentation if telemetry is already wired elsewhere.

## Recommended order

1. Review the four credential findings first because they affect deployed identity behavior. Preview the deterministic autofixes, inspect the diff, and apply only if the result matches the deployment design.
2. Verify whether the six call sites truly lack effective output-token limits. Add limits only where the underlying agent configuration has none.
3. Verify whether the three agent pipelines already emit telemetry. Wire OpenTelemetry only for pipelines that are actually uninstrumented.
4. Build and run relevant tests after changes, then rerun `maf-doctor doctor` and compare the grade and findings.

## Remediation Progress

After the credential and output-cap changes on `fix/maf-doctor-findings`:

- All four hosted agents use an explicit system-assigned managed identity. Each hosted-agent project builds without warnings.
- The shared chat-client factory now supplies an 8192-token per-response default while preserving explicit per-call values. Override it with `MAX_OUTPUT_TOKENS` in the console app, or `Foundry:MaxOutputTokens` / `MAX_OUTPUT_TOKENS` in the web app.
- The latest maf-doctor scan remains grade **B** and still reports the six `RunAsync` cost heuristics because the scanner does not follow the shared chat-client wrapper. The focused tests verify the cap is applied before the underlying chat client and that caller options are not mutated.
- Three `UseOpenTelemetry` heuristics remain. No telemetry changes were made; the Foundry tracing guidance requires opening its trace viewer first, and the available VS Code command tool is restricted to workspace-creation flows.
- Validation: all 92 console tests, all 89 web tests, and builds of all four hosted-agent projects pass.

## Commands

Preview supported mechanical fixes without writing files by calling `MafAutoFixAll` in an MCP client with `repoPath` set to the repository root and `dryRun: true` (the default).

Apply the CLI autofixes only after reviewing the preview:

```powershell
maf-doctor autofix-all "e:\AI\.NET\blog\BlogWriter" --apply
```

Reassess after remediation:

```powershell
maf-doctor doctor "e:\AI\.NET\blog\BlogWriter"
```

Treat this assessment as advisory: verify heuristic findings against runtime and host configuration before changing code.
Loading