Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
181fd5b
Add MCP server endpoint for AI agent doc access
joebutler2 Sep 3, 2026
37090e8
Optimize devdocs_list_docsets MCP tool for context efficiency
joebutler2 Sep 9, 2026
b4067a0
Add slug validation and error handling for production deployment
joebutler2 Sep 9, 2026
fb18b3a
Fix error handling and unknown tool handling
joebutler2 Sep 12, 2026
4031504
Add input validation against declared tool schemas
joebutler2 Sep 12, 2026
73cd5ed
Fix HTML text extraction to preserve block element separators
joebutler2 Sep 12, 2026
4b313ee
Add pagination and validation to devdocs_search tool
joebutler2 Sep 12, 2026
a8aa4ea
Optimize db.json parsing with in-memory caching
joebutler2 Sep 12, 2026
c5969a1
Add test for JSON parse error handling
joebutler2 Sep 12, 2026
ebe94dc
Refactor error handling in mcp_test.rb
joebutler2 Sep 14, 2026
9225cb1
Merge branch 'main' into feature/mcp-server
simon04 Sep 14, 2026
3a18100
Strip the fragment when looking up a page in db.json
simon04 Sep 14, 2026
3281bfe
Treat tables and definition lists as block elements
simon04 Sep 14, 2026
d402b93
Walk the HTML in document order when extracting text
simon04 Sep 14, 2026
6bbfce0
Keep the whitespace of preformatted text
simon04 Sep 14, 2026
74e052b
Read the page file instead of parsing db.json
simon04 Sep 14, 2026
c8f8527
Cache the parsed search indexes
simon04 Sep 14, 2026
f86dbb6
Drop the unreachable additionalProperties check
simon04 Sep 14, 2026
9a23f2a
Reject non-object payloads with Invalid Request
simon04 Sep 14, 2026
968b01c
Validate the params of tools/call
simon04 Sep 14, 2026
5f6e950
Do not answer JSON-RPC notifications
simon04 Sep 14, 2026
bff3d8a
Report failing tools in the result, not as protocol errors
simon04 Sep 14, 2026
b73e0cd
Assert the missing search index unconditionally
simon04 Sep 14, 2026
dca4caa
Add size-bounded cache with eviction to prevent memory exhaustion
joebutler2 Sep 15, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -16,3 +16,4 @@ docs/**/*
assets/stylesheets/components/_environment.scss
assets/stylesheets/global/_icons.scss
node_modules
.mcp.json
26 changes: 26 additions & 0 deletions lib/app.rb
Original file line number Diff line number Diff line change
Expand Up @@ -141,6 +141,7 @@ class App < Sinatra::Application

configure :test do
set :docs_manifest_path, File.join(root, 'test', 'files', 'docs.json')
set :docs_path, File.join(root, 'test', 'files', 'docs')
end

def self.parse_docs
Expand Down Expand Up @@ -338,6 +339,31 @@ def service_worker_cache_name
200
end

require 'mcp/server'

post '/mcp' do
content_type :json
begin
payload = JSON.parse(request.body.read)
response = Mcp::Server.handle(payload, settings)
if response.nil?
# The payload was a notification, which takes no response.
status 202
''
else
response.to_json
end
rescue JSON::ParserError => err
error_response(nil, -32700, "Parse error: #{err.message}").to_json
rescue => err
error_response(nil, -32603, "Internal error: #{err.message}").to_json
end
end

def error_response(id, code, message)
{ 'jsonrpc' => '2.0', 'id' => id, 'error' => { 'code' => code, 'message' => message } }
end

%w(docs.json application.js application.css).each do |asset|
class_eval <<-CODE, __FILE__, __LINE__ + 1
get '/#{asset}' do
Expand Down
341 changes: 341 additions & 0 deletions lib/mcp/server.rb
Original file line number Diff line number Diff line change
@@ -0,0 +1,341 @@
module Mcp
# Dispatches a single JSON-RPC 2.0 request (already parsed into a Hash with
# string keys) to the appropriate MCP handler and returns a response Hash
# ready to be serialized back to the client.
module Server
DB_CACHE = {}
MAX_CACHE_SIZE = 50 * 1024 * 1024
TOOLS = [
{
'name' => 'devdocs_list_docsets',
'description' => 'List documentation sets available on this DevDocs instance. Returns paginated results with optional filtering.',
'inputSchema' => {
'type' => 'object',
'properties' => {
'offset' => { 'type' => 'integer', 'description' => 'Number of results to skip (default: 0)', 'minimum' => 0 },
'limit' => { 'type' => 'integer', 'description' => 'Maximum results to return (default: 50, max: 500)', 'minimum' => 1, 'maximum' => 500 },
'query' => { 'type' => 'string', 'description' => 'Filter by slug or name (case-insensitive substring match)' },
},
'additionalProperties' => false,
},
},
{
'name' => 'devdocs_search',
'description' => 'Search entry names/paths within one downloaded DevDocs doc set. Returns paginated results.',
'inputSchema' => {
'type' => 'object',
'properties' => {
'slug' => { 'type' => 'string' },
'query' => { 'type' => 'string', 'description' => 'Non-empty search query' },
'offset' => { 'type' => 'integer', 'description' => 'Number of results to skip (default: 0)', 'minimum' => 0 },
'limit' => { 'type' => 'integer', 'description' => 'Maximum results to return (default: 50, max: 500)', 'minimum' => 1, 'maximum' => 500 },
},
'required' => %w(slug query),
'additionalProperties' => false,
},
},
{
'name' => 'devdocs_get_page',
'description' => 'Fetch one entry from a DevDocs doc set as plain text.',
'inputSchema' => {
'type' => 'object',
'properties' => {
'slug' => { 'type' => 'string' },
'path' => { 'type' => 'string' },
},
'required' => %w(slug path),
'additionalProperties' => false,
},
},
].freeze

def self.handle(request, app_settings)
unless request.is_a?(Hash)
# A batch is valid JSON-RPC, but this server takes one request at a time.
detail = request.is_a?(Array) ? 'batch requests are not supported' : 'expected a JSON-RPC object'
return error(request, -32600, "Invalid Request: #{detail}")
end

# A request without an id is a notification - notifications/initialized is
# sent by every client right after the handshake - and JSON-RPC 2.0 says
# it must not be answered, not even to report an unknown method.
return nil unless request.key?('id')

case request['method']
when 'initialize'
respond(request, {
'protocolVersion' => '2024-11-05',
'capabilities' => { 'tools' => {} },
'serverInfo' => { 'name' => 'devdocs-mcp', 'version' => '1.0.0' },
})
when 'tools/list'
respond(request, { 'tools' => TOOLS })
when 'tools/call'
call_tool(request, app_settings)
else
error(request, -32601, "Unsupported method: #{request['method']}")
end
rescue => err
error(request, -32603, "Internal error: #{err.message}")
end

def self.error(request, code, message)
id = request.is_a?(Hash) ? request['id'] : nil
{ 'jsonrpc' => '2.0', 'id' => id, 'error' => { 'code' => code, 'message' => message } }
end

def self.call_tool(request, app_settings)
params = request['params']
unless params.is_a?(Hash)
return error(request, -32602, 'Invalid params: expected an object naming the tool to call')
end

tool_name = params['name']
arguments = params['arguments'] || {}
unless arguments.is_a?(Hash)
return error(request, -32602, "Invalid params: expected arguments to be an object, got #{arguments.class}")
end

tool_def = TOOLS.find { |t| t['name'] == tool_name }
unless tool_def
return error(request, -32602, "Unknown tool: #{tool_name}")
end

validation_error = validate_arguments(arguments, tool_def['inputSchema'])
if validation_error
return error(request, -32602, validation_error)
end

case tool_name
when 'devdocs_list_docsets'
tool_result(request) { list_docsets(app_settings, arguments).to_json }
when 'devdocs_search'
tool_result(request) { search_docset(app_settings, arguments['slug'], arguments['query'], arguments).to_json }
when 'devdocs_get_page'
tool_result(request) { get_page(app_settings, arguments['slug'], arguments['path']) }
end
end

# Runs a tool and wraps the text it returns in a result. A tool that fails
# reports the reason in its result with isError, as the MCP spec asks: a
# protocol error is handled by the client and never reaches the model.
def self.tool_result(request)
respond(request, { 'content' => [{ 'type' => 'text', 'text' => yield }] })
rescue => err
respond(request, { 'content' => [{ 'type' => 'text', 'text' => err.message }], 'isError' => true })
end

def self.validate_arguments(arguments, schema)
required = schema['required'] || []
properties = schema['properties'] || {}

required.each do |field|
return "Missing required field: #{field}" unless arguments.key?(field)
end

arguments.each do |field, value|
return "Unknown field: #{field}" unless properties.key?(field)
error_msg = validate_value(value, properties[field])
return error_msg if error_msg
end

nil
end

def self.validate_value(value, schema)
type = schema['type']

case type
when 'string'
return "Expected string, got #{value.class}" unless value.is_a?(String)
when 'integer'
return "Expected integer, got #{value.class}" unless value.is_a?(Integer)
when 'number'
return "Expected number, got #{value.class}" unless value.is_a?(Numeric)
end

if schema['minimum'] && value < schema['minimum']
return "Value #{value} is below minimum #{schema['minimum']}"
end
if schema['maximum'] && value > schema['maximum']
return "Value #{value} exceeds maximum #{schema['maximum']}"
end

nil
end

def self.list_docsets(app_settings, args)
offset = (args['offset'] || 0).to_i
limit = [(args['limit'] || 50).to_i, 500].min
query = args['query']&.downcase

all_docsets = app_settings.docs.values.map do |docset|
{
'slug' => docset['slug'],
'name' => docset['name'],
'version' => docset['version'],
}
end

filtered = if query
all_docsets.select do |docset|
docset['slug'].downcase.include?(query) || docset['name'].downcase.include?(query)
end
else
all_docsets
end

total_count = filtered.length
paginated = filtered.drop(offset).take(limit)

{
'docsets' => paginated,
'offset' => offset,
'limit' => limit,
'total' => total_count,
'returned' => paginated.length,
}
end

def self.validate_slug(app_settings, slug)
unless app_settings.docs.key?(slug)
raise ArgumentError, "Invalid docset slug: #{slug}"
end
slug
end

def self.get_page(app_settings, slug, path)
validate_slug(app_settings, slug)
html_to_text(File.read(page_path(app_settings, slug, path)))
end

# Resolves an entry path to the file its page is stored in, the same way the
# client does (Entry#_filePath): entries sharing a page carry a #fragment,
# and the path leaves out the .html extension. Reading the page beats
# looking it up in db.json, which would mean parsing up to 100MB of JSON.
def self.page_path(app_settings, slug, path)
docset_path = File.expand_path(File.join(app_settings.docs_path, slug))
unless Dir.exist?(docset_path)
raise "Pages not available for #{slug}. They are served from the CDN."
end

file = path.sub(/#.*/, '')
file += '.html' unless file.end_with?('.html')
file_path = File.expand_path(File.join(docset_path, file))

# The path comes from the caller, so keep it inside the docset.
unless file_path.start_with?(docset_path + File::SEPARATOR) && File.file?(file_path)
raise "Page not found: #{path}"
end
file_path
end

def self.html_to_text(html)
doc = Nokogiri::HTML::DocumentFragment.parse(html)
segments = []
collect_text(doc, segments, false)
# Collapse whitespace outside <pre> only; code samples keep their
# indentation and blank lines verbatim.
segments.map { |segment|
segment[:pre] ? segment[:text] : segment[:text].squeeze(' ').gsub(/\n\s*\n/, "\n")
}.join.strip
end

# Walks the tree in document order, wrapping the text of each block element
# in newlines. Nokogiri's #traverse is post-order, which emitted a block's
# separator only after its text and ran the text before it into the block.
# Text is collected into runs of equal preformattedness so that the
# whitespace collapsing above can skip the preformatted ones.
def self.collect_text(node, segments, preformatted)
node.children.each do |child|
if child.text?
append_text(segments, child.text, preformatted)
elsif block_element?(child.name)
append_text(segments, "\n", preformatted)
collect_text(child, segments, preformatted || child.name.casecmp('pre').zero?)
append_text(segments, "\n", preformatted)
else
collect_text(child, segments, preformatted)
end
end
end

def self.append_text(segments, text, preformatted)
last = segments.last
if last && last[:pre] == preformatted
last[:text] << text
else
segments << { pre: preformatted, text: +text }
end
end

def self.block_element?(tag_name)
return false unless tag_name
%w(p div h1 h2 h3 h4 h5 h6 ul ol li dl dt dd
table caption thead tbody tfoot tr th td
blockquote pre br).include?(tag_name.downcase)
end

def self.search_docset(app_settings, slug, query, args = {})
raise "Query cannot be empty" if query.to_s.strip.empty?

validate_slug(app_settings, slug)

offset = (args['offset'] || 0).to_i
limit = [(args['limit'] || 50).to_i, 500].min

index = load_index(app_settings, slug)
query_lower = query.downcase

all_matches = index['entries'].select do |entry|
entry['name'].downcase.include?(query_lower) || entry['path'].downcase.include?(query_lower)
end
Comment thread
Copilot marked this conversation as resolved.

total_count = all_matches.length
paginated = all_matches.drop(offset).take(limit)

{
'entries' => paginated,
'offset' => offset,
'limit' => limit,
'total' => total_count,
'returned' => paginated.length,
}
end

# Caches the parsed index of every docset searched so far. Unlike db.json,
# the indexes are small (16.5MB for all of the docsets here), and they would
# otherwise be re-parsed on every search. A re-scraped docset is picked up
# again by way of the mtime and the size.
def self.load_index(app_settings, slug)
index_path = File.join(app_settings.docs_path, slug, 'index.json')
stat = begin
File.stat(index_path)
rescue Errno::ENOENT
raise "Search index not available for #{slug}. The search index is served from the CDN."
end

stamp = [stat.mtime, stat.size]
cached = DB_CACHE[index_path]
return cached[:index] if cached && cached[:stamp] == stamp

index = JSON.parse(File.read(index_path))
index_size = File.size(index_path)

if cache_size + index_size > MAX_CACHE_SIZE
DB_CACHE.clear
end

DB_CACHE[index_path] = { stamp: stamp, index: index }
index
end

def self.cache_size
DB_CACHE.values.sum { |entry| entry.is_a?(Hash) && entry[:index] ? entry[:index].to_json.bytesize : 0 }
end

def self.respond(request, result)
{ 'jsonrpc' => '2.0', 'id' => request['id'], 'result' => result }
end
end
end
2 changes: 1 addition & 1 deletion test/files/docs.json
Original file line number Diff line number Diff line change
@@ -1 +1 @@
[{"name":"CSS","slug":"css","type":"mdn","release":null,"mtime":1420139788,"db_size":3460507,"alias":null},{"name":"DOM","slug":"dom","type":"mdn","release":null,"mtime":1420139789,"db_size":11399128,"alias":null},{"name":"DOM Events","slug":"dom_events","type":"mdn","release":null,"mtime":1420139790,"db_size":889020,"alias":null},{"name":"HTML","slug":"html~5","type":"mdn","version":"5","mtime":1420139791,"db_size":1835647,"alias":null},{"name":"HTML","slug":"html~4","type":"mdn","version":"4","mtime":1420139790,"db_size":1835646,"alias":null},{"name":"HTTP","slug":"http","type":"rfc","release":null,"mtime":1420139790,"db_size":183083,"alias":null},{"name":"JavaScript","slug":"javascript","type":"mdn","release":null,"mtime":1420139791,"db_size":4125477,"alias":"js"}]
[{"name":"CSS","slug":"css","type":"mdn","release":null,"mtime":1420139788,"db_size":3460507,"alias":null},{"name":"DOM","slug":"dom","type":"mdn","release":null,"mtime":1420139789,"db_size":11399128,"alias":null},{"name":"DOM Events","slug":"dom_events","type":"mdn","release":null,"mtime":1420139790,"db_size":889020,"alias":null},{"name":"HTML","slug":"html~5","type":"mdn","version":"5","mtime":1420139791,"db_size":1835647,"alias":null},{"name":"HTML","slug":"html~4","type":"mdn","version":"4","mtime":1420139790,"db_size":1835646,"alias":null},{"name":"HTTP","slug":"http","type":"rfc","release":null,"mtime":1420139790,"db_size":183083,"alias":null},{"name":"JavaScript","slug":"javascript","type":"mdn","release":null,"mtime":1420139791,"db_size":4125477,"alias":"js"},{"name":"MCP Fixture","slug":"mcp_fixture","type":"test","release":null,"mtime":1420139791,"db_size":1024,"alias":null}]
1 change: 1 addition & 0 deletions test/files/docs/mcp_fixture/array/blocks.html
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
<div>Options are:<ul><li>one</li><li>two</li></ul></div>
3 changes: 3 additions & 0 deletions test/files/docs/mcp_fixture/array/code.html
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
<p>Example:</p><pre><code>def push(x)
items &lt;&lt; x
end</code></pre>
Loading