chopratejas commited on
Commit
75a9bd3
·
1 Parent(s): 2d19f32

Add headroom_read MCP tool with session caching (feature flag)

Browse files

New MCP tool that reads files with smart caching via CCR:
- First read: returns full content, stores original in CCR (1h TTL)
- Re-read (unchanged): returns ~20 token cache marker + CCR hash
- Re-read (changed): detects new content hash, returns fresh content
- CCR expired: silently falls back to fresh read (no broken markers)
- fresh=true: bypass cache for subagents or post-compaction

Behind feature flag: HEADROOM_MCP_READ=on (off by default).
Works with existing headroom_compress and headroom_retrieve —
CCR is the shared store, tools don't conflict.

Based on analysis of real Claude Code sessions: 74% of Read calls
are re-reads of files already in context (server.py read 39x in
one session). Cache markers save ~50K tokens per re-read.

Files changed (1) hide show
  1. headroom/ccr/mcp_server.py +159 -1
headroom/ccr/mcp_server.py CHANGED
@@ -64,9 +64,20 @@ except ImportError:
64
  CCR_TOOL_NAME = "headroom_retrieve"
65
  COMPRESS_TOOL_NAME = "headroom_compress"
66
  STATS_TOOL_NAME = "headroom_stats"
 
67
 
68
  logger = logging.getLogger("headroom.ccr.mcp")
69
 
 
 
 
 
 
 
 
 
 
 
70
  DEFAULT_PROXY_URL = os.environ.get("HEADROOM_PROXY_URL", "http://127.0.0.1:8787")
71
 
72
 
@@ -315,6 +326,8 @@ class HeadroomMCPServer:
315
  self._stats = SessionStats()
316
  self._local_store: Any = None # Lazy-initialized CompressionStore
317
  self._compressor_initialized = False
 
 
318
 
319
  if not MCP_AVAILABLE:
320
  raise ImportError("MCP SDK not installed. Install with: pip install mcp")
@@ -461,7 +474,7 @@ class HeadroomMCPServer:
461
 
462
  @self.server.list_tools()
463
  async def list_tools() -> list[Tool]:
464
- return [
465
  Tool(
466
  name=COMPRESS_TOOL_NAME,
467
  description=(
@@ -525,6 +538,41 @@ class HeadroomMCPServer:
525
  ),
526
  ]
527
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
528
  @self.server.call_tool()
529
  async def call_tool(name: str, arguments: dict[str, Any]) -> list[TextContent]:
530
  try:
@@ -534,6 +582,8 @@ class HeadroomMCPServer:
534
  return await self._handle_retrieve(arguments)
535
  elif name == STATS_TOOL_NAME:
536
  return await self._handle_stats()
 
 
537
  else:
538
  return [
539
  TextContent(
@@ -670,6 +720,114 @@ class HeadroomMCPServer:
670
  result["cost_saved_usd"] = cost.get("total_saved", cost.get("saved", 0))
671
  return result if result else None
672
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
673
  async def run_stdio(self) -> None:
674
  """Run the server with stdio transport."""
675
  async with stdio_server() as (read_stream, write_stream):
 
64
  CCR_TOOL_NAME = "headroom_retrieve"
65
  COMPRESS_TOOL_NAME = "headroom_compress"
66
  STATS_TOOL_NAME = "headroom_stats"
67
+ READ_TOOL_NAME = "headroom_read"
68
 
69
  logger = logging.getLogger("headroom.ccr.mcp")
70
 
71
+ # Feature flag: enable headroom_read tool (file read caching via CCR)
72
+ # Set HEADROOM_MCP_READ=on to enable
73
+ _READ_ENABLED = os.environ.get("HEADROOM_MCP_READ", "off").lower().strip() in (
74
+ "on",
75
+ "true",
76
+ "1",
77
+ "yes",
78
+ "enabled",
79
+ )
80
+
81
  DEFAULT_PROXY_URL = os.environ.get("HEADROOM_PROXY_URL", "http://127.0.0.1:8787")
82
 
83
 
 
326
  self._stats = SessionStats()
327
  self._local_store: Any = None # Lazy-initialized CompressionStore
328
  self._compressor_initialized = False
329
+ # File read cache: path → (content_hash, ccr_hash, line_count, token_count)
330
+ self._file_cache: dict[str, tuple[str, str, int, int]] = {}
331
 
332
  if not MCP_AVAILABLE:
333
  raise ImportError("MCP SDK not installed. Install with: pip install mcp")
 
474
 
475
  @self.server.list_tools()
476
  async def list_tools() -> list[Tool]:
477
+ tools = [
478
  Tool(
479
  name=COMPRESS_TOOL_NAME,
480
  description=(
 
538
  ),
539
  ]
540
 
541
+ # Conditionally add headroom_read (behind feature flag)
542
+ if _READ_ENABLED:
543
+ tools.append(
544
+ Tool(
545
+ name=READ_TOOL_NAME,
546
+ description=(
547
+ "Read a file with smart caching. First read returns full content "
548
+ "and caches it. Subsequent reads of the same unchanged file return "
549
+ "a lightweight cache marker (~20 tokens instead of thousands). "
550
+ "Use headroom_retrieve with the hash to get full content if needed. "
551
+ "Use this INSTEAD of the built-in Read tool for significant token savings."
552
+ ),
553
+ inputSchema={
554
+ "type": "object",
555
+ "properties": {
556
+ "file_path": {
557
+ "type": "string",
558
+ "description": "Absolute path to the file to read.",
559
+ },
560
+ "fresh": {
561
+ "type": "boolean",
562
+ "description": (
563
+ "Force a fresh read, bypassing cache. Use after context "
564
+ "compaction, in subagents, or when you need guaranteed "
565
+ "current content."
566
+ ),
567
+ },
568
+ },
569
+ "required": ["file_path"],
570
+ },
571
+ )
572
+ )
573
+
574
+ return tools
575
+
576
  @self.server.call_tool()
577
  async def call_tool(name: str, arguments: dict[str, Any]) -> list[TextContent]:
578
  try:
 
582
  return await self._handle_retrieve(arguments)
583
  elif name == STATS_TOOL_NAME:
584
  return await self._handle_stats()
585
+ elif name == READ_TOOL_NAME and _READ_ENABLED:
586
+ return await self._handle_read(arguments)
587
  else:
588
  return [
589
  TextContent(
 
720
  result["cost_saved_usd"] = cost.get("total_saved", cost.get("saved", 0))
721
  return result if result else None
722
 
723
+ async def _handle_read(self, arguments: dict[str, Any]) -> list[TextContent]:
724
+ """Handle headroom_read tool call — file read with session caching."""
725
+ import hashlib
726
+ from pathlib import Path
727
+
728
+ file_path = arguments.get("file_path", "")
729
+ fresh = arguments.get("fresh", False)
730
+
731
+ if not file_path:
732
+ return [
733
+ TextContent(
734
+ type="text",
735
+ text=json.dumps({"error": "file_path parameter is required"}),
736
+ )
737
+ ]
738
+
739
+ path = Path(file_path).expanduser().resolve()
740
+ if not path.exists():
741
+ return [
742
+ TextContent(
743
+ type="text",
744
+ text=json.dumps({"error": f"File not found: {file_path}"}),
745
+ )
746
+ ]
747
+ if not path.is_file():
748
+ return [
749
+ TextContent(
750
+ type="text",
751
+ text=json.dumps({"error": f"Not a file: {file_path}"}),
752
+ )
753
+ ]
754
+
755
+ # Read file from disk
756
+ try:
757
+ content = path.read_text(errors="replace")
758
+ except Exception as e:
759
+ return [
760
+ TextContent(
761
+ type="text",
762
+ text=json.dumps({"error": f"Cannot read file: {e}"}),
763
+ )
764
+ ]
765
+
766
+ content_hash = hashlib.sha256(content.encode()).hexdigest()[:24]
767
+ line_count = content.count("\n") + (1 if content and not content.endswith("\n") else 0)
768
+ str_path = str(path)
769
+
770
+ # Check cache (unless fresh=true)
771
+ if not fresh and str_path in self._file_cache:
772
+ cached_hash, ccr_hash, cached_lines, cached_tokens = self._file_cache[str_path]
773
+ if cached_hash == content_hash:
774
+ # File unchanged — but is the CCR entry still alive?
775
+ store = self._get_local_store()
776
+ if store.exists(ccr_hash):
777
+ # CCR alive — return cache marker
778
+ self._stats.record_compression(cached_tokens, 5, "read_cache_hit")
779
+ return [
780
+ TextContent(
781
+ type="text",
782
+ text=json.dumps(
783
+ {
784
+ "status": "cached",
785
+ "file": file_path,
786
+ "lines": cached_lines,
787
+ "unchanged": True,
788
+ "hash": ccr_hash,
789
+ "note": (
790
+ f"File unchanged since first read ({cached_lines} lines, "
791
+ f"~{cached_tokens} tokens). Content already in your context "
792
+ f"from the first read. Call headroom_retrieve(hash='{ccr_hash}') "
793
+ f"if you need the full content again."
794
+ ),
795
+ },
796
+ indent=2,
797
+ ),
798
+ )
799
+ ]
800
+ # CCR expired — clear stale cache, fall through to fresh read
801
+ del self._file_cache[str_path]
802
+ # File changed — fall through to fresh read
803
+
804
+ # Fresh read: store in CCR and cache the hash
805
+ store = self._get_local_store()
806
+ ccr_hash = store.store(
807
+ original=content,
808
+ compressed=f"[File: {path.name}, {line_count} lines]",
809
+ original_tokens=len(content.split()),
810
+ compressed_tokens=5,
811
+ tool_name="headroom_read",
812
+ ttl=MCP_SESSION_TTL,
813
+ )
814
+
815
+ token_estimate = len(content.split())
816
+ self._file_cache[str_path] = (content_hash, ccr_hash, line_count, token_estimate)
817
+
818
+ # Return full content with line numbers (like Claude Code's Read tool)
819
+ numbered_lines = []
820
+ for i, line in enumerate(content.split("\n"), 1):
821
+ numbered_lines.append(f"{i:>6}\t{line}")
822
+ numbered_content = "\n".join(numbered_lines)
823
+
824
+ return [
825
+ TextContent(
826
+ type="text",
827
+ text=numbered_content,
828
+ )
829
+ ]
830
+
831
  async def run_stdio(self) -> None:
832
  """Run the server with stdio transport."""
833
  async with stdio_server() as (read_stream, write_stream):