Spaces:
Build error
Build error
Merge pull request #52 from gglucass/fix/learn-with-dots-in-paths
Browse files- headroom/cli/learn.py +12 -0
- headroom/learn/scanner.py +198 -27
- tests/test_learn/test_integration.py +42 -2
- tests/test_learn/test_scanner.py +239 -0
headroom/cli/learn.py
CHANGED
|
@@ -145,11 +145,14 @@ def learn(
|
|
| 145 |
total_projects = 0
|
| 146 |
total_failures = 0
|
| 147 |
total_recommendations = 0
|
|
|
|
|
|
|
| 148 |
|
| 149 |
for agent_name, scanner, writer in agent_configs:
|
| 150 |
all_projects = scanner.discover_projects()
|
| 151 |
if not all_projects:
|
| 152 |
continue
|
|
|
|
| 153 |
|
| 154 |
# Filter to target project(s)
|
| 155 |
if analyze_all:
|
|
@@ -176,6 +179,7 @@ def learn(
|
|
| 176 |
return
|
| 177 |
|
| 178 |
for proj in targets:
|
|
|
|
| 179 |
click.echo(f"\n{'=' * 60}")
|
| 180 |
click.echo(f"[{agent_name}] {proj.name}")
|
| 181 |
click.echo(f"Path: {proj.project_path}")
|
|
@@ -223,6 +227,14 @@ def learn(
|
|
| 223 |
if result.dry_run:
|
| 224 |
click.echo("\n Dry run — use --apply to write.")
|
| 225 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 226 |
# Summary
|
| 227 |
if total_projects > 1:
|
| 228 |
click.echo(f"\n{'=' * 60}")
|
|
|
|
| 145 |
total_projects = 0
|
| 146 |
total_failures = 0
|
| 147 |
total_recommendations = 0
|
| 148 |
+
matched_projects = 0
|
| 149 |
+
available_projects: list[tuple[str, Path]] = []
|
| 150 |
|
| 151 |
for agent_name, scanner, writer in agent_configs:
|
| 152 |
all_projects = scanner.discover_projects()
|
| 153 |
if not all_projects:
|
| 154 |
continue
|
| 155 |
+
available_projects.extend((agent_name, p.project_path) for p in all_projects)
|
| 156 |
|
| 157 |
# Filter to target project(s)
|
| 158 |
if analyze_all:
|
|
|
|
| 179 |
return
|
| 180 |
|
| 181 |
for proj in targets:
|
| 182 |
+
matched_projects += 1
|
| 183 |
click.echo(f"\n{'=' * 60}")
|
| 184 |
click.echo(f"[{agent_name}] {proj.name}")
|
| 185 |
click.echo(f"Path: {proj.project_path}")
|
|
|
|
| 227 |
if result.dry_run:
|
| 228 |
click.echo("\n Dry run — use --apply to write.")
|
| 229 |
|
| 230 |
+
if project and matched_projects == 0:
|
| 231 |
+
click.echo(f"No project data found for {project.resolve()}")
|
| 232 |
+
if available_projects:
|
| 233 |
+
click.echo("\nAvailable discovered projects:")
|
| 234 |
+
for agent_name, project_path in available_projects[:10]:
|
| 235 |
+
click.echo(f" [{agent_name}] {project_path}")
|
| 236 |
+
return
|
| 237 |
+
|
| 238 |
# Summary
|
| 239 |
if total_projects > 1:
|
| 240 |
click.echo(f"\n{'=' * 60}")
|
headroom/learn/scanner.py
CHANGED
|
@@ -151,16 +151,12 @@ class ClaudeCodeScanner(ConversationScanner):
|
|
| 151 |
if not entry.is_dir() or entry.name.startswith("."):
|
| 152 |
continue
|
| 153 |
|
| 154 |
-
# Decode project path from escaped directory name
|
| 155 |
-
#
|
| 156 |
-
|
| 157 |
-
|
| 158 |
-
|
| 159 |
-
|
| 160 |
-
# Heuristic: try the decoded path, if it exists use it
|
| 161 |
-
decoded = _decode_project_path(entry.name)
|
| 162 |
-
if decoded:
|
| 163 |
-
project_path = decoded
|
| 164 |
|
| 165 |
# Derive human-readable name
|
| 166 |
name = project_path.name if project_path != Path("/") else entry.name
|
|
@@ -423,26 +419,71 @@ def _decode_project_path(escaped_name: str) -> Path | None:
|
|
| 423 |
|
| 424 |
|
| 425 |
def _greedy_path_decode(base: Path, parts: list[str]) -> Path | None:
|
| 426 |
-
"""Greedily decode remaining path parts
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 427 |
if not parts:
|
| 428 |
return base if base.exists() else None
|
| 429 |
|
| 430 |
-
|
| 431 |
-
|
| 432 |
-
|
| 433 |
-
|
| 434 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 435 |
|
| 436 |
-
|
| 437 |
-
|
| 438 |
-
|
| 439 |
-
|
| 440 |
-
|
| 441 |
-
if result:
|
| 442 |
-
return result
|
| 443 |
|
| 444 |
-
|
| 445 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 446 |
|
| 447 |
|
| 448 |
# =============================================================================
|
|
@@ -465,6 +506,12 @@ class CodexScanner(ConversationScanner):
|
|
| 465 |
self.codex_dir = codex_dir or Path.home() / ".codex"
|
| 466 |
self.sessions_dir = self.codex_dir / "sessions"
|
| 467 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 468 |
def discover_projects(self) -> list[ProjectInfo]:
|
| 469 |
"""Codex doesn't organize by project — return a single 'codex' project.
|
| 470 |
|
|
@@ -474,7 +521,7 @@ class CodexScanner(ConversationScanner):
|
|
| 474 |
if not self.sessions_dir.exists():
|
| 475 |
return []
|
| 476 |
|
| 477 |
-
session_files =
|
| 478 |
if not session_files:
|
| 479 |
return []
|
| 480 |
|
|
@@ -495,13 +542,19 @@ class CodexScanner(ConversationScanner):
|
|
| 495 |
def scan_project(self, project: ProjectInfo) -> list[SessionData]:
|
| 496 |
"""Scan all Codex session JSON files."""
|
| 497 |
sessions = []
|
| 498 |
-
for json_path in
|
| 499 |
session = self._scan_session(json_path)
|
| 500 |
if session and session.tool_calls:
|
| 501 |
sessions.append(session)
|
| 502 |
return sessions
|
| 503 |
|
| 504 |
def _scan_session(self, json_path: Path) -> SessionData | None:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 505 |
"""Parse a single Codex session file."""
|
| 506 |
try:
|
| 507 |
with open(json_path) as f:
|
|
@@ -592,3 +645,121 @@ class CodexScanner(ConversationScanner):
|
|
| 592 |
)
|
| 593 |
|
| 594 |
return SessionData(session_id=session_id, tool_calls=tool_calls)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 151 |
if not entry.is_dir() or entry.name.startswith("."):
|
| 152 |
continue
|
| 153 |
|
| 154 |
+
# Decode project path from escaped directory name. Fall back to a
|
| 155 |
+
# simple slash replacement for display if we can't recover the real
|
| 156 |
+
# on-disk path from the filesystem.
|
| 157 |
+
project_path = _decode_project_path(entry.name) or Path(
|
| 158 |
+
"/" + entry.name[1:].replace("-", "/")
|
| 159 |
+
)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 160 |
|
| 161 |
# Derive human-readable name
|
| 162 |
name = project_path.name if project_path != Path("/") else entry.name
|
|
|
|
| 419 |
|
| 420 |
|
| 421 |
def _greedy_path_decode(base: Path, parts: list[str]) -> Path | None:
|
| 422 |
+
"""Greedily decode remaining path parts using real child directories.
|
| 423 |
+
|
| 424 |
+
Claude's project directory escaping is ambiguous: both ``/`` and literal
|
| 425 |
+
punctuation such as ``-`` or ``.`` may show up as ``-`` in the escaped
|
| 426 |
+
directory name. Instead of guessing every separator combination, walk the
|
| 427 |
+
actual filesystem and match each child directory against the remaining
|
| 428 |
+
escaped tokens.
|
| 429 |
+
"""
|
| 430 |
if not parts:
|
| 431 |
return base if base.exists() else None
|
| 432 |
|
| 433 |
+
if not base.exists() or not base.is_dir():
|
| 434 |
+
return None
|
| 435 |
+
|
| 436 |
+
try:
|
| 437 |
+
children = sorted(child for child in base.iterdir() if child.is_dir())
|
| 438 |
+
except OSError:
|
| 439 |
+
return None
|
| 440 |
+
|
| 441 |
+
for child in children:
|
| 442 |
+
for tokenization in _component_tokenizations(child.name):
|
| 443 |
+
n_tokens = len(tokenization)
|
| 444 |
+
if parts[:n_tokens] != tokenization:
|
| 445 |
+
continue
|
| 446 |
+
|
| 447 |
+
result = _greedy_path_decode(child, parts[n_tokens:])
|
| 448 |
+
if result:
|
| 449 |
+
return result
|
| 450 |
+
|
| 451 |
+
return None
|
| 452 |
+
|
| 453 |
+
|
| 454 |
+
def _component_tokenizations(component: str) -> list[list[str]]:
|
| 455 |
+
"""Return possible escaped token sequences for a real path component."""
|
| 456 |
+
tokenizations: list[list[str]] = []
|
| 457 |
+
seen: set[tuple[str, ...]] = set()
|
| 458 |
|
| 459 |
+
def add(tokens: list[str]) -> None:
|
| 460 |
+
key = tuple(tokens)
|
| 461 |
+
if tokens and key not in seen:
|
| 462 |
+
seen.add(key)
|
| 463 |
+
tokenizations.append(tokens)
|
|
|
|
|
|
|
| 464 |
|
| 465 |
+
add([component])
|
| 466 |
+
|
| 467 |
+
for separator in ("-", ".", None):
|
| 468 |
+
if separator is None:
|
| 469 |
+
tokens = [token for token in re.split(r"[-.]", component) if token]
|
| 470 |
+
else:
|
| 471 |
+
tokens = [token for token in component.split(separator) if token]
|
| 472 |
+
add(tokens)
|
| 473 |
+
|
| 474 |
+
# Claude's flattened encoding can turn a leading "." in hidden directory
|
| 475 |
+
# names into an empty token followed by the remaining component tokens.
|
| 476 |
+
if component.startswith(".") and len(component) > 1:
|
| 477 |
+
hidden_component = component[1:]
|
| 478 |
+
add(["", hidden_component])
|
| 479 |
+
for separator in ("-", ".", None):
|
| 480 |
+
if separator is None:
|
| 481 |
+
tokens = [token for token in re.split(r"[-.]", hidden_component) if token]
|
| 482 |
+
else:
|
| 483 |
+
tokens = [token for token in hidden_component.split(separator) if token]
|
| 484 |
+
add(["", *tokens])
|
| 485 |
+
|
| 486 |
+
return tokenizations
|
| 487 |
|
| 488 |
|
| 489 |
# =============================================================================
|
|
|
|
| 506 |
self.codex_dir = codex_dir or Path.home() / ".codex"
|
| 507 |
self.sessions_dir = self.codex_dir / "sessions"
|
| 508 |
|
| 509 |
+
def _iter_session_files(self, root: Path | None = None) -> list[Path]:
|
| 510 |
+
"""Return all known Codex session files, including nested rollouts."""
|
| 511 |
+
search_root = root or self.sessions_dir
|
| 512 |
+
session_files = list(search_root.rglob("*.json")) + list(search_root.rglob("*.jsonl"))
|
| 513 |
+
return sorted(path for path in session_files if path.is_file())
|
| 514 |
+
|
| 515 |
def discover_projects(self) -> list[ProjectInfo]:
|
| 516 |
"""Codex doesn't organize by project — return a single 'codex' project.
|
| 517 |
|
|
|
|
| 521 |
if not self.sessions_dir.exists():
|
| 522 |
return []
|
| 523 |
|
| 524 |
+
session_files = self._iter_session_files()
|
| 525 |
if not session_files:
|
| 526 |
return []
|
| 527 |
|
|
|
|
| 542 |
def scan_project(self, project: ProjectInfo) -> list[SessionData]:
|
| 543 |
"""Scan all Codex session JSON files."""
|
| 544 |
sessions = []
|
| 545 |
+
for json_path in self._iter_session_files(project.data_path):
|
| 546 |
session = self._scan_session(json_path)
|
| 547 |
if session and session.tool_calls:
|
| 548 |
sessions.append(session)
|
| 549 |
return sessions
|
| 550 |
|
| 551 |
def _scan_session(self, json_path: Path) -> SessionData | None:
|
| 552 |
+
"""Parse a single Codex session file."""
|
| 553 |
+
if json_path.suffix == ".jsonl":
|
| 554 |
+
return self._scan_jsonl_session(json_path)
|
| 555 |
+
return self._scan_json_session(json_path)
|
| 556 |
+
|
| 557 |
+
def _scan_json_session(self, json_path: Path) -> SessionData | None:
|
| 558 |
"""Parse a single Codex session file."""
|
| 559 |
try:
|
| 560 |
with open(json_path) as f:
|
|
|
|
| 645 |
)
|
| 646 |
|
| 647 |
return SessionData(session_id=session_id, tool_calls=tool_calls)
|
| 648 |
+
|
| 649 |
+
def _scan_jsonl_session(self, jsonl_path: Path) -> SessionData | None:
|
| 650 |
+
"""Parse a modern Codex rollout session stored as JSONL."""
|
| 651 |
+
session_id = jsonl_path.stem
|
| 652 |
+
func_calls: dict[str, tuple[str, dict]] = {}
|
| 653 |
+
tool_calls: list[ToolCall] = []
|
| 654 |
+
msg_index = 0
|
| 655 |
+
|
| 656 |
+
try:
|
| 657 |
+
with open(jsonl_path) as f:
|
| 658 |
+
for line in f:
|
| 659 |
+
try:
|
| 660 |
+
entry = json.loads(line)
|
| 661 |
+
except json.JSONDecodeError:
|
| 662 |
+
continue
|
| 663 |
+
|
| 664 |
+
if entry.get("type") == "session_meta":
|
| 665 |
+
payload = entry.get("payload", {})
|
| 666 |
+
if isinstance(payload, dict):
|
| 667 |
+
session_id = payload.get("id", session_id)
|
| 668 |
+
continue
|
| 669 |
+
|
| 670 |
+
if entry.get("type") != "response_item":
|
| 671 |
+
continue
|
| 672 |
+
|
| 673 |
+
payload = entry.get("payload", {})
|
| 674 |
+
if not isinstance(payload, dict):
|
| 675 |
+
continue
|
| 676 |
+
|
| 677 |
+
msg_index += 1
|
| 678 |
+
item_type = payload.get("type", "")
|
| 679 |
+
|
| 680 |
+
if item_type in ("function_call", "custom_tool_call"):
|
| 681 |
+
call_id = payload.get("call_id", "")
|
| 682 |
+
name = payload.get("name", "")
|
| 683 |
+
parsed = self._parse_codex_arguments(payload)
|
| 684 |
+
name, parsed = self._normalize_codex_tool(name, parsed)
|
| 685 |
+
if call_id and name:
|
| 686 |
+
func_calls[call_id] = (name, parsed)
|
| 687 |
+
continue
|
| 688 |
+
|
| 689 |
+
if item_type not in ("function_call_output", "custom_tool_call_output"):
|
| 690 |
+
continue
|
| 691 |
+
|
| 692 |
+
call_id = payload.get("call_id", "")
|
| 693 |
+
if call_id not in func_calls:
|
| 694 |
+
continue
|
| 695 |
+
|
| 696 |
+
name, inp = func_calls[call_id]
|
| 697 |
+
result_content = self._parse_codex_output(payload.get("output", ""))
|
| 698 |
+
is_err = is_error_content(result_content)
|
| 699 |
+
error_cat = classify_error(result_content) if is_err else ErrorCategory.UNKNOWN
|
| 700 |
+
|
| 701 |
+
tool_calls.append(
|
| 702 |
+
ToolCall(
|
| 703 |
+
name=name,
|
| 704 |
+
tool_call_id=call_id,
|
| 705 |
+
input_data=inp,
|
| 706 |
+
output=result_content,
|
| 707 |
+
is_error=is_err,
|
| 708 |
+
error_category=error_cat,
|
| 709 |
+
msg_index=msg_index,
|
| 710 |
+
output_bytes=len(result_content.encode("utf-8")),
|
| 711 |
+
)
|
| 712 |
+
)
|
| 713 |
+
|
| 714 |
+
except OSError as e:
|
| 715 |
+
logger.debug("Failed to read Codex session %s: %s", jsonl_path, e)
|
| 716 |
+
return None
|
| 717 |
+
|
| 718 |
+
if not tool_calls:
|
| 719 |
+
return None
|
| 720 |
+
|
| 721 |
+
return SessionData(session_id=session_id, tool_calls=tool_calls)
|
| 722 |
+
|
| 723 |
+
def _parse_codex_arguments(self, payload: dict) -> dict:
|
| 724 |
+
"""Parse arguments for either legacy or rollout Codex tool calls."""
|
| 725 |
+
raw_args = payload.get("arguments", payload.get("input", ""))
|
| 726 |
+
if isinstance(raw_args, str):
|
| 727 |
+
try:
|
| 728 |
+
parsed = json.loads(raw_args)
|
| 729 |
+
return parsed if isinstance(parsed, dict) else {"raw": raw_args}
|
| 730 |
+
except (json.JSONDecodeError, TypeError):
|
| 731 |
+
return {"raw": raw_args}
|
| 732 |
+
if isinstance(raw_args, dict):
|
| 733 |
+
return raw_args
|
| 734 |
+
return {"raw": str(raw_args)}
|
| 735 |
+
|
| 736 |
+
def _normalize_codex_tool(self, name: str, parsed: dict) -> tuple[str, dict]:
|
| 737 |
+
"""Normalize modern Codex tool names to the cross-agent schema."""
|
| 738 |
+
if name == "shell" and "command" in parsed:
|
| 739 |
+
cmd = parsed["command"]
|
| 740 |
+
if isinstance(cmd, list):
|
| 741 |
+
parsed["command"] = cmd[-1] if cmd else ""
|
| 742 |
+
return "Bash", parsed
|
| 743 |
+
|
| 744 |
+
if name == "exec_command" and "cmd" in parsed:
|
| 745 |
+
parsed = dict(parsed)
|
| 746 |
+
parsed["command"] = parsed.get("cmd", "")
|
| 747 |
+
return "Bash", parsed
|
| 748 |
+
|
| 749 |
+
return name, parsed
|
| 750 |
+
|
| 751 |
+
def _parse_codex_output(self, output_raw: object) -> str:
|
| 752 |
+
"""Parse tool output from Codex rollout records."""
|
| 753 |
+
if isinstance(output_raw, str):
|
| 754 |
+
try:
|
| 755 |
+
parsed_out = json.loads(output_raw)
|
| 756 |
+
except (json.JSONDecodeError, TypeError):
|
| 757 |
+
return output_raw
|
| 758 |
+
|
| 759 |
+
if isinstance(parsed_out, dict):
|
| 760 |
+
if "output" in parsed_out:
|
| 761 |
+
return str(parsed_out["output"])
|
| 762 |
+
return json.dumps(parsed_out)
|
| 763 |
+
return output_raw
|
| 764 |
+
|
| 765 |
+
return str(output_raw)
|
tests/test_learn/test_integration.py
CHANGED
|
@@ -26,6 +26,7 @@ from headroom.learn.models import (
|
|
| 26 |
Recommendation,
|
| 27 |
RecommendationTarget,
|
| 28 |
)
|
|
|
|
| 29 |
from headroom.learn.writer import ClaudeCodeWriter, CodexWriter
|
| 30 |
|
| 31 |
# =============================================================================
|
|
@@ -122,6 +123,9 @@ class TestFalsePositiveFiltering:
|
|
| 122 |
CLAUDE_DIR = Path.home() / ".claude" / "projects"
|
| 123 |
CODEX_DIR = Path.home() / ".codex" / "sessions"
|
| 124 |
HAS_API_KEY = bool(os.environ.get("ANTHROPIC_API_KEY"))
|
|
|
|
|
|
|
|
|
|
| 125 |
|
| 126 |
|
| 127 |
@pytest.mark.skipif(not CLAUDE_DIR.exists(), reason="No Claude Code data")
|
|
@@ -192,7 +196,7 @@ class TestClaudeCodeIntegration:
|
|
| 192 |
}
|
| 193 |
with patch("headroom.learn.analyzer._call_llm", return_value=mock_response):
|
| 194 |
sessions = scanner.scan_project(best)
|
| 195 |
-
result = SessionAnalyzer().analyze(best, sessions)
|
| 196 |
recs = result.recommendations
|
| 197 |
|
| 198 |
writer = ClaudeCodeWriter()
|
|
@@ -203,7 +207,43 @@ class TestClaudeCodeIntegration:
|
|
| 203 |
assert "CLAUDE.md" in fp.name or "MEMORY.md" in fp.name
|
| 204 |
|
| 205 |
|
| 206 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 207 |
class TestCodexIntegration:
|
| 208 |
"""Integration tests against real Codex session data."""
|
| 209 |
|
|
|
|
| 26 |
Recommendation,
|
| 27 |
RecommendationTarget,
|
| 28 |
)
|
| 29 |
+
from headroom.learn.scanner import _greedy_path_decode
|
| 30 |
from headroom.learn.writer import ClaudeCodeWriter, CodexWriter
|
| 31 |
|
| 32 |
# =============================================================================
|
|
|
|
| 123 |
CLAUDE_DIR = Path.home() / ".claude" / "projects"
|
| 124 |
CODEX_DIR = Path.home() / ".codex" / "sessions"
|
| 125 |
HAS_API_KEY = bool(os.environ.get("ANTHROPIC_API_KEY"))
|
| 126 |
+
HAS_CODEX_DATA = CODEX_DIR.exists() and (
|
| 127 |
+
any(CODEX_DIR.rglob("*.json")) or any(CODEX_DIR.rglob("*.jsonl"))
|
| 128 |
+
)
|
| 129 |
|
| 130 |
|
| 131 |
@pytest.mark.skipif(not CLAUDE_DIR.exists(), reason="No Claude Code data")
|
|
|
|
| 196 |
}
|
| 197 |
with patch("headroom.learn.analyzer._call_llm", return_value=mock_response):
|
| 198 |
sessions = scanner.scan_project(best)
|
| 199 |
+
result = SessionAnalyzer(model="gpt-4o").analyze(best, sessions)
|
| 200 |
recs = result.recommendations
|
| 201 |
|
| 202 |
writer = ClaudeCodeWriter()
|
|
|
|
| 207 |
assert "CLAUDE.md" in fp.name or "MEMORY.md" in fp.name
|
| 208 |
|
| 209 |
|
| 210 |
+
class TestDecodeProjectPath:
|
| 211 |
+
"""Unit tests for _greedy_path_decode — covers dot-in-path bug (GitHub.nosync)."""
|
| 212 |
+
|
| 213 |
+
def test_dot_in_directory_name(self, tmp_path):
|
| 214 |
+
"""Paths with dots (e.g. GitHub.nosync) must decode correctly.
|
| 215 |
+
|
| 216 |
+
Claude Code encodes '/' and '.' both as '-', so 'GitHub.nosync'
|
| 217 |
+
becomes 'GitHub-nosync' in the directory name.
|
| 218 |
+
_greedy_path_decode must reconstruct it by trying '.' as a join.
|
| 219 |
+
"""
|
| 220 |
+
# base = tmp_path, remaining parts = ["GitHub", "nosync", "myproject"]
|
| 221 |
+
# which came from encoding "GitHub.nosync/myproject" as "GitHub-nosync-myproject"
|
| 222 |
+
(tmp_path / "GitHub.nosync" / "myproject").mkdir(parents=True)
|
| 223 |
+
result = _greedy_path_decode(tmp_path, ["GitHub", "nosync", "myproject"])
|
| 224 |
+
assert result == tmp_path / "GitHub.nosync" / "myproject"
|
| 225 |
+
|
| 226 |
+
def test_hyphen_in_directory_name(self, tmp_path):
|
| 227 |
+
"""Paths with literal hyphens decode correctly (existing behavior preserved)."""
|
| 228 |
+
(tmp_path / "my-project").mkdir()
|
| 229 |
+
result = _greedy_path_decode(tmp_path, ["my", "project"])
|
| 230 |
+
assert result == tmp_path / "my-project"
|
| 231 |
+
|
| 232 |
+
def test_simple_path_no_ambiguity(self, tmp_path):
|
| 233 |
+
"""Plain paths with no special chars still decode correctly."""
|
| 234 |
+
(tmp_path / "myproject").mkdir()
|
| 235 |
+
result = _greedy_path_decode(tmp_path, ["myproject"])
|
| 236 |
+
assert result == tmp_path / "myproject"
|
| 237 |
+
|
| 238 |
+
def test_dot_preferred_over_slash_when_slash_missing(self, tmp_path):
|
| 239 |
+
"""When GitHub/nosync doesn't exist but GitHub.nosync does, use dot."""
|
| 240 |
+
# Only create the dot version, not the slash version
|
| 241 |
+
(tmp_path / "GitHub.nosync").mkdir()
|
| 242 |
+
result = _greedy_path_decode(tmp_path, ["GitHub", "nosync"])
|
| 243 |
+
assert result == tmp_path / "GitHub.nosync"
|
| 244 |
+
|
| 245 |
+
|
| 246 |
+
@pytest.mark.skipif(not HAS_CODEX_DATA, reason="No Codex data")
|
| 247 |
class TestCodexIntegration:
|
| 248 |
"""Integration tests against real Codex session data."""
|
| 249 |
|
tests/test_learn/test_scanner.py
ADDED
|
@@ -0,0 +1,239 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Tests for _decode_project_path and _greedy_path_decode (issue #47).
|
| 2 |
+
|
| 3 |
+
Directory names that contain dots (e.g. ``GitHub.nosync``) or multiple
|
| 4 |
+
hyphens (e.g. ``my-cool-project``) were silently dropped because
|
| 5 |
+
_greedy_path_decode only tried joining two consecutive tokens with a hyphen,
|
| 6 |
+
making it impossible to reconstruct names formed from three or more tokens.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
from pathlib import Path
|
| 12 |
+
from uuid import uuid4
|
| 13 |
+
|
| 14 |
+
import pytest
|
| 15 |
+
|
| 16 |
+
from headroom.learn.scanner import _decode_project_path, _greedy_path_decode
|
| 17 |
+
|
| 18 |
+
# ---------------------------------------------------------------------------
|
| 19 |
+
# Helpers
|
| 20 |
+
# ---------------------------------------------------------------------------
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
def _make_dirs(base: Path, *rel_paths: str) -> None:
|
| 24 |
+
"""Create one or more relative directory paths under *base*."""
|
| 25 |
+
for rel in rel_paths:
|
| 26 |
+
(base / rel).mkdir(parents=True, exist_ok=True)
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
# ---------------------------------------------------------------------------
|
| 30 |
+
# _greedy_path_decode
|
| 31 |
+
# ---------------------------------------------------------------------------
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
class TestGreedyPathDecode:
|
| 35 |
+
"""Unit tests for _greedy_path_decode."""
|
| 36 |
+
|
| 37 |
+
def test_simple_directory(self, tmp_path: Path) -> None:
|
| 38 |
+
_make_dirs(tmp_path, "headroom")
|
| 39 |
+
result = _greedy_path_decode(tmp_path, ["headroom"])
|
| 40 |
+
assert result == tmp_path / "headroom"
|
| 41 |
+
|
| 42 |
+
def test_single_hyphen_in_dirname(self, tmp_path: Path) -> None:
|
| 43 |
+
"""Directory name contains one literal hyphen."""
|
| 44 |
+
_make_dirs(tmp_path, "my-project")
|
| 45 |
+
result = _greedy_path_decode(tmp_path, ["my", "project"])
|
| 46 |
+
assert result == tmp_path / "my-project"
|
| 47 |
+
|
| 48 |
+
def test_multiple_hyphens_in_dirname(self, tmp_path: Path) -> None:
|
| 49 |
+
"""Directory name contains multiple literal hyphens (the regression case)."""
|
| 50 |
+
_make_dirs(tmp_path, "my-cool-project")
|
| 51 |
+
result = _greedy_path_decode(tmp_path, ["my", "cool", "project"])
|
| 52 |
+
assert result == tmp_path / "my-cool-project"
|
| 53 |
+
|
| 54 |
+
def test_dot_only_in_dirname(self, tmp_path: Path) -> None:
|
| 55 |
+
"""Directory name contains a dot but no hyphen (e.g. GitHub.nosync)."""
|
| 56 |
+
_make_dirs(tmp_path, "GitHub.nosync")
|
| 57 |
+
result = _greedy_path_decode(tmp_path, ["GitHub.nosync"])
|
| 58 |
+
assert result == tmp_path / "GitHub.nosync"
|
| 59 |
+
|
| 60 |
+
def test_dot_and_single_hyphen_in_dirname(self, tmp_path: Path) -> None:
|
| 61 |
+
"""Directory name has both a dot and a single hyphen (e.g. my-project.nosync)."""
|
| 62 |
+
_make_dirs(tmp_path, "my-project.nosync")
|
| 63 |
+
result = _greedy_path_decode(tmp_path, ["my", "project.nosync"])
|
| 64 |
+
assert result == tmp_path / "my-project.nosync"
|
| 65 |
+
|
| 66 |
+
def test_dot_and_multiple_hyphens_in_dirname(self, tmp_path: Path) -> None:
|
| 67 |
+
"""Directory name has a dot and multiple hyphens (e.g. my-cool-project.nosync).
|
| 68 |
+
|
| 69 |
+
This was the primary regression: the old code only joined pairs, so it
|
| 70 |
+
could never reconstruct a three-token hyphenated name.
|
| 71 |
+
"""
|
| 72 |
+
_make_dirs(tmp_path, "my-cool-project.nosync")
|
| 73 |
+
result = _greedy_path_decode(tmp_path, ["my", "cool", "project.nosync"])
|
| 74 |
+
assert result == tmp_path / "my-cool-project.nosync"
|
| 75 |
+
|
| 76 |
+
def test_dot_dir_containing_hyphenated_subdir(self, tmp_path: Path) -> None:
|
| 77 |
+
"""Path like GitHub.nosync/my-project — dot parent + hyphen child."""
|
| 78 |
+
_make_dirs(tmp_path, "GitHub.nosync/my-project")
|
| 79 |
+
result = _greedy_path_decode(tmp_path, ["GitHub.nosync", "my", "project"])
|
| 80 |
+
assert result == tmp_path / "GitHub.nosync" / "my-project"
|
| 81 |
+
|
| 82 |
+
def test_dot_dir_with_multi_hyphen_subdir(self, tmp_path: Path) -> None:
|
| 83 |
+
"""Path like GitHub.nosync/my-cool-app — dot parent + multi-hyphen child."""
|
| 84 |
+
_make_dirs(tmp_path, "GitHub.nosync/my-cool-app")
|
| 85 |
+
result = _greedy_path_decode(tmp_path, ["GitHub.nosync", "my", "cool", "app"])
|
| 86 |
+
assert result == tmp_path / "GitHub.nosync" / "my-cool-app"
|
| 87 |
+
|
| 88 |
+
def test_multi_hyphen_dot_dir_containing_subproject(self, tmp_path: Path) -> None:
|
| 89 |
+
"""Path like my-cool-project.nosync/headroom — hardest combination."""
|
| 90 |
+
_make_dirs(tmp_path, "my-cool-project.nosync/headroom")
|
| 91 |
+
result = _greedy_path_decode(
|
| 92 |
+
tmp_path, ["my", "cool", "project.nosync", "headroom"]
|
| 93 |
+
)
|
| 94 |
+
assert result == tmp_path / "my-cool-project.nosync" / "headroom"
|
| 95 |
+
|
| 96 |
+
def test_dot_flattened_into_separate_tokens(self, tmp_path: Path) -> None:
|
| 97 |
+
"""Flattened encoding like GitHub-nosync should map back to GitHub.nosync."""
|
| 98 |
+
_make_dirs(tmp_path, "GitHub.nosync/thebest")
|
| 99 |
+
result = _greedy_path_decode(tmp_path, ["GitHub", "nosync", "thebest"])
|
| 100 |
+
assert result == tmp_path / "GitHub.nosync" / "thebest"
|
| 101 |
+
|
| 102 |
+
def test_hybrid_hyphen_and_dot_flattening(self, tmp_path: Path) -> None:
|
| 103 |
+
"""Flattened encoding should reconstruct mixed separators in one component."""
|
| 104 |
+
_make_dirs(tmp_path, "my-cool-project.nosync/headroom")
|
| 105 |
+
result = _greedy_path_decode(tmp_path, ["my", "cool", "project", "nosync", "headroom"])
|
| 106 |
+
assert result == tmp_path / "my-cool-project.nosync" / "headroom"
|
| 107 |
+
|
| 108 |
+
def test_nonexistent_path_returns_none(self, tmp_path: Path) -> None:
|
| 109 |
+
result = _greedy_path_decode(tmp_path, ["does", "not", "exist"])
|
| 110 |
+
assert result is None
|
| 111 |
+
|
| 112 |
+
def test_empty_parts_returns_base_when_exists(self, tmp_path: Path) -> None:
|
| 113 |
+
result = _greedy_path_decode(tmp_path, [])
|
| 114 |
+
assert result == tmp_path
|
| 115 |
+
|
| 116 |
+
def test_empty_parts_returns_none_when_not_exists(self) -> None:
|
| 117 |
+
result = _greedy_path_decode(Path("/nonexistent/path"), [])
|
| 118 |
+
assert result is None
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
# ---------------------------------------------------------------------------
|
| 122 |
+
# _decode_project_path
|
| 123 |
+
# ---------------------------------------------------------------------------
|
| 124 |
+
|
| 125 |
+
|
| 126 |
+
class TestDecodeProjectPath:
|
| 127 |
+
"""Integration-level tests for _decode_project_path.
|
| 128 |
+
|
| 129 |
+
Note: _decode_project_path's greedy branch only activates for paths whose
|
| 130 |
+
first component is ``Users`` (the common macOS home prefix). Tests that
|
| 131 |
+
exercise the greedy decoder therefore synthesise an encoded name rooted at
|
| 132 |
+
``/Users/<username>/…`` inside a real temporary directory created under
|
| 133 |
+
that prefix. When the temp directory does not exist under ``/Users`` the
|
| 134 |
+
tests fall back to ``/tmp`` and rely only on the fast simple-replace path.
|
| 135 |
+
"""
|
| 136 |
+
|
| 137 |
+
def test_returns_none_for_non_absolute_encoded_name(self) -> None:
|
| 138 |
+
assert _decode_project_path("Users-foo-bar") is None
|
| 139 |
+
|
| 140 |
+
def test_simple_replace_finds_dot_path(self, users_tmp: Path) -> None:
|
| 141 |
+
"""Simple replace-all works when no dir names contain hyphens.
|
| 142 |
+
|
| 143 |
+
The encoded name maps directly to the real path because every ``-`` is
|
| 144 |
+
a path separator; dots in directory names are preserved unchanged.
|
| 145 |
+
"""
|
| 146 |
+
project = users_tmp / "GitHub.nosync" / "headroom"
|
| 147 |
+
project.mkdir(parents=True)
|
| 148 |
+
# Build the encoded name exactly as Claude Code does (/ → -)
|
| 149 |
+
encoded = "-" + str(project)[1:].replace("/", "-")
|
| 150 |
+
result = _decode_project_path(encoded)
|
| 151 |
+
if str(users_tmp).startswith("/Users/"):
|
| 152 |
+
assert result == project
|
| 153 |
+
else:
|
| 154 |
+
assert result is None or result == project
|
| 155 |
+
|
| 156 |
+
# ------------------------------------------------------------------
|
| 157 |
+
# Greedy-decoder tests — require a /Users-rooted path to activate.
|
| 158 |
+
# We try to create a temp dir under the real /Users tree; if that is
|
| 159 |
+
# not writable we skip rather than fail (CI typically runs as a real
|
| 160 |
+
# macOS user whose home IS under /Users).
|
| 161 |
+
# ------------------------------------------------------------------
|
| 162 |
+
|
| 163 |
+
@pytest.fixture()
|
| 164 |
+
def users_tmp(self, tmp_path: Path) -> Path:
|
| 165 |
+
"""Return a temporary directory whose path starts with /Users/…
|
| 166 |
+
|
| 167 |
+
On macOS the system temp dir is under /private/var, so we create a
|
| 168 |
+
disposable directory directly inside the real user's home instead.
|
| 169 |
+
Falls back to tmp_path so tests still run on non-macOS platforms
|
| 170 |
+
(where the greedy branch isn't reached but no crash occurs either).
|
| 171 |
+
"""
|
| 172 |
+
|
| 173 |
+
home = Path.home()
|
| 174 |
+
if str(home).startswith("/Users/"):
|
| 175 |
+
base = home / ".pytest_headroom_tmp"
|
| 176 |
+
try:
|
| 177 |
+
base.mkdir(exist_ok=True)
|
| 178 |
+
except PermissionError:
|
| 179 |
+
pytest.skip("Cannot create /Users-rooted temp dir in this environment")
|
| 180 |
+
# Use a sub-directory unique to this test invocation
|
| 181 |
+
unique = base / uuid4().hex
|
| 182 |
+
try:
|
| 183 |
+
unique.mkdir()
|
| 184 |
+
except PermissionError:
|
| 185 |
+
pytest.skip("Cannot create /Users-rooted temp dir in this environment")
|
| 186 |
+
yield unique
|
| 187 |
+
import shutil
|
| 188 |
+
|
| 189 |
+
shutil.rmtree(unique, ignore_errors=True)
|
| 190 |
+
else:
|
| 191 |
+
yield tmp_path
|
| 192 |
+
|
| 193 |
+
def test_dot_and_hyphen_in_dirname_via_greedy(self, users_tmp: Path) -> None:
|
| 194 |
+
"""GitHub.nosync/my-project — dot parent + hyphenated child (issue #47).
|
| 195 |
+
|
| 196 |
+
Simple replace-all gives ``…/GitHub.nosync/my/project`` which does not
|
| 197 |
+
exist, so the greedy decoder must reconstruct ``my-project``.
|
| 198 |
+
"""
|
| 199 |
+
project = users_tmp / "GitHub.nosync" / "my-project"
|
| 200 |
+
project.mkdir(parents=True)
|
| 201 |
+
|
| 202 |
+
encoded = "-" + str(project)[1:].replace("/", "-")
|
| 203 |
+
result = _decode_project_path(encoded)
|
| 204 |
+
|
| 205 |
+
if str(users_tmp).startswith("/Users/"):
|
| 206 |
+
assert result == project
|
| 207 |
+
else:
|
| 208 |
+
# Greedy branch not reached outside /Users; just confirm no crash
|
| 209 |
+
assert result is None or result == project
|
| 210 |
+
|
| 211 |
+
def test_multi_hyphen_dot_dirname_via_greedy(self, users_tmp: Path) -> None:
|
| 212 |
+
"""my-cool-project.nosync/app — primary regression from issue #47.
|
| 213 |
+
|
| 214 |
+
Three tokens joined by hyphens form the parent dir name; the old code
|
| 215 |
+
only tried pairs and therefore could never reconstruct this component.
|
| 216 |
+
"""
|
| 217 |
+
project = users_tmp / "my-cool-project.nosync" / "app"
|
| 218 |
+
project.mkdir(parents=True)
|
| 219 |
+
|
| 220 |
+
encoded = "-" + str(project)[1:].replace("/", "-")
|
| 221 |
+
result = _decode_project_path(encoded)
|
| 222 |
+
|
| 223 |
+
if str(users_tmp).startswith("/Users/"):
|
| 224 |
+
assert result == project
|
| 225 |
+
else:
|
| 226 |
+
assert result is None or result == project
|
| 227 |
+
|
| 228 |
+
def test_flattened_dot_dirname_via_greedy(self, users_tmp: Path) -> None:
|
| 229 |
+
"""GitHub.nosync/thebest should decode from GitHub-nosync-thebest."""
|
| 230 |
+
project = users_tmp / "GitHub.nosync" / "thebest"
|
| 231 |
+
project.mkdir(parents=True)
|
| 232 |
+
|
| 233 |
+
encoded = "-" + str(project)[1:].replace("/", "-").replace(".", "-")
|
| 234 |
+
result = _decode_project_path(encoded)
|
| 235 |
+
|
| 236 |
+
if str(users_tmp).startswith("/Users/"):
|
| 237 |
+
assert result == project
|
| 238 |
+
else:
|
| 239 |
+
assert result is None or result == project
|