Przeglądaj źródła

Initial commit — Wobkey RGB CLI with LICENSE (GPLv3)

- Pure-Go HID layer (/dev/hidraw* + /sys/class/hidraw uevent parsing)
- Keyboard discovery (VID/PID matching from keyboards.json)
- 8 RGB subcommands: effect, brightness, speed, color, mode, enable, disable, info
- Proper RGB to HSV conversion for color command
- QMK rgblight param constants (Hue=26, Sat=27, Effect=48, etc.)
- Full test suite: device, hid, rgb packages
- Fix: HID scanner off-by-one (HID_ID=0003: :12 not :11)
- Fix: /sys/bus/hid/devices/ entries are symlinks, not dirs
Paul Klumpp 2 tygodni temu
commit
b0d6e24f95

+ 3 - 0
.claude/CLAUDE.md

@@ -0,0 +1,3 @@
+# graphify
+- **graphify** (`.claude/skills/graphify/SKILL.md`) - any input to knowledge graph. Trigger: `/graphify`
+When the user types `/graphify`, use the installed graphify skill or instructions before doing anything else.

+ 26 - 0
.claude/settings.json

@@ -0,0 +1,26 @@
+{
+  "hooks": {
+    "PreToolUse": [
+      {
+        "matcher": "Bash|Grep",
+        "hooks": [
+          {
+            "type": "command",
+            "command": "graphify hook-guard search",
+            "timeout": 10
+          }
+        ]
+      },
+      {
+        "matcher": "Read|Glob",
+        "hooks": [
+          {
+            "type": "command",
+            "command": "graphify hook-guard read",
+            "timeout": 10
+          }
+        ]
+      }
+    ]
+  }
+}

+ 1 - 0
.claude/skills/graphify/.graphify_version

@@ -0,0 +1 @@
+0.9.67

+ 713 - 0
.claude/skills/graphify/SKILL.md

@@ -0,0 +1,713 @@
+---
+name: graphify
+description: "Use for any question about a codebase, its architecture, file relationships, or project content — especially when graphify-out/ exists, where the question should be treated as a graphify query first. Turns any input (code, docs, papers, images, videos) into a persistent knowledge graph with god nodes, community detection, and query/path/explain tools."
+---
+
+# /graphify
+
+Turn any folder of files into a navigable knowledge graph with community detection, an honest audit trail, and three outputs: interactive HTML, GraphRAG-ready JSON, and a plain-language GRAPH_REPORT.md.
+
+## Usage
+
+```
+/graphify                                             # full pipeline on current directory (HTML viz; add --obsidian for a vault)
+/graphify <path>                                      # full pipeline on specific path
+/graphify https://github.com/<owner>/<repo>           # clone repo then run full pipeline on it
+/graphify https://github.com/<owner>/<repo> --branch <branch>  # clone a specific branch
+/graphify <url1> <url2> ...                           # clone multiple repos, build each, merge into one cross-repo graph
+/graphify <path> --mode deep                          # thorough extraction, richer INFERRED edges
+/graphify <path> --update                             # incremental - re-extract only new/changed files
+/graphify <path> --directed                            # build directed graph (preserves edge direction: source→target)
+/graphify <path> --whisper-model medium                # use a larger Whisper model for better transcription accuracy
+/graphify <path> --cluster-only                       # rerun clustering on existing graph
+/graphify <path> --no-viz                             # skip visualization, just report + JSON
+/graphify <path> --html                               # (HTML is generated by default - this flag is a no-op)
+/graphify <path> --svg                                # also export graph.svg (embeds in Notion, GitHub)
+/graphify <path> --graphml                            # export graph.graphml (Gephi, yEd)
+/graphify <path> --neo4j                              # generate graphify-out/cypher.txt for Neo4j
+/graphify <path> --neo4j-push bolt://localhost:7687   # push directly to Neo4j
+/graphify <path> --falkordb                           # generate graphify-out/cypher.txt for FalkorDB
+/graphify <path> --falkordb-push falkordb://localhost:6379   # push directly to FalkorDB
+/graphify <path> --mcp                                # start MCP stdio server for agent access
+/graphify <path> --watch                              # watch folder, auto-rebuild on code changes (no LLM needed)
+/graphify <path> --wiki                               # build agent-crawlable wiki (index.md + one article per community)
+/graphify <path> --obsidian --obsidian-dir ~/vaults/my-project  # write vault to custom path (e.g. existing vault)
+/graphify add <url>                                   # fetch URL, save to ./raw, update graph
+/graphify add <url> --author "Name"                   # tag who wrote it
+/graphify add <url> --contributor "Name"              # tag who added it to the corpus
+/graphify query "<question>"                          # BFS traversal - broad context
+/graphify query "<question>" --dfs                    # DFS - trace a specific path
+/graphify query "<question>" --budget 1500            # cap answer at N tokens
+/graphify path "AuthModule" "Database"                # shortest path between two concepts
+/graphify explain "SwinTransformer"                   # plain-language explanation of a node
+```
+
+## What graphify is for
+
+Drop any folder of code, docs, papers, images, or video into graphify and get a queryable knowledge graph. Persistent across sessions, honest audit trail (EXTRACTED/INFERRED/AMBIGUOUS), community detection surfaces cross-document connections you wouldn't think to ask about.
+
+## What You Must Do When Invoked
+
+If the user invoked `/graphify --help` or `/graphify -h` (with no other arguments), print the contents of the `## Usage` section above verbatim and stop. Do not run any commands, do not detect files, do not default the path to `.`. Just print the Usage block and return.
+
+**Fast path — existing graph:** Before doing anything else, check whether `graphify-out/graph.json` exists. The expected location is `graphify-out/graph.json` relative to the **current working directory** (i.e. the project root where you are running commands). If it exists AND the user's request is a natural-language question about the codebase (e.g. "How does X work?", "What calls Y?", "Trace the data flow through Z") and NOT an explicit rebuild command (`--update`, `--cluster-only`, or a bare path/URL that implies fresh extraction): **skip Steps 1–5 entirely and jump straight to `## For /graphify query`.** Run `graphify query "<question>"` immediately. Do not run detect. Do not check corpus size. Do not ask the user to narrow. The graph is already built — use it.
+
+If no path was given, use `.` (current directory). Do not ask the user for a path.
+
+If the path argument starts with `https://github.com/` or `http://github.com/`, treat it as a GitHub URL - run Step 0 before anything else, then continue with the resolved local path.
+
+Follow these steps in order. Do not skip steps.
+
+### Step 0 - GitHub repos and multi-path merge (only if a URL or several paths)
+
+Only when the path is one or more `https://github.com/...` URLs, or several local subfolders to merge. See `references/github-and-merge.md` for the clone, cross-repo merge, and monorepo flow, then continue with the resolved local path. A plain local path skips this step.
+
+### Step 1 - Ensure graphify is installed
+
+```bash
+# Detect the correct Python interpreter (handles uv tool, pipx, venv, system installs)
+PYTHON=""
+GRAPHIFY_BIN=$(which graphify 2>/dev/null)
+# 1. uv tool installs — most reliable on modern Mac/Linux
+if [ -z "$PYTHON" ] && command -v uv >/dev/null 2>&1; then
+    _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null)
+    if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi
+fi
+# 2. Read shebang from graphify binary (pipx and direct pip installs)
+if [ -z "$PYTHON" ] && [ -n "$GRAPHIFY_BIN" ]; then
+    _SHEBANG=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!')
+    case "$_SHEBANG" in
+        *[!a-zA-Z0-9/_.@-]*) ;;
+        *) "$_SHEBANG" -c "import graphify" 2>/dev/null && PYTHON="$_SHEBANG" ;;
+    esac
+fi
+# 3. Fall back to python3
+if [ -z "$PYTHON" ]; then PYTHON="python3"; fi
+if ! "$PYTHON" -c "import graphify" 2>/dev/null; then
+    if command -v uv >/dev/null 2>&1; then
+        uv tool install --upgrade graphifyy -q 2>&1 | tail -3
+        _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null)
+        if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi
+    else
+        "$PYTHON" -m pip install graphifyy -q 2>/dev/null \
+          || "$PYTHON" -m pip install graphifyy -q --break-system-packages 2>&1 | tail -3
+    fi
+fi
+# Write interpreter path for all subsequent steps (persists across invocations)
+mkdir -p graphify-out
+"$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)"
+# Save scan root so `graphify update` (no args) knows where to look next time
+echo "$(cd INPUT_PATH && pwd)" > graphify-out/.graphify_root
+```
+
+If the import succeeds, print nothing and move straight to Step 2.
+
+**In every subsequent bash block, replace `python3` with `$(cat graphify-out/.graphify_python)` to use the correct interpreter.**
+
+### Step 2 - Detect files
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from graphify.detect import detect
+from pathlib import Path
+result = detect(Path('INPUT_PATH'))
+# Write the sidecar from Python, not a shell redirect, so the same block renders
+# on PowerShell hosts without console-encoding drift (#2528).
+Path('graphify-out/.graphify_detect.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\")
+print(f'Detected {result[\"total_files\"]} files')
+"
+```
+
+Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead:
+
+```
+Corpus: X files · ~Y words
+  code:     N files (.py .ts .go ...)
+  docs:     N files (.md .txt ...)
+  papers:   N files (.pdf ...)
+  images:   N files
+  video:    N files (.mp4 .mp3 ...)
+```
+
+Omit any category with 0 files from the summary.
+
+Then act on it:
+- If `total_files` is 0: stop with "No supported files found in [path]."
+- If `skipped_sensitive` is non-empty: report the count and list the skipped file names, so a wrongly-flagged source or doc is visible and can be renamed or moved (#2106).
+- If `total_words` > 2,000,000 OR `total_files` > 500: show the warning. Then compute the top 5 first-level subdirectories by file count:
+  - Read `scan_root` from the detect JSON (always an absolute path to the resolved INPUT_PATH).
+  - Concatenate all file lists across all types (`code`, `document`, `paper`, `image`, `video`).
+  - Filter out any path that starts with `scan_root + "/graphify-out/"` to exclude converted sidecars.
+  - For each file, strip the `scan_root` prefix and take the first path component. Files directly in `scan_root` with no subdirectory count as `(root)`.
+  - If all files are in `(root)` with no subdirectories, do not ask to narrow — no subfolders exist. Instead suggest `--no-cluster` to skip the expensive clustering step and proceed.
+  - Otherwise rank by count, show the top 5 with file counts, then ask which subfolder to run on. Wait for the user's answer before proceeding.
+- Otherwise: proceed directly to Step 2.5 if video files were detected, or Step 3 if not.
+
+### Step 2.5 - Video and audio (only if video files detected)
+
+Skip this step entirely if `detect` returned zero `video` files. When the corpus has video or audio, see `references/transcribe.md` to transcribe them to text first, then treat the transcripts as doc files in Step 3.
+
+### Step 3 - Extract entities and relationships
+
+**Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it.
+
+This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (LLM, costs tokens).
+
+> **graphify needs no API key. Never ask the user for one, and never block on one.** Code is extracted structurally (AST) with no LLM and no key at all — a code-only corpus (the common `/graphify .` on a repo) skips semantic extraction entirely, so it needs nothing here: go straight to Part A and skip Part B. Semantic extraction (only for docs, papers, and images) uses Gemini **only if** `GEMINI_API_KEY`/`GOOGLE_API_KEY` is already set; otherwise the host agent itself is the LLM. graphify does **not** read `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, or any other provider key. If you catch yourself about to prompt for, wait on, or stop because of a missing API key, that is a misread of this skill — proceed without one.
+
+**Before semantic extraction:** check whether `GEMINI_API_KEY` or `GOOGLE_API_KEY` is set. If neither is set, print this one-liner to the user:
+> Tip: set `GEMINI_API_KEY` or `GOOGLE_API_KEY` to use Gemini for semantic extraction (`pip install 'graphifyy[gemini]'`).
+
+Print it once, then continue — do not wait for the user to supply a key. If `GEMINI_API_KEY` or `GOOGLE_API_KEY` IS set, use `graphify.llm.extract_corpus_parallel(files, backend="gemini")` for semantic extraction instead of dispatching subagents. The default Gemini model is `gemini-3-flash-preview`; set `GRAPHIFY_GEMINI_MODEL` or pass `--model` in headless CLI flows to override it.
+
+> **No other API keys are read.** When `GEMINI_API_KEY`/`GOOGLE_API_KEY` are unset, semantic extraction falls to the host agent itself — the running session is the LLM. On a host that dispatches subagents (e.g. Claude Code), dispatch them as written in Part B. On a host that runs the CLI directly in a terminal and cannot dispatch subagents, do not stall: a code-only corpus has no semantic work, so write the empty semantic file (Part B "Fast path") and continue to Part C; for a corpus with docs/papers/images, either set a Gemini key or extract those inline yourself, but in no case prompt for `ANTHROPIC_API_KEY` — that prompt is a misread of this skill.
+
+**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.**
+
+Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers.
+
+#### Part A - Structural extraction for code files
+
+For any code files detected, run AST extraction in parallel with Part B subagents:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import sys, json
+from graphify.extract import collect_files, extract
+from pathlib import Path
+import json
+
+code_files = []
+detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
+for f in detect.get('files', {}).get('code', []):
+    code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)])
+
+if code_files:
+    result = extract(code_files, cache_root=Path('INPUT_PATH'))
+    Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding=\"utf-8\")
+    print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges')
+else:
+    Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding=\"utf-8\")
+    print('No code files - skipping AST extraction')
+"
+```
+
+#### Part B - Semantic extraction (parallel subagents)
+
+**Fast path:** If detection found zero docs, papers, and images (code-only corpus), skip Part B entirely and go straight to Part C. AST handles code - there is nothing for semantic subagents to do. **First write an empty semantic file** so Part C's merge has its input (it reads `.graphify_semantic.json` unconditionally; without this a code-only run hits `FileNotFoundError`):
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+Path('graphify-out/.graphify_semantic.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8')
+"
+```
+
+**MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden - it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.**
+
+Before dispatching subagents, print a timing estimate:
+- Load `total_words` and file counts from `graphify-out/.graphify_detect.json`
+- Estimate agents needed: `ceil(uncached_non_code_files / 22)` (chunk size is 20-25)
+- Estimate time: ~45s per agent batch (they run in parallel, so total ≈ 45s × ceil(agents/parallel_limit))
+- Print: "Semantic extraction: ~N files → X agents, estimated ~Ys"
+
+**Step B0 - Check extraction cache first**
+
+Before dispatching any subagents, check which files already have cached extraction results:
+
+SPEC_PATH below is the **absolute** path of the `references/extraction-spec.md` that ships beside this SKILL.md — the same file Step B2 loads and hands to every subagent. It is the extraction prompt, so cache entries are attributed to it: when a graphify upgrade changes the prompt, entries produced by the old one are re-extracted instead of replayed, and unchanged prompts keep their entries (#1939). Substitute the real path in both Step B0 and Step B3 — pass the same one to each, and do not drop the argument.
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from graphify.cache import check_semantic_cache
+from pathlib import Path
+
+detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
+# Only content files go to semantic extraction. Code is already covered structurally
+# by the AST pass (Part A); flattening every category here makes subagents re-read
+# every source file (#1392). Video is transcribed to a document in Step 2.5 first.
+all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])]
+
+cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files, root='INPUT_PATH', prompt_file='SPEC_PATH')
+
+# Always (re)write the cache file: write hits, else DELETE any leftover from a prior
+# run so Part C never merges a stale .graphify_cached.json (#1392).
+if cached_nodes or cached_edges or cached_hyperedges:
+    Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding=\"utf-8\")
+else:
+    Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True)
+Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding=\"utf-8\")
+print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction')
+"
+```
+
+Only dispatch subagents for files listed in `graphify-out/.graphify_uncached.txt`. If all files are cached, skip to Part C directly.
+
+**Step B1 - Split into chunks**
+
+Load files from `graphify-out/.graphify_uncached.txt`. Split into chunks of 20-25 files each. Each image gets its own chunk (vision needs separate context). When splitting, group files from the same directory together so related artifacts land in the same chunk and cross-file relationships are more likely to be extracted.
+
+**Step B2 - Dispatch ALL subagents in a single message**
+
+Call the Agent tool multiple times IN THE SAME RESPONSE - one call per chunk. This is the only way they run in parallel. If you make one Agent call, wait, then make another, you are doing it sequentially and defeating the purpose.
+
+**IMPORTANT - subagent type:** Always use `subagent_type="general-purpose"`. Do NOT use `Explore` - it is read-only and cannot write chunk files to disk, which silently drops extraction results. General-purpose has Write and Bash access which the subagent needs.
+
+Concrete example for 3 chunks:
+```
+[Agent tool call 1: files 1-15, subagent_type="general-purpose"]
+[Agent tool call 2: files 16-30, subagent_type="general-purpose"]
+[Agent tool call 3: files 31-45, subagent_type="general-purpose"]
+```
+All three in one message. Not three separate messages.
+
+Each subagent receives this exact prompt (substitute FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH).
+
+CHUNK_PATH must be an **absolute** path — derive it before dispatching:
+```bash
+PROJECT_ROOT=$(pwd)  # cwd — where Part C globs graphify-out/ (NOT .graphify_root/scan dir, #1392)
+# Then for chunk N: CHUNK_PATH="${PROJECT_ROOT}/graphify-out/.graphify_chunk_0N.json"
+```
+
+Subagent prompt template:
+
+See `references/extraction-spec.md` for the exact subagent prompt (JSON schema, node-ID rules, confidence rubric, frontmatter, hyperedge, and vision rules). Load it only here, only when at least one chunk holds a doc, paper, or image; a pure-code corpus has skipped Part B and never reads it. Pass each subagent that prompt verbatim with FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH substituted, and have it write the result to CHUNK_PATH.
+
+**Step B3 - Collect, cache, and merge**
+
+Wait for all subagents. For each result:
+- Check that `graphify-out/.graphify_chunk_NN.json` exists on disk — this is the success signal
+- If the file exists and contains valid JSON with `nodes` and `edges`, include it and save to cache
+- If the file is missing, the subagent was likely dispatched as read-only (Explore type) — print a warning: "chunk N missing from disk — subagent may have been read-only. Re-run with general-purpose agent." Do not silently skip.
+- If a subagent failed or returned invalid JSON, print a warning and skip that chunk - do not abort
+
+If more than half the chunks failed or are missing, stop and tell the user to re-run and ensure `subagent_type="general-purpose"` is used.
+
+Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent call completes, read the real token counts from the Agent tool result's `usage` field and write them back into the chunk JSON before merging** — the chunk JSON itself always has placeholder zeros. Then run:
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json, glob
+from pathlib import Path
+
+chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json'))
+all_nodes, all_edges, all_hyperedges = [], [], []
+total_in, total_out = 0, 0
+for c in chunks:
+    d = json.loads(Path(c).read_text(encoding=\"utf-8\"))
+    all_nodes += d.get('nodes', [])
+    all_edges += d.get('edges', [])
+    all_hyperedges += d.get('hyperedges', [])
+    total_in += d.get('input_tokens', 0)
+    total_out += d.get('output_tokens', 0)
+Path('graphify-out/.graphify_semantic_new.json').write_text(json.dumps({
+    'nodes': all_nodes, 'edges': all_edges, 'hyperedges': all_hyperedges,
+    'input_tokens': total_in, 'output_tokens': total_out,
+}, indent=2, ensure_ascii=False), encoding=\"utf-8\")
+print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens')
+"
+```
+
+Save new results to cache. Pass the same SPEC_PATH as Step B0 — it stamps each entry with the prompt that produced it, and a write under a different prompt than the read lands where the next run won't look (#1939):
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from graphify.cache import save_semantic_cache
+from pathlib import Path
+
+new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]}
+uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line]
+saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []), root='INPUT_PATH', allowed_source_files=uncached, prompt_file='SPEC_PATH')
+print(f'Cached {saved} files')
+"
+```
+
+Merge cached + new results into `graphify-out/.graphify_semantic.json`:
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+
+cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]}
+new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]}
+
+all_nodes = cached['nodes'] + new.get('nodes', [])
+all_edges = cached['edges'] + new.get('edges', [])
+all_hyperedges = cached.get('hyperedges', []) + new.get('hyperedges', [])
+seen = set()
+deduped = []
+for n in all_nodes:
+    if n['id'] not in seen:
+        seen.add(n['id'])
+        deduped.append(n)
+
+merged = {
+    'nodes': deduped,
+    'edges': all_edges,
+    'hyperedges': all_hyperedges,
+    'input_tokens': new.get('input_tokens', 0),
+    'output_tokens': new.get('output_tokens', 0),
+}
+Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\")
+print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)')
+"
+```
+Clean up temp files: `rm -f graphify-out/.graphify_cached.json graphify-out/.graphify_uncached.txt graphify-out/.graphify_semantic_new.json`
+
+#### Part C - Merge AST + semantic into final extraction
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import sys, json
+from pathlib import Path
+
+ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding=\"utf-8\"))
+sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding=\"utf-8\"))
+
+# Merge: AST nodes first, semantic nodes deduplicated by id
+seen = {n['id'] for n in ast['nodes']}
+merged_nodes = list(ast['nodes'])
+for n in sem['nodes']:
+    if n['id'] not in seen:
+        merged_nodes.append(n)
+        seen.add(n['id'])
+
+merged_edges = ast['edges'] + sem['edges']
+merged_hyperedges = sem.get('hyperedges', [])
+merged = {
+    'nodes': merged_nodes,
+    'edges': merged_edges,
+    'hyperedges': merged_hyperedges,
+    'input_tokens': sem.get('input_tokens', 0),
+    'output_tokens': sem.get('output_tokens', 0),
+}
+Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\")
+total = len(merged_nodes)
+edges = len(merged_edges)
+print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)')
+"
+```
+
+### Step 4 - Build graph, cluster, analyze, generate outputs
+
+**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source→target), otherwise `False` (the default undirected `Graph`). Substitute it the same way you substitute `INPUT_PATH` — do not leave the literal `IS_DIRECTED` in the code.
+
+```bash
+mkdir -p graphify-out
+$(cat graphify-out/.graphify_python) -c "
+import sys, json
+from graphify.build import build_from_json
+from graphify.cluster import cluster, score_all
+from graphify.analyze import god_nodes, surprising_connections, suggest_questions
+from graphify.report import generate
+from graphify.export import to_json
+from pathlib import Path
+
+extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
+detection  = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
+
+# root= mirrors the --update runbook (#1361): relativize source_file to the same
+# base so the full build and incremental --update never drift apart on re-extract.
+G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED)
+# Guard BEFORE any write: an empty extraction must not clobber a good graph.json /
+# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392).
+if G.number_of_nodes() == 0:
+    print('ERROR: Graph is empty - extraction produced no nodes.')
+    print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.')
+    raise SystemExit(1)
+communities = cluster(G)
+cohesion = score_all(G, communities)
+tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)}
+gods = god_nodes(G)
+surprises = surprising_connections(G, communities)
+labels = {cid: 'Community ' + str(cid) for cid in communities}
+# Placeholder questions - regenerated with real labels in Step 5
+questions = suggest_questions(G, communities, labels)
+
+# Export FIRST and honor the #479 shrink-guard: to_json returns False (writing
+# nothing) when the new graph is smaller than the existing graph.json. Only write
+# GRAPH_REPORT.md + the analysis sidecar when the graph was actually written, so
+# they never describe a graph that graph.json doesn't contain (#1392).
+wrote = to_json(G, communities, 'graphify-out/graph.json')
+if not wrote:
+    print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).')
+    print('If this shrink is intentional (you deleted files), re-run a full build with --force.')
+    raise SystemExit(1)
+report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions)
+Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\")
+analysis = {
+    'communities': {str(k): v for k, v in communities.items()},
+    'cohesion': {str(k): v for k, v in cohesion.items()},
+    'gods': gods,
+    'surprises': surprises,
+    'questions': questions,
+}
+Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding=\"utf-8\")
+print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities')
+"
+```
+
+If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization.
+
+Replace INPUT_PATH with the actual path.
+
+### Step 4.5 - Graph health check (read-only integrity gate)
+
+A non-destructive diagnostic on the extraction, before labeling. It surfaces edge collapse, dangling/missing endpoints, and self-loops — the silent-corruption modes of incremental updates and AST/LLM id mismatches. Read-only; never aborts.
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+from graphify.diagnostics import diagnose_extraction, format_diagnostic_report
+
+extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
+summary = diagnose_extraction(extraction, directed=IS_DIRECTED, root='INPUT_PATH')
+print(format_diagnostic_report(summary))
+flags = [f'{summary[k]} {label}' for k, label in (
+    ('dangling_endpoint_edges', 'dangling-endpoint edges'),
+    ('missing_endpoint_edges', 'missing-endpoint edges'),
+    ('self_loop_edges', 'self-loop edges'),
+    ('directed_same_endpoint_collapsed_edges', 'collapsed (directed) edges'),
+    ('undirected_same_endpoint_collapsed_edges', 'collapsed (undirected) edges'),
+) if summary.get(k, 0)]
+print('GRAPH HEALTH WARNING: ' + '; '.join(flags) + ' - graph may be incomplete/corrupt.' if flags else 'Graph health: OK (no dangling/missing/collapsed edges).')
+"
+```
+
+Substitute `IS_DIRECTED` and `INPUT_PATH` as in Step 4. If a `GRAPH HEALTH WARNING` prints, surface it in the final summary (do not abort — the graph is still usable, but the integrity issue must be visible, per the Honesty Rules).
+
+### Step 5 - Label communities
+
+Read `graphify-out/.graphify_analysis.json`. For each community key, look at its node labels and write a 2-5 word plain-language name (e.g. "Attention Mechanism", "Training Pipeline", "Data Loading").
+
+Then regenerate the report and save the labels for the visualizer:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import sys, json
+from graphify.build import build_from_json
+from graphify.cluster import score_all
+from graphify.analyze import god_nodes, surprising_connections, suggest_questions
+from graphify.report import generate
+from graphify.export import to_json
+from pathlib import Path
+
+extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
+detection  = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
+analysis   = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding=\"utf-8\"))
+
+# root= as in Step 4 / the --update runbook (#1361) — same base for node-key parity.
+G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED)
+communities = {int(k): v for k, v in analysis['communities'].items()}
+cohesion = {int(k): v for k, v in analysis['cohesion'].items()}
+tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)}
+
+# LABELS - replace these with the names you chose above
+labels = LABELS_DICT
+
+# Regenerate questions with real community labels (labels affect question phrasing)
+questions = suggest_questions(G, communities, labels)
+
+report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions)
+Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\")
+Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\")
+# Re-export so graph.json nodes carry the curated community_name (#2490).
+# Same extraction as Step 4, so the #479 shrink-guard passes on node count;
+# if it still refuses, surface the guard message - do not force past it.
+wrote = to_json(G, communities, 'graphify-out/graph.json', community_labels=labels)
+if not wrote:
+    print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).')
+    print('If this shrink is intentional (you deleted files), re-run a full build with --force.')
+print('Report updated with community labels')
+"
+```
+
+Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`).
+Replace INPUT_PATH with the actual path.
+
+### Step 6 - Generate Obsidian vault (opt-in) + HTML
+
+**Generate HTML always** (unless `--no-viz`). **Obsidian vault only if `--obsidian` was explicitly given** — skip it otherwise, it generates one file per node.
+
+If `--obsidian` was given:
+
+- If `--obsidian-dir <path>` was also given, pass it via `--dir`. Otherwise defaults to `graphify-out/obsidian`.
+
+```bash
+graphify export obsidian
+# or with custom dir: graphify export obsidian --dir ~/vaults/my-project
+```
+
+Generate the HTML graph (always, unless `--no-viz`):
+
+```bash
+graphify export html  # auto-aggregates to community view if graph > 5000 nodes
+# or: graphify export html --no-viz
+```
+
+### Steps 6b-8 - Wiki, Neo4j, FalkorDB, SVG, GraphML, MCP, benchmark (only on their flags)
+
+These run only when their flag is present (`--wiki`, `--neo4j`/`--neo4j-push`, `--falkordb`/`--falkordb-push`, `--svg`, `--graphml`, `--mcp`) or, for the token-reduction benchmark, when `total_words` exceeds 5,000. A default run with no export flags skips all of them. See `references/exports.md` for each one. Run any `--wiki` export before Step 9 cleanup so `.graphify_labels.json` is still available.
+
+---
+
+### Step 9 - Save manifest, update cost tracker, clean up, and report
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+from datetime import datetime, timezone
+from graphify.detect import save_manifest
+
+# Save manifest for --update
+detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
+extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
+# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
+# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
+# root= relativizes the manifest keys to the scan root (same base as the build),
+# so the on-disk manifest is portable across clones/machines and a later --update
+# matches cached files instead of missing every one (#1417).
+#
+# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
+# a detected file whose chunk failed or was omitted must stay unstamped so the
+# next --update re-queues it, otherwise it is marked done and its content is lost
+# forever (#2015). This mirrors the library extract path exactly
+# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
+# raw corpus. Code files are always stamped (AST is deterministic); only semantic
+# types are gated on output.
+from graphify.cli import _stamped_manifest_files
+_corpus = detect.get('all_files') or detect['files']
+_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
+# Files dispatched this run (the changed subset) but NOT stamped above still carry
+# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
+# them instead of reading them as unchanged (#1948).
+_sem_types = ('document', 'paper', 'image')
+_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
+_stamped = {f for fl in _manifest_files.values() for f in fl}
+_cleared = _dispatched - _stamped
+# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
+# files newly excluded since last run are dropped rather than masquerading as
+# deletions; untouched files' prior rows are still preserved (#1908).
+_scan = {f for fl in _corpus.values() for f in fl}
+save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
+
+# Update cumulative cost tracker
+input_tok = extract.get('input_tokens', 0)
+output_tok = extract.get('output_tokens', 0)
+
+cost_path = Path('graphify-out/cost.json')
+if cost_path.exists():
+    cost = json.loads(cost_path.read_text(encoding=\"utf-8\"))
+else:
+    cost = {'runs': [], 'total_input_tokens': 0, 'total_output_tokens': 0}
+
+cost['runs'].append({
+    'date': datetime.now(timezone.utc).isoformat(),
+    'input_tokens': input_tok,
+    'output_tokens': output_tok,
+    'files': detect.get('total_files', 0),
+})
+cost['total_input_tokens'] += input_tok
+cost['total_output_tokens'] += output_tok
+cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding=\"utf-8\")
+
+print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens')
+print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)')
+"
+rm -f graphify-out/.graphify_detect.json graphify-out/.graphify_extract.json graphify-out/.graphify_ast.json graphify-out/.graphify_semantic.json graphify-out/.graphify_analysis.json
+find graphify-out -maxdepth 1 -name '.graphify_chunk_*.json' -delete 2>/dev/null
+rm -f graphify-out/.needs_update 2>/dev/null || true
+```
+
+Replace INPUT_PATH with the actual path (same value used in Steps 4-5) so the manifest is relativized to the scan root.
+
+Tell the user (omit the obsidian line unless --obsidian was given):
+```
+Graph complete. Outputs in PATH_TO_DIR/graphify-out/
+
+  graph.html            - interactive graph, open in browser
+  GRAPH_REPORT.md       - audit report
+  graph.json            - raw graph data
+  obsidian/             - Obsidian vault (only if --obsidian was given)
+```
+
+If graphify saved you time, consider supporting it: https://github.com/sponsors/safishamsi
+
+Replace PATH_TO_DIR with the actual absolute path of the directory that was processed.
+
+Then paste these sections from GRAPH_REPORT.md directly into the chat:
+- God Nodes
+- Surprising Connections
+- Suggested Questions
+
+Do NOT paste the full report - just those three sections. Keep it concise.
+
+Then immediately offer to explore. Pick the single most interesting suggested question from the report - the one that crosses the most community boundaries or has the most surprising bridge node - and ask:
+
+> "The most interesting question this graph can answer: **[question]**. Want me to trace it?"
+
+If the user says yes, run `/graphify query "[question]"` on the graph and walk them through the answer using the graph structure - which nodes connect, which community boundaries get crossed, what the path reveals. Keep going as long as they want to explore. Each answer should end with a natural follow-up ("this connects to X - want to go deeper?") so the session feels like navigation, not a one-shot report.
+
+The graph is the map. Your job after the pipeline is to be the guide.
+
+---
+
+## Interpreter guard for subcommands
+
+Before running any subcommand below (`--update`, `--cluster-only`, `query`, `path`, `explain`, `add`), check that `.graphify_python` exists. If it's missing (e.g. user deleted `graphify-out/`), re-resolve the interpreter first:
+
+```bash
+if [ ! -f graphify-out/.graphify_python ]; then
+    GRAPHIFY_BIN=$(which graphify 2>/dev/null)
+    if [ -n "$GRAPHIFY_BIN" ]; then
+        PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!')
+        case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac
+    else
+        PYTHON="python3"
+    fi
+    mkdir -p graphify-out
+    "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)"
+fi
+```
+
+## For --update and --cluster-only
+
+Both are non-default subcommands. `--update` re-extracts only new or changed files; `--cluster-only` reruns clustering on the existing graph. See `references/update.md` for both flows.
+
+---
+
+## For /graphify query
+
+When `graphify-out/graph.json` already exists and the user asks a question about the corpus, answer from the graph rather than rebuilding it:
+
+```bash
+graphify query "<question>"
+```
+
+Before traversal, expand the question against the graph's own vocabulary so a wording mismatch does not collapse the answer to noise. If the `graphify query` CLI is unavailable, fall back to an inline NetworkX traversal of `graphify-out/graph.json`. Answer using only what the graph output contains, and quote `source_location` when citing a specific fact. For that vocab-expansion step, the BFS/DFS traversal modes, the `--budget` cap, the NetworkX fallback, `save-result` feedback, and the `/graphify path` and `/graphify explain` flows, see `references/query.md`.
+
+---
+
+## For /graphify add and --watch
+
+Neither is part of the default build. When the user runs `/graphify add <url>` to fetch a URL into the corpus, or passes `--watch` to auto-rebuild on file changes, see `references/add-watch.md`.
+
+---
+
+## For the commit hook and native CLAUDE.md integration
+
+When the user asks to install the post-commit auto-rebuild hook or wire graphify into a project's CLAUDE.md, see `references/hooks.md`.
+
+---
+
+## Honesty Rules
+
+- Never invent an edge. If unsure, use AMBIGUOUS.
+- Never skip the corpus check warning.
+- Always show token cost in the report.
+- Never hide cohesion scores behind symbols - show the raw number.
+- Never run HTML viz on a graph with more than 5,000 nodes without warning the user.

+ 56 - 0
.claude/skills/graphify/references/add-watch.md

@@ -0,0 +1,56 @@
+# graphify reference: add a URL and watch a folder
+
+Load this when the user ran `/graphify add <url>` or passed `--watch`. Neither is part of the default build.
+
+## For /graphify add
+
+Fetch a URL and add it to the corpus, then update the graph.
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import sys
+from graphify.ingest import ingest
+from pathlib import Path
+
+try:
+    out = ingest('URL', Path('./raw'), author='AUTHOR', contributor='CONTRIBUTOR')
+    print(f'Saved to {out}')
+except ValueError as e:
+    print(f'error: {e}', file=sys.stderr)
+    sys.exit(1)
+except RuntimeError as e:
+    print(f'error: {e}', file=sys.stderr)
+    sys.exit(1)
+"
+```
+
+Replace `URL` with the actual URL, `AUTHOR` with the user's name if provided, `CONTRIBUTOR` likewise. If the command exits with an error, tell the user what went wrong - do not silently continue. After a successful save, automatically run the `--update` pipeline on `./raw` to merge the new file into the existing graph.
+
+Supported URL types (auto-detected):
+- YouTube / any video URL → audio downloaded via yt-dlp, transcribed to `.txt` on next run (requires `pip install 'graphifyy[video]'`)
+- Twitter/X → fetched via oEmbed, saved as `.md` with tweet text and author
+- arXiv → abstract + metadata saved as `.md`
+- PDF → downloaded as `.pdf`
+- Images (.png/.jpg/.webp) → downloaded, Claude vision extracts on next run
+- Any webpage → converted to markdown via html2text
+
+---
+
+## For --watch
+
+Start a background watcher that monitors a folder and auto-updates the graph when files change.
+
+```bash
+$(cat graphify-out/.graphify_python) -m graphify.watch INPUT_PATH --debounce 3
+```
+
+Replace INPUT_PATH with the folder to watch. Behavior depends on what changed:
+
+- **Code files only (.py, .ts, .go, etc.):** re-runs AST extraction + rebuild + cluster immediately, no LLM needed. `graph.json` and `GRAPH_REPORT.md` are updated automatically.
+- **Docs, papers, or images:** writes a `graphify-out/needs_update` flag and prints a notification to run `/graphify --update` (LLM semantic re-extraction required).
+
+Debounce (default 3s): waits until file activity stops before triggering, so a wave of parallel agent writes doesn't trigger a rebuild per file.
+
+Press Ctrl+C to stop.
+
+For agentic workflows: run `--watch` in a background terminal. Code changes from agent waves are picked up automatically between waves. If agents are also writing docs or notes, you'll need a manual `/graphify --update` after those waves.

+ 87 - 0
.claude/skills/graphify/references/exports.md

@@ -0,0 +1,87 @@
+# graphify reference: extra exports and benchmark
+
+Load this when the user passed one of the export flags (`--wiki`, `--neo4j`, `--neo4j-push`, `--falkordb`, `--falkordb-push`, `--svg`, `--graphml`, `--mcp`), or when the corpus is large enough for the token-reduction benchmark. Each step runs only for its own flag.
+
+### Step 6b - Wiki (only if --wiki flag)
+
+**Only run this step if `--wiki` was explicitly given in the original command.**
+
+Run this before Step 9 (cleanup) so `.graphify_labels.json` is still available.
+
+```bash
+graphify export wiki
+```
+
+### Step 7 - Neo4j export (only if --neo4j or --neo4j-push flag)
+
+**If `--neo4j`** - generate a Cypher file for manual import:
+
+```bash
+graphify export neo4j
+```
+
+**If `--neo4j-push <uri>`** - push directly to a running Neo4j instance. Ask the user for credentials if not provided:
+
+```bash
+graphify export neo4j --push bolt://localhost:7687 --user neo4j --password PASSWORD
+```
+
+Default URI is `bolt://localhost:7687`, default user is `neo4j`. Uses MERGE - safe to re-run without creating duplicates.
+
+### Step 7a - FalkorDB export (only if --falkordb or --falkordb-push flag)
+
+**If `--falkordb`** - generate a Cypher file. The statements are OpenCypher, but FalkorDB's `GRAPH.QUERY` runs one statement at a time (no bulk script import like Neo4j's `cypher-shell`), so prefer `--falkordb-push` to load a graph. Use this only when you want the portable `cypher.txt` artifact:
+
+```bash
+graphify export falkordb
+```
+
+**If `--falkordb-push <uri>`** - push directly to a running FalkorDB instance. Credentials are optional; ask the user only if the instance requires auth:
+
+```bash
+graphify export falkordb --push falkordb://localhost:6379
+```
+
+Default URI is `falkordb://localhost:6379` (the scheme is informational - `redis://` or a bare `host:port` work too), auth is optional, and the target graph defaults to `graphify`. Uses MERGE - safe to re-run without creating duplicates.
+
+### Step 7b - SVG export (only if --svg flag)
+
+```bash
+graphify export svg
+```
+
+### Step 7c - GraphML export (only if --graphml flag)
+
+```bash
+graphify export graphml
+```
+
+### Step 7d - MCP server (only if --mcp flag)
+
+```bash
+$(cat graphify-out/.graphify_python) -m graphify.serve graphify-out/graph.json
+```
+
+This starts a stdio MCP server that exposes tools: `query_graph`, `get_node`, `get_neighbors`, `get_community`, `god_nodes`, `graph_stats`, `shortest_path`. Add to Claude Desktop or any MCP-compatible agent orchestrator so other agents can query the graph live.
+
+To configure in Claude Desktop, add to `claude_desktop_config.json`. Claude Desktop can't run `$(...)`, and under `uv tool install` the system `python3` can't import graphify — so set `command` to the **absolute interpreter path** printed by `cat graphify-out/.graphify_python`:
+```json
+{
+  "mcpServers": {
+    "graphify": {
+      "command": "<absolute path from: cat graphify-out/.graphify_python>",
+      "args": ["-m", "graphify.serve", "/absolute/path/to/graphify-out/graph.json"]
+    }
+  }
+}
+```
+
+### Step 8 - Token reduction benchmark (only if total_words > 5000)
+
+If `total_words` from `graphify-out/.graphify_detect.json` is greater than 5,000, run:
+
+```bash
+graphify benchmark
+```
+
+Print the output directly in chat. If `total_words <= 5000`, skip silently - the graph value is structural clarity, not token compression, for small corpora.

+ 70 - 0
.claude/skills/graphify/references/extraction-spec.md

@@ -0,0 +1,70 @@
+# graphify reference: extraction subagent prompt
+
+Load this in Step 3 Part B when the corpus has at least one doc, paper, or image chunk. A pure-code corpus skips Part B and never reads this file. Each semantic subagent receives the prompt below verbatim (substitute FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH).
+
+```
+You are a graphify extraction subagent. Read the files listed and extract a knowledge graph fragment.
+Output ONLY valid JSON matching the schema below - no explanation, no markdown fences, no preamble.
+
+Files (chunk CHUNK_NUM of TOTAL_CHUNKS):
+FILE_LIST
+
+Rules:
+- EXTRACTED: relationship explicit in source (import, call, citation, "see §3.2")
+- INFERRED: reasonable inference (shared data structure, implied dependency)
+- AMBIGUOUS: uncertain - flag for review, do not omit
+
+Code files: focus on semantic edges AST cannot find (call relationships, shared data, arch patterns).
+  Do not re-extract imports - AST already has those.
+Doc/paper files: extract named concepts, entities, citations. For rationale (WHY decisions were made, trade-offs, design intent): store as a `rationale` attribute on the relevant concept node — do NOT create a separate rationale node or fragment node. Only create a node for something that is itself a named entity or concept. Use `file_type:"rationale"` for concept-like nodes (ideas, principles, mechanisms, design patterns). `file_type` MUST be one of exactly these six values: `code`, `document`, `paper`, `image`, `rationale`, `concept`. Any other value is invalid and will be rejected.
+Code files: when adding `calls` edges, source MUST be the caller (the function/class doing the calling), target MUST be the callee. Never reverse this direction. `calls` edges MUST stay within one language: a Python function cannot `calls` a JS/TS/Go/Rust/Java symbol and vice versa — cross-language call edges are phantom artifacts, never emit them.
+Image files: use vision to understand what the image IS - do not just OCR.
+  UI screenshot: layout patterns, design decisions, key elements, purpose.
+  Chart: metric, trend/insight, data source.
+  Tweet/post: claim as node, author, concepts mentioned.
+  Diagram: components and connections.
+  Research figure: what it demonstrates, method, result.
+  Handwritten/whiteboard: ideas and arrows, mark uncertain readings AMBIGUOUS.
+
+DEEP_MODE (if --mode deep was given): be aggressive with INFERRED edges - indirect deps,
+  shared assumptions, latent couplings. Mark uncertain ones AMBIGUOUS instead of omitting.
+
+Semantic similarity: if two concepts in this chunk solve the same problem or represent the same idea without any structural link (no import, no call, no citation), add a `semantically_similar_to` edge marked INFERRED with a confidence_score reflecting how similar they are (0.6-0.95). Examples:
+- Two functions that both validate user input but never call each other
+- A class in code and a concept in a paper that describe the same algorithm
+- Two error types that handle the same failure mode differently
+Only add these when the similarity is genuinely non-obvious and cross-cutting. Do not add them for trivially similar things.
+
+Hyperedges: if 3 or more nodes clearly participate together in a shared concept, flow, or pattern that is not captured by pairwise edges alone, add a hyperedge to a top-level `hyperedges` array. Examples:
+- All classes that implement a common protocol or interface
+- All functions in an authentication flow (even if they don't all call each other)
+- All concepts from a paper section that form one coherent idea
+Use sparingly — only when the group relationship adds information beyond the pairwise edges. Maximum 3 hyperedges per chunk.
+
+If a file has YAML frontmatter (--- ... ---), copy source_url, captured_at, author,
+  contributor onto every node from that file.
+
+confidence_score is REQUIRED on every edge - never omit it, never use 0.5 as a default:
+- EXTRACTED edges: confidence_score = 1.0 always
+- INFERRED edges: pick exactly ONE value from this set — never 0.5:
+    0.95  direct structural evidence (shared data structure, named cross-file reference).
+    0.85  strong inference (clear functional alignment, no direct symbol link).
+    0.75  reasonable inference (shared problem domain + similar shape, requires interpretation).
+    0.65  weak inference (thematically related, no shape evidence).
+    0.55  speculative but plausible (surface-level co-occurrence only).
+  Models follow discrete rubrics better than continuous ranges; the bimodal
+  distribution observed in production (>50% at 0.5, >40% at 0.85+) shows the
+  range guidance is being collapsed to a binary. If no value above fits, mark
+  the edge AMBIGUOUS rather than picking 0.4 or below.
+- AMBIGUOUS edges: 0.1-0.3
+
+Node ID format: lowercase, only `[a-z0-9_]`, no dots or slashes. Format: `{stem}_{entity}` where stem is the **full repo-relative path with the extension dropped**, every path segment kept and joined with `_` (each segment lowercased with non-alphanumeric chars replaced by `_`), and entity is the symbol name similarly normalized. Use every directory level, not just the immediate parent — this keeps same-named files in different directories distinct. Examples: `src/auth/session.py` + `ValidateToken` → `src_auth_session_validatetoken`; `lib/utils/helpers.py` + `parse_url` → `lib_utils_helpers_parse_url`; `tests/test_foo.py` + `_helper` → `tests_test_foo_helper`; `docs/v1/api/README.md` + `getUser` → `docs_v1_api_readme_getuser`. Top-level files (no parent dir, e.g. `setup.py`) use just the filename stem: `setup_my_func`. This must match the ID the AST extractor generates — using just the filename (e.g., `session_validatetoken`) or only the immediate parent (e.g., `auth_session_validatetoken`) will create orphan ghost-duplicate nodes. If you are re-extracting a project built under the old immediate-parent format, the user should run `graphify extract --force` to rebuild cleanly. CRITICAL: never append chunk numbers, sequence numbers, or any suffix to an ID (no `_c1`, `_c2`, `_chunk2`, etc.). IDs must be deterministic from the label alone — the same entity must always produce the same ID regardless of which chunk processes it.
+
+Generate the extraction JSON matching this schema exactly:
+{"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"<FILE_LIST path verbatim>","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"<FILE_LIST path verbatim>","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":"<FILE_LIST path verbatim>"}],"input_tokens":0,"output_tokens":0}
+
+source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate.
+
+Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost):
+CHUNK_PATH
+```

+ 46 - 0
.claude/skills/graphify/references/github-and-merge.md

@@ -0,0 +1,46 @@
+# graphify reference: GitHub clone and cross-repo merge
+
+Load this when the user passed one or more `https://github.com/...` URLs, or named several local subfolders to merge into one graph.
+
+### Step 0 - Clone GitHub repo(s) (only if a GitHub URL was given)
+
+**Single repo:**
+```bash
+LOCAL_PATH=$(graphify clone <github-url> [--branch <branch>])
+# Use LOCAL_PATH as the target for all subsequent steps
+```
+
+**Multiple repos (cross-repo graph):**
+```bash
+# Clone each repo, run the full pipeline on each, then merge
+graphify clone <url1>   # → ~/.graphify/repos/<owner1>/<repo1>
+graphify clone <url2>   # → ~/.graphify/repos/<owner2>/<repo2>
+# Run /graphify on each local path to produce their graph.json files
+# Then merge:
+graphify merge-graphs \
+  ~/.graphify/repos/<owner1>/<repo1>/graphify-out/graph.json \
+  ~/.graphify/repos/<owner2>/<repo2>/graphify-out/graph.json \
+  --out graphify-out/cross-repo-graph.json
+```
+
+Graphify clones into `~/.graphify/repos/<owner>/<repo>` and reuses existing clones on repeat runs. Each node in the merged graph carries a `repo` attribute so you can filter by origin.
+
+**Multiple local subfolders (monorepo or multi-service layout):**
+
+The skill pipeline writes all intermediate and final outputs to `graphify-out/` in the current working directory. Running the skill on each subfolder separately will clobber the same output dir. Instead, use the CLI directly for each subfolder — it places `graphify-out/` *inside* the scanned path:
+
+```bash
+graphify extract ./core/     # → ./core/graphify-out/graph.json
+graphify extract ./service/  # → ./service/graphify-out/graph.json
+graphify extract ./platform/ # → ./platform/graphify-out/graph.json
+# Add --backend gemini|kimi|openai|deepseek|claude-cli depending on which API key you have set
+
+# Then merge at the project root:
+graphify merge-graphs \
+  ./core/graphify-out/graph.json \
+  ./service/graphify-out/graph.json \
+  ./platform/graphify-out/graph.json \
+  --out graphify-out/graph.json
+```
+
+Once `graphify-out/graph.json` exists, the fast path above takes over: any codebase question runs `graphify query` directly on the merged graph — no re-extraction, no size gate.

+ 33 - 0
.claude/skills/graphify/references/hooks.md

@@ -0,0 +1,33 @@
+# graphify reference: commit hook and native CLAUDE.md integration
+
+Load this when the user asked to install the post-commit hook or wire graphify into a project's CLAUDE.md.
+
+## For git commit hook
+
+Install a post-commit hook that auto-rebuilds the graph after every commit. No background process needed - triggers once per commit, works with any editor.
+
+```bash
+graphify hook install    # install
+graphify hook uninstall  # remove
+graphify hook status     # check
+```
+
+After every `git commit`, the hook detects which code files changed (via `git diff HEAD~1`), re-runs AST extraction on those files, and rebuilds `graph.json` and `GRAPH_REPORT.md`. Doc/image changes are ignored by the hook - run `/graphify --update` manually for those.
+
+If a post-commit hook already exists, graphify appends to it rather than replacing it.
+
+---
+
+## For native CLAUDE.md integration
+
+Run once per project to make graphify always-on in Claude Code sessions:
+
+```bash
+graphify claude install
+```
+
+This writes a `## graphify` section to the local `CLAUDE.md` that instructs Claude to check the graph before answering codebase questions and rebuild it after code changes. No manual `/graphify` needed in future sessions.
+
+```bash
+graphify claude uninstall  # remove the section
+```

+ 311 - 0
.claude/skills/graphify/references/query.md

@@ -0,0 +1,311 @@
+# graphify reference: query, path, explain
+
+Load this when the user asks a question against an existing graph, or runs `/graphify path` or `/graphify explain`. The core's query stub points here for the full traversal flow. These flows use the `graphify query` CLI when it is available and fall back to an inline NetworkX traversal otherwise.
+
+Two traversal modes - choose based on the question:
+
+| Mode | Flag | Best for |
+|------|------|----------|
+| BFS (default) | _(none)_ | "What is X connected to?" - broad context, nearest neighbors first |
+| DFS | `--dfs` | "How does X reach Y?" - trace a specific chain or dependency path |
+
+First check the graph exists:
+```bash
+$(cat graphify-out/.graphify_python) -c "
+from pathlib import Path
+if not Path('graphify-out/graph.json').exists():
+    print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
+    raise SystemExit(1)
+"
+```
+If it fails, stop and tell the user to run `/graphify <path>` first.
+
+### Step 0 — Constrained query expansion (REQUIRED before traversal)
+
+graphify's `query` CLI matches nodes via case-folded substring + IDF — there is **no stemming, no synonyms, no cross-language match** inside the binary, and the inline fallback below matches the same way. If the user's question uses different language or different domain vocabulary than the graph's labels (user says "обработчик" / graph says "handler"; user says "authentication" / graph says "Guardian"), the literal matcher returns 0 hits and the answer collapses to noise.
+
+Fix this **without inventing tokens** by expanding the query against the actual graph vocabulary first:
+
+1. Extract the token vocabulary from node labels:
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json, re
+from pathlib import Path
+data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
+vocab = set()
+for n in data['nodes']:
+    for c in re.findall(r'[^\W\d_]+', n.get('label','') or '', re.UNICODE):
+        parts = re.findall(r'[A-Z]+(?=[A-Z][a-z])|[A-Z]?[a-z]+|[A-Z]+', c) or [c]
+        for p in parts:
+            t = p.lower()
+            if 3 <= len(t) <= 30:
+                vocab.add(t)
+Path('graphify-out/.vocab.txt').write_text('\n'.join(sorted(vocab)), encoding='utf-8')
+print(f'vocab: {len(vocab)} tokens')
+"
+```
+
+2. Read `graphify-out/.vocab.txt`. Then for the user's question, select **up to 12 tokens from this exact list** that semantically match the query intent. Hard constraints:
+   - You MUST pick only tokens present in the vocabulary file. Do NOT invent tokens.
+   - If a query concept has no plausible token in the vocab, skip it — do not substitute a near-synonym from training memory.
+   - If **no** vocab tokens match the query at all, output an empty list and tell the user the corpus has no relevant vocabulary for this question. Do not fabricate a search.
+   - Translate cross-language: Russian "аутентификация" → look for `auth`, `credential`, `token`, `security` IFF present in vocab.
+   - Morphology: "handlers" maps to `handler` IFF present; "todos" maps to `todo` IFF present.
+
+3. Print the selection explicitly to the user before running the query, so the expansion is auditable:
+```
+Query expanded to (from graph vocab, N tokens): [token1, token2, ...]
+```
+If the list is empty, say so plainly and stop — do not proceed to traversal.
+
+### Step 1 — Traversal
+
+Build the **expanded query string** by joining the selected tokens with spaces. Use this string as `QUESTION` below — NOT the original user question. (The original question is preserved only for `save-result` at the end.)
+
+Prefer the CLI when it is installed:
+```bash
+graphify query "QUESTION"
+# or: graphify query "QUESTION" --dfs --budget 3000
+```
+
+If the CLI is unavailable, load `graphify-out/graph.json` and run the traversal inline:
+
+1. Find the 1-3 nodes whose label best matches the expanded tokens.
+2. Run the appropriate traversal from each starting node.
+3. Read the subgraph - node labels, edge relations, confidence tags, source locations.
+4. Answer using **only** what the graph contains. Quote `source_location` when citing a specific fact.
+5. If the graph lacks enough information, say so - do not hallucinate edges.
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import sys, json
+from networkx.readwrite import json_graph
+import networkx as nx
+from pathlib import Path
+
+data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
+G = json_graph.node_link_graph(data, edges='links')
+
+question = 'QUESTION'
+mode = 'MODE'  # 'bfs' or 'dfs'
+terms = [t.lower() for t in question.split() if len(t) >= 3]  # match the vocab threshold; keeps api/jwt/ios (#1392)
+
+# Find best-matching start nodes
+scored = []
+for nid, ndata in G.nodes(data=True):
+    label = ndata.get('label', '').lower()
+    score = sum(1 for t in terms if t in label)
+    if score > 0:
+        scored.append((score, nid))
+scored.sort(reverse=True)
+start_nodes = [nid for _, nid in scored[:3]]
+
+if not start_nodes:
+    print('No matching nodes found for query terms:', terms)
+    sys.exit(0)
+
+subgraph_nodes = set()
+subgraph_edges = []
+
+if mode == 'dfs':
+    # DFS: follow one path as deep as possible before backtracking.
+    # Depth-limited to 6 to avoid traversing the whole graph.
+    visited = set()
+    stack = [(n, 0) for n in reversed(start_nodes)]
+    while stack:
+        node, depth = stack.pop()
+        if node in visited or depth > 6:
+            continue
+        visited.add(node)
+        subgraph_nodes.add(node)
+        for neighbor in G.neighbors(node):
+            if neighbor not in visited:
+                stack.append((neighbor, depth + 1))
+                subgraph_edges.append((node, neighbor))
+else:
+    # BFS: explore all neighbors layer by layer up to depth 3.
+    frontier = set(start_nodes)
+    subgraph_nodes = set(start_nodes)
+    for _ in range(3):
+        next_frontier = set()
+        for n in frontier:
+            for neighbor in G.neighbors(n):
+                if neighbor not in subgraph_nodes:
+                    next_frontier.add(neighbor)
+                    subgraph_edges.append((n, neighbor))
+        subgraph_nodes.update(next_frontier)
+        frontier = next_frontier
+
+# Token-budget aware output: rank by relevance, cut at budget (~4 chars/token)
+token_budget = BUDGET  # default 2000
+char_budget = token_budget * 4
+
+# Score each node by term overlap for ranked output
+def relevance(nid):
+    label = G.nodes[nid].get('label', '').lower()
+    return sum(1 for t in terms if t in label)
+
+ranked_nodes = sorted(subgraph_nodes, key=relevance, reverse=True)
+
+lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get(\"label\",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes']
+for nid in ranked_nodes:
+    d = G.nodes[nid]
+    lines.append(f'  NODE {d.get(\"label\", nid)} [src={d.get(\"source_file\",\"\")} loc={d.get(\"source_location\",\"\")}]')
+for u, v in subgraph_edges:
+    if u in subgraph_nodes and v in subgraph_nodes:
+        _raw = G[u][v]; d = next(iter(_raw.values()), {}) if isinstance(G, nx.MultiGraph) else _raw
+        lines.append(f'  EDGE {G.nodes[u].get(\"label\",u)} --{d.get(\"relation\",\"\")} [{d.get(\"confidence\",\"\")}]--> {G.nodes[v].get(\"label\",v)}')
+
+output = '\n'.join(lines)
+if len(output) > char_budget:
+    output = output[:char_budget] + f'\n... (truncated at ~{token_budget} token budget - use --budget N for more)'
+print(output)
+"
+```
+
+Replace `QUESTION` with the **expanded** query string, `MODE` with `bfs` or `dfs`, and `BUDGET` with the token budget (default `2000`, or whatever `--budget N` specifies). Then answer based on the subgraph output above, using only what the graph contains.
+
+After writing the answer, save it back into the graph so it improves future queries. Include the expanded tokens inside the `--answer` text (e.g. `"Expanded from original query via vocab: [tokens]. Then traversed..."`) so the next `--update` extracts the expansion history as a graph node:
+
+```bash
+$(cat graphify-out/.graphify_python) -m graphify save-result --question "ORIGINAL_QUESTION" --answer "ANSWER" --type query --nodes NODE1 NODE2
+```
+
+Replace `ORIGINAL_QUESTION` with the user's verbatim question, `ANSWER` with your full answer text (containing the expanded-token trace), `NODE1 NODE2` with the list of node labels you cited. This closes the feedback loop: the next `--update` will extract this Q&A as a node in the graph.
+
+**Work memory (self-improving loop).** Add an `--outcome` so future sessions learn from this one — append `--outcome useful|dead_end|corrected` to the `save-result` command (and `--correction "the right answer"` when correcting):
+
+- `useful` — the cited nodes answered the question well (they become *preferred sources*).
+- `dead_end` — the question/path led nowhere; don't re-derive it next time.
+- `corrected` — the saved answer was wrong; `--correction` records what was right.
+
+At the **start** of graph work, refresh and read the lessons: run `graphify reflect --if-stale` (cheap, deterministic, no LLM; `--if-stale` makes it a no-op when `LESSONS.md` is already newer than every input, e.g. when the git hook just refreshed it), then read `graphify-out/reflections/LESSONS.md`. It lists **preferred sources** (start there), **known dead ends** (skip them), and prior **corrections**. Running `reflect` yourself keeps the lessons current even without the git hook installed; if the post-commit hook *is* installed, `--if-stale` means your session-start run costs almost nothing.
+
+---
+
+## For /graphify path
+
+Find the shortest path between two named concepts in the graph. Prefer the CLI when installed:
+
+```bash
+graphify path "NODE_A" "NODE_B"
+```
+
+If the CLI is unavailable, run it inline:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json, sys
+import networkx as nx
+from networkx.readwrite import json_graph
+from pathlib import Path
+
+data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
+G = json_graph.node_link_graph(data, edges='links')
+
+a_term = 'NODE_A'
+b_term = 'NODE_B'
+
+def find_node(term):
+    term = term.lower()
+    scored = sorted(
+        [(sum(1 for w in term.split() if w in G.nodes[n].get('label','').lower()), n)
+         for n in G.nodes()],
+        reverse=True
+    )
+    return scored[0][1] if scored and scored[0][0] > 0 else None
+
+src = find_node(a_term)
+tgt = find_node(b_term)
+
+if not src or not tgt:
+    print(f'Could not find nodes matching: {a_term!r} or {b_term!r}')
+    sys.exit(0)
+
+try:
+    path = nx.shortest_path(G, src, tgt)
+    print(f'Shortest path ({len(path)-1} hops):')
+    for i, nid in enumerate(path):
+        label = G.nodes[nid].get('label', nid)
+        if i < len(path) - 1:
+            _raw = G[nid][path[i+1]]; edge = next(iter(_raw.values()), {}) if isinstance(G, nx.MultiGraph) else _raw
+            rel = edge.get('relation', '')
+            conf = edge.get('confidence', '')
+            print(f'  {label} --{rel}--> [{conf}]')
+        else:
+            print(f'  {label}')
+except nx.NetworkXNoPath:
+    print(f'No path found between {a_term!r} and {b_term!r}')
+except nx.NodeNotFound as e:
+    print(f'Node not found: {e}')
+"
+```
+
+Replace `NODE_A` and `NODE_B` with the actual concept names from the user. Then explain the path in plain language - what each hop means, why it's significant.
+
+After writing the explanation, save it back:
+
+```bash
+$(cat graphify-out/.graphify_python) -m graphify save-result --question "Path from NODE_A to NODE_B" --answer "ANSWER" --type path_query --nodes NODE_A NODE_B
+```
+
+---
+
+## For /graphify explain
+
+Give a plain-language explanation of a single node - everything connected to it. Prefer the CLI when installed:
+
+```bash
+graphify explain "NODE_NAME"
+```
+
+If the CLI is unavailable, run it inline:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json, sys
+import networkx as nx
+from networkx.readwrite import json_graph
+from pathlib import Path
+
+data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
+G = json_graph.node_link_graph(data, edges='links')
+
+term = 'NODE_NAME'
+term_lower = term.lower()
+
+# Find best matching node
+scored = sorted(
+    [(sum(1 for w in term_lower.split() if w in G.nodes[n].get('label','').lower()), n)
+     for n in G.nodes()],
+    reverse=True
+)
+if not scored or scored[0][0] == 0:
+    print(f'No node matching {term!r}')
+    sys.exit(0)
+
+nid = scored[0][1]
+data_n = G.nodes[nid]
+print(f'NODE: {data_n.get(\"label\", nid)}')
+print(f'  source: {data_n.get(\"source_file\",\"unknown\")}')
+print(f'  type: {data_n.get(\"file_type\",\"unknown\")}')
+print(f'  degree: {G.degree(nid)}')
+print()
+print('CONNECTIONS:')
+for neighbor in G.neighbors(nid):
+    _raw = G[nid][neighbor]; edge = next(iter(_raw.values()), {}) if isinstance(G, nx.MultiGraph) else _raw
+    nlabel = G.nodes[neighbor].get('label', neighbor)
+    rel = edge.get('relation', '')
+    conf = edge.get('confidence', '')
+    src_file = G.nodes[neighbor].get('source_file', '')
+    print(f'  --{rel}--> {nlabel} [{conf}] ({src_file})')
+"
+```
+
+Replace `NODE_NAME` with the concept the user asked about. Then write a 3-5 sentence explanation of what this node is, what it connects to, and why those connections are significant. Use the source locations as citations.
+
+After writing the explanation, save it back:
+
+```bash
+$(cat graphify-out/.graphify_python) -m graphify save-result --question "Explain NODE_NAME" --answer "ANSWER" --type explain --nodes NODE_NAME
+```

+ 52 - 0
.claude/skills/graphify/references/transcribe.md

@@ -0,0 +1,52 @@
+# graphify reference: transcribe video and audio
+
+Load this only when `detect` reported one or more `video` files. A corpus with no video never reads this.
+
+### Step 2.5 - Transcribe video / audio files (only if video files detected)
+
+Skip this step entirely if `detect` returned zero `video` files.
+
+Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
+
+**Strategy:** Read the god nodes from `graphify-out/.graphify_detect.json` (or the analysis file if it exists from a previous run). You are already a language model — write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
+
+**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
+
+**Step 1 - Write the Whisper prompt yourself.**
+
+Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
+
+- Labels: `transformer, attention, encoder, decoder` → `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
+- Labels: `kubernetes, deployment, pod, helm` → `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
+
+**Export** it as `GRAPHIFY_WHISPER_PROMPT` (the exact name the transcriber reads — and it must be `export`ed so the child Python process sees it) for the next command.
+
+**Step 2 - Transcribe:**
+
+```bash
+export GRAPHIFY_WHISPER_MODEL=base  # or whatever --whisper-model the user passed (must be exported)
+export GRAPHIFY_WHISPER_PROMPT="<the one-sentence domain hint you composed in Step 1>"
+$(cat graphify-out/.graphify_python) -c "
+import json, os, sys
+from pathlib import Path
+from graphify.transcribe import transcribe_all
+
+detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
+video_files = detect.get('files', {}).get('video', [])
+prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
+
+transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
+# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper
+# print progress to stdout, which would otherwise corrupt the JSON file (#1392).
+Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\")
+print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr)
+"
+```
+
+After transcription:
+- Read the transcript paths from `graphify-out/.graphify_transcripts.json`
+- Add them to the docs list before dispatching semantic subagents in Step 3B
+- Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs`
+- If transcription fails for a file, print a warning and continue with the rest
+
+**Whisper model:** Default is `base`. If the user passed `--whisper-model <name>`, `export GRAPHIFY_WHISPER_MODEL=<name>` (it must be exported, not just assigned) before running the command above.

+ 210 - 0
.claude/skills/graphify/references/update.md

@@ -0,0 +1,210 @@
+# graphify reference: incremental update and cluster-only
+
+Load this only when the user passed `--update` or `--cluster-only`. A first-time full build never reads this file.
+
+## For --update (incremental re-extraction)
+
+Use when you've added or modified files since the last run. Only re-extracts changed files - saves tokens and time.
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import sys, json
+from graphify.detect import detect_incremental, save_manifest
+from pathlib import Path
+
+result = detect_incremental(Path('INPUT_PATH'))
+new_total = result.get('new_total', 0)
+print(json.dumps(result, indent=2, ensure_ascii=False))
+Path('graphify-out/.graphify_incremental.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\")
+deleted = list(result.get('deleted_files', []))
+if new_total == 0 and not deleted:
+    print('No files changed since last run. Nothing to update.')
+    raise SystemExit(0)
+if deleted:
+    print(f'{len(deleted)} deleted file(s) to prune.')
+if new_total > 0:
+    print(f'{new_total} new/changed file(s) to re-extract.')
+"
+```
+
+Then populate `.graphify_detect.json` so Steps 3A–6 (which read it unconditionally) see the right state for an incremental run. `files` carries the changed subset (drives Step 3A AST + Step 3B0 cache check on only what changed); `all_files` carries the full corpus for any step that needs corpus-wide context:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+r = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\"))
+Path('graphify-out/.graphify_detect.json').write_text(json.dumps({
+    'files': r.get('new_files', {}),
+    'all_files': r.get('files', {}),
+    'total_files': r.get('new_total', 0),
+    'total_words': r.get('total_words', 0),
+    'skipped_sensitive': r.get('skipped_sensitive', []),
+    'needs_graph': True,
+}, ensure_ascii=False), encoding=\"utf-8\")
+"
+```
+
+If new files exist, first check whether all changed files are code files:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+
+result = json.loads(open('graphify-out/.graphify_incremental.json', encoding='utf-8').read()) if Path('graphify-out/.graphify_incremental.json').exists() else {}
+code_exts = {'.py','.ts','.js','.go','.rs','.java','.cpp','.c','.rb','.swift','.kt','.cs','.scala','.php','.cc','.cxx','.hpp','.h','.kts','.lua','.toc','.f','.F','.f90','.F90','.f95','.F95','.f03','.F03','.f08','.F08'}
+new_files = result.get('new_files', {})
+all_changed = [f for files in new_files.values() for f in files]
+code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed)
+print('code_only:', code_only)
+"
+```
+
+If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8.
+
+If `code_only` is False (any changed file is a doc/paper/image/video): **first, if any changed file is in `new_files['video']`, run `references/transcribe.md` (Step 2.5) on those files, then rewrite `.graphify_detect.json` to move the resulting transcript paths into `files['document']` and drop `files['video']`** — otherwise raw `.mp4/.mp3` paths are fed to semantic subagents as unreadable media (#1392). Then run the full Steps 3A–3C pipeline as normal.
+
+
+If no new files exist (only deletions), create an empty extraction so the merge step can prune:
+
+```bash
+if [ ! -f graphify-out/.graphify_extract.json ]; then
+    echo '[graphify update] Only deletions -- creating empty extraction for merge.'
+    $(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8')
+"
+fi
+```
+
+
+Then:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from pathlib import Path
+from graphify.build import build_merge
+from graphify.detect import save_manifest
+
+# Load new extraction and incremental state
+new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
+incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\"))
+deleted = list(incremental.get('deleted_files', []))
+# prune_sources is ONLY for genuinely DELETED files. Changed/re-extracted files are
+# handled by build_merge's replace-on-re-extract (#1344): every source_file in
+# new_chunks is dropped from the base before merge, so old/stale nodes don't survive.
+# Do NOT add `changed` here: with root= passed, prune_set relativizes to the same base
+# as the freshly merged nodes and would DELETE the re-extracted content (#1178 is moot
+# now that replace — not the dedup pass — reconciles changed files).
+prune = list(deleted) or None
+
+# Use build_merge() — reads graph.json directly without NetworkX round-trip
+# so edge direction (calls, implements, imports) is always preserved (#801).
+# Pass root= so prune_sources (absolute paths from detect_incremental) are
+# relativized to match the graph's relative source_file values; without it
+# nothing is pruned and stale nodes accumulate on every update (#1361).
+# directed=IS_DIRECTED: replace IS_DIRECTED with True if --directed was given, else
+# False. Without it a --directed --update silently rebuilds undirected and collapses
+# reciprocal A<->B edges (#1392).
+G = build_merge(
+    [new_extraction],
+    graph_path='graphify-out/graph.json',
+    prune_sources=prune,
+    root='INPUT_PATH',
+    directed=IS_DIRECTED,
+)
+print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges')
+
+# Write merged result back to .graphify_extract.json so Step 4 sees the full graph
+merged_out = {
+    'nodes': [{'id': n, **d} for n, d in G.nodes(data=True)],
+    'edges': [
+        # Explicit source/target last so they win over any stale attrs in d.
+        {**{k: val for k, val in d.items() if k not in ('_src', '_tgt', 'source', 'target')},
+         'source': d.get('_src', u), 'target': d.get('_tgt', v)}
+        for u, v, d in G.edges(data=True)
+    ],
+    # G.graph["hyperedges"] holds hyperedges from both existing graph.json
+    # and new_extraction (build_merge combines them). Falling back to
+    # new_extraction only would silently drop prior-run hyperedges (#801).
+    'hyperedges': list(G.graph.get('hyperedges', [])),
+    'input_tokens': new_extraction.get('input_tokens', 0),
+    'output_tokens': new_extraction.get('output_tokens', 0),
+}
+Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged_out, ensure_ascii=False), encoding=\"utf-8\")
+print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])} nodes, {len(merged_out[\"edges\"])} edges)')
+
+# Save manifest so next --update diffs against today's state, not the
+# prior run's baseline (prevents ghost-node reports on subsequent updates).
+# root= matches the build_merge call above so the manifest keys stay relative to
+# the scan root — portable across clones/machines, so --update keeps matching
+# cached files instead of missing every one after a move (#1417).
+#
+# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
+# THIS run (new_extraction is this run's fresh extraction, read above before the
+# merge overwrote the file): a changed doc whose chunk failed must stay unstamped
+# so the next --update re-queues it, otherwise it is marked done and its content
+# is lost forever (#2015). Mirrors the library extract path
+# (cli._stamped_manifest_files + clear_semantic + scan_corpus).
+from graphify.cli import _stamped_manifest_files
+_manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
+# Changed semantic files dispatched this run but NOT stamped had their chunk fail
+# or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
+_sem_types = ('document', 'paper', 'image')
+_dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
+_stamped = {f for fl in _manifest_files.values() for f in fl}
+_cleared = _dispatched - _stamped
+# scan_corpus = the RAW full corpus so in-root files newly excluded since last run
+# are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
+_scan = {f for fl in incremental['files'].values() for f in fl}
+save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
+print('[graphify update] Manifest saved.')
+"
+```
+
+Then run Steps 4–8 on the merged graph as normal.
+
+After Step 4, show the graph diff:
+
+```bash
+$(cat graphify-out/.graphify_python) -c "
+import json
+from graphify.analyze import graph_diff
+from graphify.build import build_from_json
+from networkx.readwrite import json_graph
+import networkx as nx
+from pathlib import Path
+
+# Load old graph (before update) from backup written before merge
+old_data = json.loads(Path('graphify-out/.graphify_old.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_old.json').exists() else None
+new_extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
+G_new = build_from_json(new_extract, directed=IS_DIRECTED)
+
+if old_data:
+    G_old = json_graph.node_link_graph(old_data, edges='links')
+    diff = graph_diff(G_old, G_new)
+    print(diff['summary'])
+    if diff['new_nodes']:
+        print('New nodes:', ', '.join(n['label'] for n in diff['new_nodes'][:5]))
+    if diff['new_edges']:
+        print('New edges:', len(diff['new_edges']))
+"
+```
+
+Before the merge step, save the old graph: `cp graphify-out/graph.json graphify-out/.graphify_old.json`
+Clean up after: `rm -f graphify-out/.graphify_old.json`
+
+---
+
+## For --cluster-only
+
+Skip Steps 1–3. Re-run clustering on the existing graph:
+
+```bash
+graphify cluster-only .
+```
+
+`graphify cluster-only .` is **self-contained**: it re-clusters, names communities, and regenerates `GRAPH_REPORT.md`, `graph.json`, and `graph.html` from the existing graph. **Do not re-run Steps 5–9** — they read intermediate files (`.graphify_extract.json`, `.graphify_detect.json`, `.graphify_analysis.json`) that a prior build's cleanup (Step 9) already deleted, so they raise `FileNotFoundError` (#1392). When it finishes, present the refreshed `GRAPH_REPORT.md` summary as usual.

+ 6 - 0
.opencode/opencode.json

@@ -0,0 +1,6 @@
+{
+  "$schema": "https://opencode.ai/config.json",
+  "plugin": [
+    ".opencode/plugins/graphify.js"
+  ]
+}

+ 30 - 0
.opencode/plugins/graphify.js

@@ -0,0 +1,30 @@
+// graphify OpenCode plugin
+// Injects a knowledge graph reminder before bash tool calls when the graph exists.
+//
+// IMPORTANT: keep the reminder string free of backticks and $(...) constructs.
+// The hook prepends `echo "<reminder>" && <cmd>` to the user's bash command;
+// backticks inside the double-quoted echo trigger bash command substitution,
+// which both corrupts tool output and silently executes the very graphify
+// command we are only suggesting. Plain words render fine in opencode's TUI.
+import { existsSync } from "fs";
+import { join } from "path";
+
+export const GraphifyPlugin = async ({ directory }) => {
+  let reminded = false;
+
+  return {
+    "tool.execute.before": async (input, output) => {
+      if (reminded) return;
+      if (!existsSync(join(directory, "graphify-out", "graph.json"))) return;
+
+      if (input.tool === "bash") {
+        // ';' not '&&' — Windows PowerShell 5.1 rejects '&&' as a statement
+        // separator, breaking the first bash command of the session (#1646).
+        output.args.command =
+          'echo "[graphify] knowledge graph at graphify-out/. For focused questions, run graphify query with your question (scoped subgraph, usually much smaller than GRAPH_REPORT.md) instead of grepping raw files. Read GRAPH_REPORT.md only for broad architecture context." ; ' +
+          output.args.command;
+        reminded = true;
+      }
+    },
+  };
+};

+ 130 - 0
AGENTS.md

@@ -0,0 +1,130 @@
+# Agent Guidelines — Wobkey Impact 80 RGB CLI
+
+## Objective
+Cross-platform Go CLI library for programmatic/agent-friendly control of VIA-compatible keyboard RGB lighting (starting with Wobkey Impact 80).
+
+## Architecture
+
+```
+cmd/                    # Cobra-based CLI entrypoints
+internal/device/        # HID device discovery (VID/PID matching)
+internal/via/           # VIA protocol implementation + LED subsystem
+internal/rgb/           # Effects, values, color definitions
+keyboards.json          # VID/PID database + keyboard metadata
+```
+
+## CLI Interface
+
+```
+wobkey keyboard info              # Discover connected VIA-compatible keyboards
+wobkey rgb effect <name>          # Set RGB effect (e.g. "solid", "breathing", "wave")
+wobkey rgb brightness <val>       # Set brightness (0-255)
+wobkey rgb speed <val>            # Set effect speed (0-255)
+wobkey rgb color <hex>            # Set solid color (e.g. "ff0000")
+wobkey rgb mode <index>           # Set mode by index
+wobkey rgb enable                 # Enable RGB lighting
+wobkey rgb disable                # Disable RGB lighting
+wobkey rgb info                   # Show current RGB state
+```
+
+## VIA Protocol Essentials
+
+- Transport: HID reports, `Report ID 0x52`
+- Message types:
+  - `0x01` — Enable/Handshake
+  - `0x02` — Set Value
+  - `0x03` — Get Value
+- RGB runs via QMK `rgblight` subsystem (`LED-Type 0x01`)
+- No existing Go package for VIA/QMK — must implement minimal protocol
+
+## Keyboards (Known Wobkey VID/PID)
+
+| Keyboard         | VID    | PID    |
+|------------------|--------|--------|
+| Wobkey Rainy 75  | 0x6666 | 0x0001 |
+| Wobkey Impact 80 | TBD    | TBD    |
+
+`keyboards.json` maps VID+PID → VIA protocol config + RGB layout metadata.
+
+## Stack
+
+- Go 1.26+ (cross-platform: Linux, macOS, Windows)
+- `github.com/sstallion/go-hid` for HID access
+- `github.com/spf13/cobra` for CLI framework
+
+## Constraints
+
+- CLI-first, no GUI — designed for automation and agent consumption
+- Output should be machine-parseable (JSON when possible)
+- Cross-platform from day one
+
+## Go Development
+
+### Tooling
+
+- `gopls` — Go Language Server, built into Go toolchain (LSP, diagnostics, go-to-def, rename, references)
+- `gofmt` / `goimports` — formatting. `goimports` adds/removes imports automatically. Always run before committing.
+- `go test` — built-in test framework. Tests live in `*_test.go` files alongside source.
+- `go vet` — static analysis. Run before committing.
+- `go build` — compiles without installing. Fast, cached.
+- `go mod tidy` — adds missing deps, removes unused ones. Run after every import change.
+
+### Style (Effective Go + Code Review Comments)
+
+- **Formatting**: always `gofmt` or `goimports`. Tabs for indentation, no line-length limit but avoid uncomfortably long lines.
+- **Names**: `MixedCaps` / `mixedCaps`, no underscores. Short local variables (`i`, `c`, `r`). Descriptive for globals.
+- **Initialisms**: consistent case — `URL`, `ID`, `HTTP` → `appID`, `urlPony`, `ServeHTTP`.
+- **Comments**: doc comments start with the name, end with period. `// Package rgb provides RGB effect handling.`
+- **Imports**: standard library first, blank line, then third-party. Grouped.
+
+### Error Handling
+
+- Return errors, never `panic` for normal control flow.
+- Indent error flow early — happy path at minimal indentation:
+  ```go
+  if err != nil {
+      return err
+  }
+  // normal code
+  ```
+- Error strings: lowercase, no period — `fmt.Errorf("device not found")`.
+- Do not discard errors with `_`.
+
+### Interfaces & Types
+
+- Define interfaces in the **consumer** package, not the producer.
+- Use pointer receivers when in doubt (mutation, large structs, sync fields).
+- Return concrete types from constructors — let consumers mock if needed.
+- Empty slices: `var t []string` (nil), not `t := []string{}` (non-nil), unless JSON encoding requires `[]`.
+
+### Concurrency
+
+- Pass `context.Context` as **first parameter**. Never store in structs.
+- Clear goroutine lifetimes — document when/why they exit. Avoid leaks via channels.
+- Prefer synchronous functions — callers can add concurrency, not remove it.
+
+### Testing
+
+- Table-driven tests preferred for multiple cases.
+- Failure messages: `t.Errorf("Effect(%q) = %d, want %d", input, got, want)`.
+- Use `*_test.go` files. Test package should match source package (internal tests).
+- Example functions (`func Example...`) double as docs and tests.
+
+### Modules & Packages
+
+- Module path: `github.com/<user>/<project>` (when published).
+- Package names: short, single-word, lowercase. No `util`, `common`, `api`, `types`.
+- Zero values should be useful (e.g. `bytes.Buffer`, `sync.Mutex`).
+
+## graphify
+
+This project has a knowledge graph at graphify-out/ with god nodes, community structure, and cross-file relationships.
+
+When the user types `/graphify`, use the installed graphify skill or instructions before doing anything else.
+
+Rules:
+- For codebase questions, first run `graphify query "<question>"` when graphify-out/graph.json exists. Use `graphify path "<A>" "<B>"` for relationships and `graphify explain "<concept>"` for focused concepts. These return a scoped subgraph, usually much smaller than GRAPH_REPORT.md or raw grep output.
+- Dirty graphify-out/ files are expected after hooks or incremental updates; dirty graph files are not a reason to skip graphify. Only skip graphify if the task is about stale or incorrect graph output, or the user explicitly says not to use it.
+- If graphify-out/wiki/index.md exists, use it for broad navigation instead of raw source browsing.
+- Read graphify-out/GRAPH_REPORT.md only for broad architecture review or when query/path/explain do not surface enough context.
+- After modifying code, run `graphify update .` to keep the graph current (AST-only, no API cost).

+ 9 - 0
CLAUDE.md

@@ -0,0 +1,9 @@
+## graphify
+
+This project has a knowledge graph at graphify-out/ with god nodes, community structure, and cross-file relationships.
+
+Rules:
+- For codebase questions, first run `graphify query "<question>"` when graphify-out/graph.json exists. Use `graphify path "<A>" "<B>"` for relationships and `graphify explain "<concept>"` for focused concepts. These return a scoped subgraph, usually much smaller than GRAPH_REPORT.md or raw grep output.
+- If graphify-out/wiki/index.md exists, use it for broad navigation instead of raw source browsing.
+- Read graphify-out/GRAPH_REPORT.md only for broad architecture review or when query/path/explain do not surface enough context.
+- After modifying code, run `graphify update .` to keep the graph current (AST-only, no API cost).

+ 268 - 0
LICENSE

@@ -0,0 +1,268 @@
+GNU General Public License
+==========================
+
+_Version 3, 29 June 2007_
+_Copyright © Free Software Foundation, Inc. &lt;http://fsf.org/&gt;_
+
+Everyone is permitted to copy and distribute verbatim copies of this license
+document, but changing it is not allowed.
+
+## Preamble
+
+The GNU General Public License is a free, copyleft license for software and other
+kinds of works.
+
+The licenses for most software and other practical works are designed to take away
+your freedom to share and change them. By contrast, the GNU General Public License
+is intended to guarantee your freedom to share and change all versions of a
+program—to make sure it remains free software for all its users. We, the Free
+Software Foundation, use the GNU General Public License for most of our software;
+it applies also to any other work released this way by its authors. You can apply
+it to your programs, too.
+
+When we speak of free software, we are referring to freedom, not price. Our General
+Public Licenses are designed to make sure that you have the freedom to distribute
+copies of free software (and charge for them if you wish), that you receive source
+code or can get it if you want it, that you can change the software or use pieces
+of it in new free programs, and that you know you can do these things.
+
+To protect your rights, we need to prevent others from denying you these rights or
+asking you to surrender the rights. Therefore, you have certain responsibilities if
+you distribute copies of the software, or if you modify it: responsibilities to
+respect the freedom of others.
+
+For example, if you distribute copies of such a program, whether gratis or for a fee,
+you must pass on to the recipients the same freedoms that you received. You must make
+sure that they, too, receive or can get the source code. And you must show them these
+terms so they know their rights.
+
+Developers that use the GNU GPL protect your rights with two steps: **(1)** assert
+copyright on your software, and **(2)** offer you this License giving you legal
+permission to copy, distribute and/or modify it.
+
+For the developers' and authors' protection the GNU GPL clearly explains that there
+is no warranty for this free software. For both users' and authors' sake, the GNU GPL
+requires that modified versions be marked as changed, so that their problems will not
+be attributed erroneously to authors of previous versions.
+
+Some devices are designed to deny users access to install or run modified versions of
+the software inside them, although the manufacturer can do so. This is fundamentally
+incompatible with the aim of protecting users' freedom to change the software. The
+systematic pattern of such abuse occurs in the area of products for individuals to use,
+which is precisely where it is most unacceptable. Therefore, we have designed this
+version of the GNU GPL to prohibit the practice for those products. If such problems
+arise substantially in other domains, we stand ready to extend this provision to those
+domains in future versions of the GNU GPL, as needed to protect the freedom of users.
+
+Finally, every program is threatened constantly by software patents. States should not
+allow patents to restrict development and use of software on general-purpose computers,
+but in those that do, we wish to avoid the special danger that patents applied to a free
+program could make it effectively proprietary. To prevent this, the GNU GPL assures that
+patents cannot be used to render the program non-free.
+
+The full text of the GNU General Public License version 3 follows below.
+
+## TERMS AND CONDITIONS
+
+### 0. Definitions
+
+"This License" refers to version 3 of the GNU General Public License.
+
+"Copyright" also means copyright-like laws that apply to other kinds of works, such as
+semiconductor masks.
+
+"The Program" refers to any copyrightable work licensed under this License. Each licensee
+is addressed as "you". "Licensees" and "recipients" may be individuals or organizations.
+
+To "modify" a work means to copy from or adapt all or part of the work in a fashion
+requiring copyright permission, other than the making of an exact copy. The resulting work
+is called a "modified version" of the earlier work or a work "based on" the earlier work.
+
+A "covered work" means either the unmodified Program or a work based on the Program.
+
+To "propagate" a work means to do anything with it that, without permission, would make
+you directly or secondarily liable for infringement under applicable copyright law, except
+executing it on a computer or modifying a private copy. Propagation includes copying,
+distribution (with or without modification), making available to the public, and in some
+countries other activities as well.
+
+To "convey" a work means any kind of propagation that enables other parties to make or
+receive copies. Mere interaction with a user through a computer network, with no transfer
+of a copy, is not conveying.
+
+An interactive user interface displays "Appropriate Legal Notices" to the extent that it
+includes a convenient and prominently visible feature that **(1)** displays an appropriate
+copyright notice, and **(2)** tells the user that there is no warranty for the work (except
+to the extent that warranties are provided), that licensees may convey the work under this
+License, and how to view a copy of this License. If the interface presents a list of user
+commands or options, such as a menu, a prominent item in the list meets this criterion.
+
+### 1. Source Code
+
+The "source code" for a work means the preferred form of the work for making modifications
+to it. "Object code" means any non-source form of a work.
+
+A "Standard Interface" means an interface that either is an official standard defined by a
+recognized standards body, or, in the case of interfaces specified for a particular
+programming language, one that is widely used among developers working in that language.
+
+The "System Libraries" of an executable work include anything, other than the Program, that
+serves only to enable use of the Program with a major component, or to implement a Standard
+Interface for which an implementation is available to the public in source code form. A
+"Major Component", in this context, means a major essential component (kernel, window
+system, and so on) of the specific operating system (if any) on which the executable work
+runs, or a compiler used to produce the work, or an object code interpreter used to run it.
+
+The "Corresponding Source" for a work in object code form means all the source code needed to
+generate, install, and (for an executable work) run the object code and to modify the work,
+including scripts to control those activities. However, it does not include the work's System
+Libraries, or general-purpose tools or generally available free programs which are used
+unmodified in performing those activities but which are not part of the work.
+
+### 2. Basic Permissions
+
+All rights granted under this License are granted for the term of copyright on the Program,
+and are irrevocable provided the stated conditions are met. This License explicitly affirms
+your unlimited permission to run the unmodified Program. The output from running a covered
+work is covered by this License only if the output, given its content, constitutes a covered
+work.
+
+You may convey verbatim copies of the Program's source code as you receive it, in any medium,
+provided that you conspicuously and appropriately publish on each copy an appropriate
+copyright notice; keep intact all notices stating that this License and any non-permissive
+terms added in accord with section 7 apply to the code; keep intact all notices of the
+absence of any warranty; and give all recipients a copy of this License along with the
+Program.
+
+You may charge any price or no price for each copy that you convey, and you may offer support
+or warranty protection for a fee.
+
+### 3. Protecting Users' Legal Rights From Anti-Circumvention Law
+
+No covered work shall be deemed part of an effective technological measure under any
+applicable law fulfilling obligations under article 11 of the WIPO copyright treaty adopted
+on 20 December 1996, or similar laws prohibiting or restricting circumvention of such
+measures.
+
+### 4. Conveying Verbatim Copies
+
+You may convey verbatim copies of the Program's source code as you receive it, in any medium,
+provided that you comply with all conditions in this section.
+
+### 5. Conveying Modified Source Versions
+
+You may convey a work based on the Program, or the modifications to produce it from the
+Program, in the form of source code under the terms of section 4, provided that you also meet
+all of these conditions:
+
+* **a)** The work must carry prominent notices stating that you modified it, and giving a
+  relevant date.
+* **b)** The work must carry prominent notices stating that it is released under this License
+  and any conditions added under section 7.
+* **c)** You must license the entire work, as a whole, under this License to anyone who comes
+  into possession of a copy.
+* **d)** If the work has interactive user interfaces, each must display Appropriate Legal
+  Notices.
+
+### 6. Conveying Non-Source Forms
+
+You may convey a covered work in object code form under the terms of sections 4 and 5,
+provided that you also convey the machine-readable Corresponding Source.
+
+### 7. Additional Terms
+
+"Additional permissions" are terms that supplement the terms of this License by making
+exceptions from one or more of its conditions. Additional permissions that are applicable to
+the entire Program shall be treated as though they were included in this License.
+
+### 8. Termination
+
+You may not propagate or modify a covered work except as expressly provided under this
+License. Any attempt otherwise to propagate or modify it is void, and will automatically
+terminate your rights under this License.
+
+### 9. Acceptance Not Required for Having Copies
+
+You are not required to accept this License in order to receive or run a copy of the Program.
+Ancillary propagation of a covered work occurring solely as a consequence of using
+peer-to-peer transmission to receive a copy likewise does not require acceptance.
+
+### 10. Automatic Licensing of Downstream Recipients
+
+Each time you convey a covered work, the recipient automatically receives a license from the
+original licensors, to run, modify and propagate that work, subject to this License.
+
+### 11. Patents
+
+A "contributor" is a copyright holder who authorizes use under this License of the Program or
+a work on which the Program is based. The work thus licensed is called the contributor's
+"contributor version".
+
+A contributor's "essential patent claims" are all patent claims owned or controlled by the
+contributor, whether already acquired or hereafter acquired, that would be infringed by some
+manner, permitted by this License, of making, using, or selling its contributor version.
+
+Each contributor grants you a non-exclusive, worldwide, royalty-free patent license under the
+contributor's essential patent claims.
+
+### 12. No Surrender of Others' Freedom
+
+If conditions are imposed on you (whether by court order, agreement or otherwise) that
+contradict the conditions of this License, they do not excuse you from the conditions of this
+License.
+
+### 13. Use with the GNU Affero General Public License
+
+You may combine or propagate a covered work with a work licensed under version 3 of the GNU
+Affero General Public License into a single combined work, and to convey the resulting work.
+
+### 14. Revised Versions of this License
+
+The Free Software Foundation may publish revised and/or new versions of the GNU General Public
+License from time to time.
+
+If the Program specifies that a certain numbered version of the GNU General Public License "or
+any later version" applies to it, you have the option of following the terms and conditions
+either of that numbered version or of any later version published by the Free Software
+Foundation.
+
+### 15. Disclaimer of Warranty
+
+THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY APPLICABLE LAW. EXCEPT WHEN
+OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS
+IS" WITHOUT WARRANTY OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO,
+THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE. THE ENTIRE RISK
+AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM IS WITH YOU.
+
+### 16. Limitation of Liability
+
+IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING WILL ANY COPYRIGHT
+HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS THE PROGRAM AS PERMITTED ABOVE, BE
+LIABLE TO YOU FOR DAMAGES, INCLUDING ANY GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES
+ARISING OUT OF THE USE OR INABILITY TO USE THE PROGRAM.
+
+### 17. Interpretation of Sections 15 and 16
+
+If the disclaimer of warranty and limitation of liability provided above cannot be given local
+legal effect according to their terms, reviewing courts shall apply local law that most
+closely approximates an absolute waiver of all civil liability in connection with the Program.
+
+END OF TERMS AND CONDITIONS
+
+## How to apply these terms to your program
+
+    Wobkey RGB CLI — Cross-platform CLI for VIA-compatible keyboard RGB lighting
+    Copyright (C) 2025  Wobkey contributors
+
+    This program is free software: you can redistribute it and/or modify
+    it under the terms of the GNU General Public License as published by
+    the Free Software Foundation, either version 3 of the License, or
+    (at your option) any later version.
+
+    This program is distributed in the hope that it will be useful,
+    but WITHOUT ANY WARRANTY; without even the implied warranty of
+    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+    GNU General Public License for more details.
+
+    You should have received a copy of the GNU General Public License
+    along with this program.  If not, see <https://www.gnu.org/licenses/>.

+ 99 - 0
PLAN.md

@@ -0,0 +1,99 @@
+# Wobkey RGB CLI — Implementation Plan
+
+## Status
+
+- ✅ Go module initialized (`github.com/wobkey/rgb`)
+- ✅ `git init` + AGENTS.md + README.md
+- ✅ Directory structure: `cmd/wobkey/`, `internal/device/`, `internal/hid/`, `internal/rgb/`, `internal/via/`
+- ✅ Pure-Go HID layer: reads `/dev/hidraw*` + parses `/sys/bus/hid/devices/*/uevent` (no cgo, no libudev)
+- ✅ Keyboard discovery: detects Impact 80 (VID `0x36B0`, PID `0x309F`) and Rainy 75
+- ✅ All 8 RGB subcommands implemented (effect, brightness, speed, color, mode, enable, disable, info)
+- ✅ `go build` ✅ `go vet` ✅ `go test ./internal/rgb/...` (18/18 pass) ✅ `gofmt` clean
+- ✅ VIA protocol: Enable handshake + Set/Get commands
+- ✅ `keyboards.json` with decimal VID/PID
+
+## Open Issues
+
+### 1. Linux HID Permissions — BLOCKS ALL FUNCTIONAL TESTING
+
+All `/dev/hidraw*` are `root:root 600`. The tool can't open them as `paul`.
+
+**Action for next session:**
+1. Run: `sudo chown root:adm /dev/hidraw* && sudo chmod 660 /dev/hidraw*`
+2. Or: apply the udev rule from README.md for a permanent fix
+3. Verify: `ls -la /dev/hidraw*` shows group `adm` with `rw` for group
+4. Run: `/tmp/wobkey keyboard info` — should output JSON with the Impact 80
+
+### 2. Test the Keyboard — Needs Step 1 First
+
+Once permissions are fixed:
+1. `/tmp/wobkey keyboard info` — verify Impact 80 is discovered
+2. `/tmp/wobkey rgb disable` — test disable
+3. `/tmp/wobkey rgb enable` — test enable
+4. `/tmp/wobkey rgb info` — verify it returns JSON state
+5. `/tmp/wobkey rgb effect rainbow` — test effect switching
+6. `/tmp/wobkey rgb brightness 200` — test brightness
+7. `/tmp/wobkey rgb color ff0000` — test red color
+
+### 3. Color Command — RGB to HSV Conversion
+
+Current implementation is a hack: it treats R→hue, G→saturation with no actual RGB→HSV conversion.
+
+**Action:**
+- Add `rgbto.hsv()` conversion in `internal/rgb/effects.go`
+- R=red channel → HSV H, G=green → HSV S, B=blue → use as V when setting static
+- This makes `wobkey rgb color ff0000` actually red instead of unpredictable
+
+### 4. QMK rgblight Parameter Numbers
+
+Current code uses hardcoded numbers that need verification against QMK spec:
+- `rgb.Mode` (param 0) ✅
+- `rgb.Brightness` (param 1) ✅  
+- `rgb.Speed` (param 2) ✅
+- `rgb.Enable` (param 3) ✅
+- Hue = 26, Sat = 27 — unverified
+- Effect param = 48, effect_sw = 49, bright_set = 50 — missing
+
+**Action:**
+- Add all QMK rgblight param constants to `internal/rgb/effects.go`
+- Map them properly
+
+### 5. Test Coverage
+
+Currently only `internal/rgb/effects_test.go` has tests (18/18 pass).
+
+**Action:**
+- `internal/device/device_test.go` — test `LoadKeyboards` (parse JSON file)
+- `internal/hid/hid_test.go` — test `parseHex` and `splitLines` helper functions
+- `internal/via/protocol_test.go` — mock-based tests for protocol methods
+
+### 6. Cross-Compile Binaries
+
+The pure-Go HID layer should work on macOS/Windows, but needs verification.
+
+**Action:**
+- `GOOS=darwin GOARCH=amd64 go build -o dist/wobkey-darwin-amd64 ./cmd/wobkey/`
+- `GOOS=darwin GOARCH=arm64 go build -o dist/wobkey-darwin-arm64 ./cmd/wobkey/`
+- `GOOS=windows GOARCH=amd64 go build -o dist/wobkey-windows-amd64.exe ./cmd/wobkey/`
+- The Linux HID layer (`/dev/hidraw*`) won't work on macOS/Windows — need conditional compilation or platform-specific HID backends
+
+## Quick Commands for Next Session
+
+```bash
+# 1. Fix permissions (requires sudo)
+sudo chown root:adm /dev/hidraw* && sudo chmod 660 /dev/hidraw*
+
+# 2. Rebuild with latest code
+go build -o /tmp/wobkey ./cmd/wobkey/
+
+# 3. Test discovery
+/tmp/wobkey keyboard info
+
+# 4. Test RGB commands
+/tmp/wobkey rgb disable
+/tmp/wobkey rgb enable
+/tmp/wobkey rgb info
+
+# 5. Run all tests
+go test ./...
+```

+ 114 - 0
README.md

@@ -0,0 +1,114 @@
+# Wobkey RGB CLI
+
+Cross-platform CLI for programmatic control of VIA-compatible keyboard RGB lighting. Designed for automation, scripting, and agent consumption.
+
+## Installation
+
+```bash
+git clone <repo-url>
+cd wobkey-rgb
+go build -o wobkey ./cmd/wobkey/
+```
+
+Or install globally:
+
+```bash
+go install ./cmd/wobkey/
+```
+
+## Usage
+
+```bash
+# Discover connected keyboards
+./wobkey keyboard info
+
+# RGB commands
+./wobkey rgb enable
+./wobkey rgb disable
+./wobkey rgb info
+./wobkey rgb effect rainbow
+./wobkey rgb effect breathing
+./wobkey rgb brightness 128
+./wobkey rgb speed 100
+./wobkey rgb color ff0000
+./wobkey rgb mode 7
+
+# Specify a target device (when multiple are connected)
+./wobkey rgb --device /dev/hidraw0 enable
+```
+
+All output is machine-parseable JSON when applicable.
+
+## Linux HID Device Permissions
+
+The tool accesses keyboards via `/dev/hidraw*` which requires root permissions by default. To run without `sudo`, choose one of the methods below.
+
+### Option 1: Quick (Per Session)
+
+Grant your user group access to all HIDRAW devices:
+
+```bash
+sudo chown root:adm /dev/hidraw*
+sudo chmod 660 /dev/hidraw*
+```
+
+This works for the current session only. Permissions reset after reboot.
+
+### Option 2: Permanent (udev Rule)
+
+Create a persistent rule for the Wobkey Impact 80:
+
+```bash
+echo 'SUBSYSTEM=="hidraw", ATTRS{idVendor}=="36b0", ATTRS{idProduct}=="309f", MODE="0660", GROUP="adm"' \
+  | sudo tee /etc/udev/rules.d/99-wobkey.rules
+
+sudo udevadm control --reload-rules
+sudo udevadm trigger
+sudo udevadm reload
+```
+
+This makes the Impact 80 accessible to members of the `adm` group on every boot.
+
+After either option, run the tool as your regular user (no `sudo` needed).
+
+## Supported Keyboards
+
+| Keyboard        | VID    | PID    | Status    |
+|-----------------|--------|--------|-----------|
+| Wobkey Rainy 75 | 0x6666 | 0x0001 | Supported |
+| Wobkey Impact 80| 0x36B0 | 0x309F | Supported |
+
+New keyboards can be added to `keyboards.json`.
+
+## CLI Reference
+
+| Command                         | Description                              |
+|---------------------------------|------------------------------------------|
+| `wobkey keyboard info`          | Discover connected VIA-compatible keyboards |
+| `wobkey rgb enable`             | Enable RGB lighting                      |
+| `wobkey rgb disable`            | Disable RGB lighting                     |
+| `wobkey rgb info`               | Show current RGB state (JSON)            |
+| `wobkey rgb effect <name>`      | Set effect (static, breathing, rainbow…) |
+| `wobkey rgb brightness <val>`   | Set brightness (0–255)                   |
+| `wobkey rgb speed <val>`        | Set effect speed (0–255)                 |
+| `wobkey rgb color <hex>`        | Set solid color (e.g. `ff0000`)          |
+| `wobkey rgb mode <index>`       | Set mode by numeric index                |
+
+## Architecture
+
+```
+cmd/wobkey/       # Cobra-based CLI
+cmd/wobkey/rgb/   # RGB subcommands
+internal/device/  # HID discovery + keyboards.json loader
+internal/hid/     # Pure-Go HID access (/dev/hidraw*)
+internal/rgb/     # Effects, colors, state
+internal/via/     # VIA protocol implementation
+```
+
+## Protocol
+
+Communicates via the VIA protocol over HID reports (report ID `0x52`). RGB is controlled through the QMK `rgblight` subsystem.
+
+- `0x01` — Enable/Handshake
+- `0x02` — Set Value
+- `0x03` — Get Value

+ 81 - 0
cmd/wobkey/main.go

@@ -0,0 +1,81 @@
+package main
+
+import (
+	"encoding/json"
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/cmd/wobkey/rgb"
+	"github.com/wobkey/rgb/internal/device"
+)
+
+var rootCmd = &cobra.Command{
+	Use:   "wobkey",
+	Short: "Wobkey RGB CLI — control VIA-compatible keyboard lighting",
+}
+
+func Execute() {
+	if err := rootCmd.Execute(); err != nil {
+		fmt.Fprintln(os.Stderr, err)
+		os.Exit(1)
+	}
+}
+
+func main() {
+	Execute()
+}
+
+var keyboardCmd = &cobra.Command{
+	Use:   "keyboard",
+	Short: "Keyboard management commands",
+}
+
+var keyboardInfoCmd = &cobra.Command{
+	Use:   "info",
+	Short: "Discover connected VIA-compatible keyboards",
+	Long:  "Scan for connected VIA-compatible keyboards and print device info as JSON.",
+	Run: func(cmd *cobra.Command, args []string) {
+		keyboards, err := device.LoadKeyboards()
+		if err != nil {
+			fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+			os.Exit(1)
+		}
+
+		devices, err := device.Discover(keyboards)
+		if err != nil {
+			fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+			os.Exit(1)
+		}
+
+		if len(devices) == 0 {
+			fmt.Println(`{"devices":[]}`)
+			return
+		}
+
+		type Info struct {
+			Devices []device.Device `json:"devices"`
+			Total   int             `json:"total"`
+		}
+		out := Info{Devices: devices, Total: len(devices)}
+		data, _ := json.MarshalIndent(out, "", "  ")
+		fmt.Println(string(data))
+	},
+}
+
+func init() {
+	rootCmd.AddCommand(keyboardCmd)
+	keyboardCmd.AddCommand(keyboardInfoCmd)
+	rgbCmd := rgb.Init()
+	rgbCmd.AddCommand(
+		rgb.NewEffectCmd(),
+		rgb.NewBrightnessCmd(),
+		rgb.NewSpeedCmd(),
+		rgb.NewColorCmd(),
+		rgb.NewModeCmd(),
+		rgb.NewEnableCmd(),
+		rgb.NewDisableCmd(),
+		rgb.NewInfoCmd(),
+	)
+	rootCmd.AddCommand(rgbCmd)
+}

+ 45 - 0
cmd/wobkey/rgb/brightness.go

@@ -0,0 +1,45 @@
+package rgb
+
+import (
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+func NewBrightnessCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "brightness <val>",
+		Short: "Set RGB brightness",
+		Long:  "Set the RGB brightness value (0-255).",
+		Args:  cobra.ExactArgs(1),
+		Run: func(cmd *cobra.Command, args []string) {
+			val, err := ParseUint8(args[0])
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+
+			proto, _, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Brightness), val); err != nil {
+				fmt.Fprintf(os.Stderr, "Error setting brightness: %v\n", err)
+				os.Exit(1)
+			}
+
+			fmt.Printf("Brightness set to %d\n", val)
+		},
+	}
+}

+ 55 - 0
cmd/wobkey/rgb/color.go

@@ -0,0 +1,55 @@
+package rgb
+
+import (
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+func NewColorCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "color <hex>",
+		Short: "Set solid color",
+		Long:  "Set the RGB color using a 6-digit hex code (e.g. \"ff0000\" for red).",
+		Args:  cobra.ExactArgs(1),
+		Run: func(cmd *cobra.Command, args []string) {
+			c, err := rgb.ParseHexColor(args[0])
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+
+			proto, _, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Mode), uint8(rgb.Static)); err != nil {
+				fmt.Fprintf(os.Stderr, "Error setting mode to static: %v\n", err)
+				os.Exit(1)
+			}
+
+			h, s, _ := c.HSV()
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Hue), h); err != nil {
+				fmt.Fprintf(os.Stderr, "Error setting hue: %v\n", err)
+				os.Exit(1)
+			}
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Saturation), s); err != nil {
+				fmt.Fprintf(os.Stderr, "Error setting saturation: %v\n", err)
+				os.Exit(1)
+			}
+
+			fmt.Printf("Color set to %s (static mode)\n", args[0])
+		},
+	}
+}

+ 37 - 0
cmd/wobkey/rgb/disable.go

@@ -0,0 +1,37 @@
+package rgb
+
+import (
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+func NewDisableCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "disable",
+		Short: "Disable RGB lighting",
+		Run: func(cmd *cobra.Command, args []string) {
+			proto, _, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Enable), 0); err != nil {
+				fmt.Fprintf(os.Stderr, "Error disabling RGB: %v\n", err)
+				os.Exit(1)
+			}
+
+			fmt.Println("RGB disabled")
+		},
+	}
+}

+ 72 - 0
cmd/wobkey/rgb/effect.go

@@ -0,0 +1,72 @@
+package rgb
+
+import (
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	intdevice "github.com/wobkey/rgb/internal/device"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+var effectName string
+
+func NewEffectCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "effect <name>",
+		Short: "Set RGB effect",
+		Long:  "Set the RGB lighting effect on the connected keyboard.",
+		Args:  cobra.ExactArgs(1),
+		Run: func(cmd *cobra.Command, args []string) {
+			proto, kb, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			e, err := rgb.ParseEffect(args[0])
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+
+			availEffects := GetAvailableEffects(kb)
+			if availEffects != nil {
+				found := false
+				for _, ef := range availEffects {
+					if ef == e {
+						found = true
+						break
+					}
+				}
+				if !found {
+					fmt.Fprintf(os.Stderr, "Warning: effect %q may not be supported by %s\n", args[0], kb.Name)
+				}
+			}
+
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Mode), uint8(e)); err != nil {
+				fmt.Fprintf(os.Stderr, "Error setting effect: %v\n", err)
+				os.Exit(1)
+			}
+
+			fmt.Printf("Effect set to %q\n", args[0])
+		},
+	}
+}
+
+func GetAvailableEffects(kb intdevice.Keyboard) []rgb.Effect {
+	if kb.Name == "Wobkey Impact 80" {
+		return []rgb.Effect{
+			rgb.Static, rgb.Breathing, rgb.Rainbow, rgb.RainbowWave,
+			rgb.Snake, rgb.Tail, rgb.Gradient,
+		}
+	}
+	return nil
+}

+ 37 - 0
cmd/wobkey/rgb/enable.go

@@ -0,0 +1,37 @@
+package rgb
+
+import (
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+func NewEnableCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "enable",
+		Short: "Enable RGB lighting",
+		Run: func(cmd *cobra.Command, args []string) {
+			proto, _, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Enable), 1); err != nil {
+				fmt.Fprintf(os.Stderr, "Error enabling RGB: %v\n", err)
+				os.Exit(1)
+			}
+
+			fmt.Println("RGB enabled")
+		},
+	}
+}

+ 68 - 0
cmd/wobkey/rgb/info.go

@@ -0,0 +1,68 @@
+package rgb
+
+import (
+	"encoding/json"
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+func NewInfoCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "info",
+		Short: "Show current RGB state",
+		Run: func(cmd *cobra.Command, args []string) {
+			proto, _, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			var state rgb.State
+
+			resp, err := proto.GetValue(via.RGBLight, uint8(rgb.Brightness))
+			if err == nil && len(resp) >= 5 {
+				state.Brightness = uint8(resp[4])
+			}
+
+			resp, err = proto.GetValue(via.RGBLight, uint8(rgb.Speed))
+			if err == nil && len(resp) >= 5 {
+				state.Speed = uint8(resp[4])
+			}
+
+			resp, err = proto.GetValue(via.RGBLight, uint8(rgb.Mode))
+			if err == nil && len(resp) >= 5 {
+				state.Mode = rgb.Effect(resp[4])
+			}
+
+			resp, err = proto.GetValue(via.RGBLight, uint8(rgb.Enable))
+			if err == nil && len(resp) >= 5 {
+				state.Enabled = resp[4] != 0
+			}
+
+			out := struct {
+				Enabled    bool   `json:"enabled"`
+				Mode       string `json:"mode"`
+				Brightness uint8  `json:"brightness"`
+				Speed      uint8  `json:"speed"`
+			}{
+				Enabled:    state.Enabled,
+				Mode:       state.Mode.String(),
+				Brightness: state.Brightness,
+				Speed:      state.Speed,
+			}
+
+			data, _ := json.MarshalIndent(out, "", "  ")
+			fmt.Println(string(data))
+		},
+	}
+}

+ 45 - 0
cmd/wobkey/rgb/mode.go

@@ -0,0 +1,45 @@
+package rgb
+
+import (
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+func NewModeCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "mode <index>",
+		Short: "Set effect mode by index",
+		Long:  "Set the RGB effect mode by numeric index (0-255).",
+		Args:  cobra.ExactArgs(1),
+		Run: func(cmd *cobra.Command, args []string) {
+			val, err := ParseUint8(args[0])
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+
+			proto, _, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Mode), val); err != nil {
+				fmt.Fprintf(os.Stderr, "Error setting mode: %v\n", err)
+				os.Exit(1)
+			}
+
+			fmt.Printf("Mode set to index %d\n", val)
+		},
+	}
+}

+ 69 - 0
cmd/wobkey/rgb/rgb.go

@@ -0,0 +1,69 @@
+package rgb
+
+import (
+	"fmt"
+	"strconv"
+
+	"github.com/spf13/cobra"
+	intdevice "github.com/wobkey/rgb/internal/device"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+var targetDevice string
+
+func OpenDevice() (*via.Protocol, intdevice.Keyboard, error) {
+	keyboards, err := intdevice.LoadKeyboards()
+	if err != nil {
+		return nil, intdevice.Keyboard{}, fmt.Errorf("load keyboards: %w", err)
+	}
+
+	devices, err := intdevice.Discover(keyboards)
+	if err != nil {
+		return nil, intdevice.Keyboard{}, fmt.Errorf("discover: %w", err)
+	}
+
+	if len(devices) == 0 {
+		return nil, intdevice.Keyboard{}, fmt.Errorf("no VIA-compatible keyboard found")
+	}
+
+	var dev intdevice.Device
+	if targetDevice != "" {
+		for _, d := range devices {
+			if d.Path == targetDevice {
+				dev = d
+				break
+			}
+		}
+		if dev.Path == "" {
+			return nil, intdevice.Keyboard{}, fmt.Errorf("device %s not found", targetDevice)
+		}
+	} else if len(devices) == 1 {
+		dev = devices[0]
+	} else {
+		return nil, intdevice.Keyboard{}, fmt.Errorf("multiple devices found, use --device to specify")
+	}
+
+	proto, err := via.New(dev)
+	if err != nil {
+		return nil, intdevice.Keyboard{}, fmt.Errorf("open protocol: %w", err)
+	}
+
+	return proto, dev.Keyboard, nil
+}
+
+func ParseUint8(s string) (uint8, error) {
+	v, err := strconv.ParseUint(s, 10, 8)
+	if err != nil {
+		return 0, fmt.Errorf("invalid value: %w", err)
+	}
+	return uint8(v), nil
+}
+
+func Init() *cobra.Command {
+	cmd := &cobra.Command{
+		Use:   "rgb",
+		Short: "RGB lighting commands",
+	}
+	cmd.Flags().StringVar(&targetDevice, "device", "", "HID device path to use")
+	return cmd
+}

+ 45 - 0
cmd/wobkey/rgb/speed.go

@@ -0,0 +1,45 @@
+package rgb
+
+import (
+	"fmt"
+	"os"
+
+	"github.com/spf13/cobra"
+	"github.com/wobkey/rgb/internal/rgb"
+	"github.com/wobkey/rgb/internal/via"
+)
+
+func NewSpeedCmd() *cobra.Command {
+	return &cobra.Command{
+		Use:   "speed <val>",
+		Short: "Set effect speed",
+		Long:  "Set the RGB effect speed value (0-255).",
+		Args:  cobra.ExactArgs(1),
+		Run: func(cmd *cobra.Command, args []string) {
+			val, err := ParseUint8(args[0])
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+
+			proto, _, err := OpenDevice()
+			if err != nil {
+				fmt.Fprintf(os.Stderr, "Error: %v\n", err)
+				os.Exit(1)
+			}
+			defer proto.Close()
+
+			if err := proto.EnableHandshake(); err != nil {
+				fmt.Fprintf(os.Stderr, "Handshake failed: %v\n", err)
+				os.Exit(1)
+			}
+
+			if err := proto.SetValue(via.RGBLight, uint8(rgb.Speed), val); err != nil {
+				fmt.Fprintf(os.Stderr, "Error setting speed: %v\n", err)
+				os.Exit(1)
+			}
+
+			fmt.Printf("Speed set to %d\n", val)
+		},
+	}
+}

+ 10 - 0
go.mod

@@ -0,0 +1,10 @@
+module github.com/wobkey/rgb
+
+go 1.26.6
+
+require github.com/spf13/cobra v1.10.2
+
+require (
+	github.com/inconshreveable/mousetrap v1.1.0 // indirect
+	github.com/spf13/pflag v1.0.9 // indirect
+)

+ 10 - 0
go.sum

@@ -0,0 +1,10 @@
+github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
+github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8=
+github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw=
+github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
+github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU=
+github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4=
+github.com/spf13/pflag v1.0.9 h1:9exaQaMOCwffKiiiYk6/BndUBv+iRViNW+4lEMi0PvY=
+github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg=
+go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
+gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=

+ 98 - 0
internal/device/device.go

@@ -0,0 +1,98 @@
+package device
+
+import (
+	"encoding/json"
+	"fmt"
+	"os"
+	"path/filepath"
+
+	"github.com/wobkey/rgb/internal/hid"
+)
+
+// Keyboard represents a VIA-compatible keyboard from keyboards.json.
+type Keyboard struct {
+	Name       string `json:"name"`
+	VendorID   uint16 `json:"vendorId"`
+	ProductID  uint16 `json:"productId"`
+	Protocol   string `json:"protocol"`
+	VIAVersion int    `json:"viaVersion"`
+	LEDLayout  string `json:"ledLayout"`
+}
+
+// Device represents a connected keyboard device.
+type Device struct {
+	Keyboard
+	Path      string
+	VendorID  uint16
+	ProductID uint16
+}
+
+// LoadKeyboards reads keyboard definitions from keyboards.json.
+func LoadKeyboards() ([]Keyboard, error) {
+	path, err := findKeyboardsJSON()
+	if err != nil {
+		return nil, fmt.Errorf("find keyboards.json: %w", err)
+	}
+
+	data, err := os.ReadFile(path)
+	if err != nil {
+		return nil, fmt.Errorf("read keyboards.json: %w", err)
+	}
+
+	var keyboards []Keyboard
+	if err := json.Unmarshal(data, &keyboards); err != nil {
+		return nil, fmt.Errorf("parse keyboards.json: %w", err)
+	}
+
+	return keyboards, nil
+}
+
+// findKeyboardsJSON searches for keyboards.json in common locations.
+var findKeyboardsJSON = func() (string, error) {
+	exe, err := os.Executable()
+	if err == nil {
+		path := filepath.Join(filepath.Dir(exe), "keyboards.json")
+		if _, err := os.Stat(path); err == nil {
+			return path, nil
+		}
+	}
+
+	path := "keyboards.json"
+	if _, err := os.Stat(path); err == nil {
+		return path, nil
+	}
+
+	cwd, err := os.Getwd()
+	if err == nil {
+		path := filepath.Join(cwd, "keyboards.json")
+		if _, err := os.Stat(path); err == nil {
+			return path, nil
+		}
+	}
+
+	return "", fmt.Errorf("keyboards.json not found")
+}
+
+// Discover scans all HID devices and returns connected VIA-compatible keyboards.
+func Discover(keyboards []Keyboard) ([]Device, error) {
+	devices, err := hid.DiscoverAll()
+	if err != nil {
+		return nil, fmt.Errorf("enumerate HID devices: %w", err)
+	}
+
+	var found []Device
+	for _, kb := range keyboards {
+		for _, d := range devices {
+			if d.VendorID == kb.VendorID && d.ProductID == kb.ProductID {
+				found = append(found, Device{
+					Keyboard:  kb,
+					Path:      d.Path,
+					VendorID:  d.VendorID,
+					ProductID: d.ProductID,
+				})
+			}
+		}
+	}
+
+	return found, nil
+}

+ 97 - 0
internal/device/device_test.go

@@ -0,0 +1,97 @@
+package device
+
+import (
+	"os"
+	"path/filepath"
+	"testing"
+)
+
+func TestLoadKeyboards(t *testing.T) {
+	// Create a temporary keyboards.json for testing
+	tmpDir := t.TempDir()
+	tmpFile := filepath.Join(tmpDir, "keyboards.json")
+
+	testData := `[
+		{"name": "Test Keyboard 1", "vendorId": 22222, "productId": 1, "protocol": "via"},
+		{"name": "Test Keyboard 2", "vendorId": 33333, "productId": 2, "protocol": "viapro"}
+	]`
+
+	if err := os.WriteFile(tmpFile, []byte(testData), 0644); err != nil {
+		t.Fatalf("failed to write test file: %v", err)
+	}
+
+	// Save original function and override search path for testing
+	origFind := findKeyboardsJSON
+	findKeyboardsJSON = func() (string, error) {
+		return tmpFile, nil
+	}
+	defer func() { findKeyboardsJSON = origFind }()
+
+	keyboards, err := LoadKeyboards()
+	if err != nil {
+		t.Fatalf("LoadKeyboards() returned error: %v", err)
+	}
+
+	if len(keyboards) != 2 {
+		t.Errorf("LoadKeyboards() returned %d keyboards, want 2", len(keyboards))
+	}
+
+	if keyboards[0].Name != "Test Keyboard 1" {
+		t.Errorf("First keyboard name = %q, want %q", keyboards[0].Name, "Test Keyboard 1")
+	}
+
+	if keyboards[0].VendorID != 22222 {
+		t.Errorf("First keyboard VID = %d, want %d", keyboards[0].VendorID, 22222)
+	}
+
+	if keyboards[1].ProductID != 2 {
+		t.Errorf("Second keyboard PID = %d, want %d", keyboards[1].ProductID, 2)
+	}
+}
+
+func TestLoadKeyboardsInvalid(t *testing.T) {
+	// Test with invalid JSON
+	tmpDir := t.TempDir()
+	tmpFile := filepath.Join(tmpDir, "keyboards.json")
+
+	invalidData := `[this is not valid json`
+
+	if err := os.WriteFile(tmpFile, []byte(invalidData), 0644); err != nil {
+		t.Fatalf("failed to write test file: %v", err)
+	}
+
+	origFind := findKeyboardsJSON
+	findKeyboardsJSON = func() (string, error) {
+		return tmpFile, nil
+	}
+	defer func() { findKeyboardsJSON = origFind }()
+
+	_, err := LoadKeyboards()
+	if err == nil {
+		t.Error("LoadKeyboards() expected error for invalid JSON, got nil")
+	}
+}
+
+func TestLoadKeyboardsEmptyFile(t *testing.T) {
+	tmpDir := t.TempDir()
+	tmpFile := filepath.Join(tmpDir, "keyboards.json")
+
+	if err := os.WriteFile(tmpFile, []byte("[]"), 0644); err != nil {
+		t.Fatalf("failed to write test file: %v", err)
+	}
+
+	origFind := findKeyboardsJSON
+	findKeyboardsJSON = func() (string, error) {
+		return tmpFile, nil
+	}
+	defer func() { findKeyboardsJSON = origFind }()
+
+	keyboards, err := LoadKeyboards()
+	if err != nil {
+		t.Fatalf("LoadKeyboards() returned error: %v", err)
+	}
+
+	if len(keyboards) != 0 {
+		t.Errorf("LoadKeyboards() returned %d keyboards, want 0", len(keyboards))
+	}
+}

+ 182 - 0
internal/hid/hid.go

@@ -0,0 +1,182 @@
+package hid
+
+import (
+	"fmt"
+	"os"
+	"strings"
+	"syscall"
+	"unsafe"
+)
+
+// HID device access via /dev/hidraw* using pure Go syscalls.
+// No cgo, no libudev.
+
+const (
+	ioctlRDGET = 0x80000000 + 1
+	ioctlWRSET = 0x40000000 + 2
+	ioctlGET   = 0x30000000 + 3
+	ioctlSET   = 0x40000000 + 4
+)
+
+// Device represents a connected HID device.
+type Device struct {
+	fd        int
+	path      string
+	vendorID  uint16
+	productID uint16
+}
+
+// Devices lists all connected HID devices with VID/PID.
+type DevicesInfo struct {
+	Devices []DeviceInfo
+}
+
+// DeviceInfo holds device metadata without open handle.
+type DeviceInfo struct {
+	Path      string
+	VendorID  uint16
+	ProductID uint16
+}
+
+// OpenPath opens a HID device by path (e.g. "/dev/hidraw0").
+func OpenPath(path string) (*Device, error) {
+	fd, err := syscall.Open(path, syscall.O_RDWR, 0)
+	if err != nil {
+		return nil, fmt.Errorf("open %s: %w", path, err)
+	}
+	return &Device{fd: fd, path: path}, nil
+}
+
+// VID returns the vendor ID.
+func (d *Device) VID() uint16 { return d.vendorID }
+
+// PID returns the product ID.
+func (d *Device) PID() uint16 { return d.productID }
+
+// Path returns the device file path.
+func (d *Device) Path() string { return d.path }
+
+// SendReport sends an HID report.
+// report[0] must be the report ID (0 for single-report devices).
+func (d *Device) SendReport(reportID byte, report []byte) (int, error) {
+	buf := make([]byte, 1+len(report))
+	buf[0] = reportID
+	copy(buf[1:], report)
+
+	n, _, errno := syscall.Syscall(syscall.SYS_WRITE, uintptr(d.fd), uintptr(unsafe.Pointer(&buf[0])), uintptr(len(buf)))
+	if errno != 0 {
+		return 0, fmt.Errorf("write: %w", errno)
+	}
+	return int(n), nil
+}
+
+// Read reads a response report.
+func (d *Device) Read(buf []byte) (int, error) {
+	n, _, errno := syscall.Syscall(syscall.SYS_READ, uintptr(d.fd), uintptr(unsafe.Pointer(&buf[0])), uintptr(len(buf)))
+	if errno != 0 {
+		return 0, fmt.Errorf("read: %w", errno)
+	}
+	return int(n), nil
+}
+
+// Close closes the device.
+func (d *Device) Close() error {
+	return syscall.Close(d.fd)
+}
+
+// DiscoverAll returns a list of HID devices with their VID/PID.
+// It iterates /dev/hidraw* and reads VID/PID from /sys/class/hidraw/*/device/uevent.
+func DiscoverAll() ([]DeviceInfo, error) {
+	entries, err := os.ReadDir("/dev")
+	if err != nil {
+		return nil, fmt.Errorf("read /dev: %w", err)
+	}
+
+	var devices []DeviceInfo
+
+	for _, entry := range entries {
+		if !strings.HasPrefix(entry.Name(), "hidraw") {
+			continue
+		}
+
+		hidrawPath := "/dev/" + entry.Name()
+		sysEventPath := "/sys/class/hidraw/" + entry.Name() + "/device/uevent"
+
+		data, err := os.ReadFile(sysEventPath)
+		if err != nil {
+			continue
+		}
+
+		vid, pid := parseUEVENT(string(data))
+		if vid == 0 || pid == 0 {
+			continue
+		}
+
+		devices = append(devices, DeviceInfo{
+			Path:      hidrawPath,
+			VendorID:  vid,
+			ProductID: pid,
+		})
+	}
+
+	return devices, nil
+}
+
+func parseUEVENT(s string) (vid, pid uint16) {
+	for _, line := range splitLines(s) {
+		if len(line) >= 25 && line[:12] == "HID_ID=0003:" {
+			data := line[12:]
+			colon := -1
+			for i, c := range data {
+				if c == ':' {
+					colon = i
+					break
+				}
+			}
+			if colon > 0 && colon+1 < len(data) {
+				if v, e := parseHex(data[:colon]); e == nil {
+					vid = v
+				}
+				if p, e := parseHex(data[colon+1:]); e == nil {
+					pid = p
+				}
+			}
+		}
+	}
+	return vid, pid
+}
+
+func splitLines(s string) []string {
+	var lines []string
+	var current string
+	for _, c := range s {
+		if c == '\n' {
+			lines = append(lines, current)
+			current = ""
+		} else {
+			current += string(c)
+		}
+	}
+	if current != "" {
+		lines = append(lines, current)
+	}
+	return lines
+}
+
+func parseHex(s string) (uint16, error) {
+	var val uint16
+	for _, c := range s {
+		val <<= 4
+		switch {
+		case c >= '0' && c <= '9':
+			val |= uint16(c - '0')
+		case c >= 'a' && c <= 'f':
+			val |= uint16(c - 'a' + 10)
+		case c >= 'A' && c <= 'F':
+			val |= uint16(c - 'A' + 10)
+		default:
+			return 0, fmt.Errorf("invalid hex char: %c", c)
+		}
+	}
+	return val, nil
+}

+ 99 - 0
internal/hid/hid_test.go

@@ -0,0 +1,99 @@
+package hid
+
+import (
+	"testing"
+)
+
+func TestParseHex(t *testing.T) {
+	cases := []struct {
+		name  string
+		input string
+		want  uint16
+		err   bool
+	}{
+		{"0x36B0", "36b0", 0x36b0, false},
+		{"0x6666", "6666", 0x6666, false},
+		{"0x0001", "0001", 0x0001, false},
+		{"0xFFFF", "ffff", 0xffff, false},
+		{"0x0000", "0000", 0x0000, false},
+		{"uppercase", "ABCD", 0xabcd, false},
+		{"mixed case", "AbCd", 0xabcd, false},
+		{"6 chars", "0036b0", 0x36b0, false},
+		{"invalid hex", "12gh", 0, true},
+	}
+
+	for _, tc := range cases {
+		t.Run(tc.name, func(t *testing.T) {
+			got, err := parseHex(tc.input)
+			if tc.err {
+				if err == nil {
+					t.Errorf("parseHex(%q) expected error, got nil", tc.input)
+				}
+				return
+			}
+			if err != nil {
+				t.Errorf("parseHex(%q) unexpected error: %v", tc.input, err)
+				return
+			}
+			if got != tc.want {
+				t.Errorf("parseHex(%q) = %d, want %d", tc.input, got, tc.want)
+			}
+		})
+	}
+}
+
+func TestSplitLines(t *testing.T) {
+	cases := []struct {
+		name  string
+		input string
+		want  int
+	}{
+		{"single line", "hello", 1},
+		{"two lines", "hello\nworld", 2},
+		{"trailing newline", "hello\n", 1},
+		{"empty", "", 0},
+		{"single newline", "\n", 1},
+		{"double newline", "\n\n", 2},
+		{"with content and newlines", "foo\nbar\n", 2},
+	}
+
+	for _, tc := range cases {
+		t.Run(tc.name, func(t *testing.T) {
+			got := splitLines(tc.input)
+			if len(got) != tc.want {
+				t.Errorf("splitLines(%q) = %d lines, want %d lines", tc.input, len(got), tc.want)
+			}
+		})
+	}
+}
+
+func TestParseUEVENT(t *testing.T) {
+	ueventData := `DRIVER=hid-generic
+HID_ID=0003:000036B0:0000309F
+HID_NAME=RDMCTMZT Impact 80
+HID_PHYS=usb-0000:10:00.0-5.4.1.4.4.1/input0
+HID_UNIQ=
+MODALIAS=hid:b0003g0001v000036B0p0000309F
+`
+
+	vid, pid := parseUEVENT(ueventData)
+	if vid != 0x36B0 {
+		t.Errorf("parseUEVENT() vid = 0x%04X, want 0x36B0", vid)
+	}
+	if pid != 0x309F {
+		t.Errorf("parseUEVENT() pid = 0x%04X, want 0x309F", pid)
+	}
+
+	// Mouse
+	mouse := `DRIVER=hid-generic
+HID_ID=0003:0000046D:0000C041
+HID_NAME=Logitech USB Gaming Mouse
+`
+	vid, pid = parseUEVENT(mouse)
+	if vid != 0x046D {
+		t.Errorf("parseUEVENT() mouse vid = 0x%04X, want 0x046D", vid)
+	}
+	if pid != 0xC041 {
+		t.Errorf("parseUEVENT() mouse pid = 0x%04X, want 0xC041", pid)
+	}
+}

+ 205 - 0
internal/rgb/effects.go

@@ -0,0 +1,205 @@
+package rgb
+
+import (
+	"fmt"
+	"math"
+)
+
+// Effect defines an RGB lighting effect.
+type Effect uint8
+
+const (
+	Static       Effect = 0
+	Breathing    Effect = 1
+	EffectAlmond Effect = 2
+	BandSat      Effect = 3
+	BandVal      Effect = 4
+	BandHue      Effect = 5
+	Chroma       Effect = 6
+	Rainbow      Effect = 7
+	RainbowWave  Effect = 8
+	Dance        Effect = 9
+	RainbowDance Effect = 10
+	ChromaDance  Effect = 11
+	Snake        Effect = 12
+	Tail         Effect = 13
+	Split        Effect = 14
+	Heartbeat    Effect = 15
+	Blink        Effect = 16
+	BlinkStatic  Effect = 17
+	Gradient     Effect = 18
+	Testing      Effect = 19
+)
+
+// EffectName returns the human-readable name of an effect.
+func (e Effect) String() string {
+	names := map[Effect]string{
+		Static:       "static",
+		Breathing:    "breathing",
+		EffectAlmond: "almond",
+		BandSat:      "band_sat",
+		BandVal:      "band_val",
+		BandHue:      "band_hue",
+		Chroma:       "chroma",
+		Rainbow:      "rainbow",
+		RainbowWave:  "rainbow_wave",
+		Dance:        "dance",
+		RainbowDance: "rainbow_dance",
+		ChromaDance:  "chroma_dance",
+		Snake:        "snake",
+		Tail:         "tail",
+		Split:        "split",
+		Heartbeat:    "heartbeat",
+		Blink:        "blink",
+		BlinkStatic:  "blink_static",
+		Gradient:     "gradient",
+		Testing:      "testing",
+	}
+	if name, ok := names[e]; ok {
+		return name
+	}
+	return "unknown"
+}
+
+// ParseEffect converts a string name to an Effect.
+func ParseEffect(name string) (Effect, error) {
+	switch name {
+	case "static":
+		return Static, nil
+	case "breathing":
+		return Breathing, nil
+	case "almond":
+		return EffectAlmond, nil
+	case "band_sat":
+		return BandSat, nil
+	case "band_val":
+		return BandVal, nil
+	case "band_hue":
+		return BandHue, nil
+	case "chroma":
+		return Chroma, nil
+	case "rainbow":
+		return Rainbow, nil
+	case "rainbow_wave":
+		return RainbowWave, nil
+	case "dance":
+		return Dance, nil
+	case "rainbow_dance":
+		return RainbowDance, nil
+	case "chroma_dance":
+		return ChromaDance, nil
+	case "snake":
+		return Snake, nil
+	case "tail":
+		return Tail, nil
+	case "split":
+		return Split, nil
+	case "heartbeat":
+		return Heartbeat, nil
+	case "blink":
+		return Blink, nil
+	case "blink_static":
+		return BlinkStatic, nil
+	case "gradient":
+		return Gradient, nil
+	case "testing":
+		return Testing, nil
+	default:
+		return 0, fmt.Errorf("unknown effect: %s", name)
+	}
+}
+
+// LEDParam defines QMK rgblight parameter indices.
+type LEDParam uint8
+
+const (
+	Mode         LEDParam = 0
+	Brightness   LEDParam = 1
+	Speed        LEDParam = 2
+	Enable       LEDParam = 3
+	Hue          LEDParam = 26
+	Saturation   LEDParam = 27
+	EffectID     LEDParam = 48
+	EffectSwitch LEDParam = 49
+	BrightSet    LEDParam = 50
+)
+
+// Color represents an RGB color.
+type Color struct {
+	R uint8
+	G uint8
+	B uint8
+}
+
+// ParseHexColor parses a 6-digit hex color string (e.g. "ff0000").
+func ParseHexColor(s string) (Color, error) {
+	if len(s) != 6 {
+		return Color{}, fmt.Errorf("invalid hex color: %s", s)
+	}
+	r := hexDigit(s[0])<<4 + hexDigit(s[1])
+	g := hexDigit(s[2])<<4 + hexDigit(s[3])
+	b := hexDigit(s[4])<<4 + hexDigit(s[5])
+	return Color{R: r, G: g, B: b}, nil
+}
+
+func hexDigit(c byte) uint8 {
+	switch {
+	case c >= '0' && c <= '9':
+		return c - '0'
+	case c >= 'a' && c <= 'f':
+		return c - 'a' + 10
+	case c >= 'A' && c <= 'F':
+		return c - 'A' + 10
+	default:
+		return 0
+	}
+}
+
+// State represents the current RGB state from the keyboard.
+type State struct {
+	Enabled    bool
+	Mode       Effect
+	Brightness uint8
+	Speed      uint8
+	Color      Color
+}
+
+// HSV converts the color to HSV (QMK rgblight uses 0-255 ranges for H, S, V).
+func (c Color) HSV() (h, s, v uint8) {
+	r, g, b := float64(c.R), float64(c.G), float64(c.B)
+	max := math.Max(r, math.Max(g, b))
+	min := math.Min(r, math.Min(g, b))
+	v = uint8(max)
+
+	d := max - min
+	if d == 0 {
+		return 0, 0, v
+	}
+
+	if max > 0 {
+		s = uint8((d / max) * 255)
+	} else {
+		return 0, 0, 0
+	}
+
+	var hFloat float64
+	switch {
+	case max == r:
+		hFloat = ((g - b) / d) * 60.0
+	case max == g:
+		hFloat = ((b-r)/d)*60.0 + 120.0
+	case max == b:
+		hFloat = ((r-g)/d)*60.0 + 240.0
+	}
+
+	if hFloat < 0 {
+		hFloat += 360.0
+	}
+
+	h = uint8((hFloat / 360.0) * 255.0)
+	if h >= 255 {
+		h = 254
+	}
+
+	return h, s, v
+}

+ 142 - 0
internal/rgb/effects_test.go

@@ -0,0 +1,142 @@
+package rgb
+
+import (
+	"testing"
+)
+
+func TestParseEffect(t *testing.T) {
+	cases := []struct {
+		name  string
+		input string
+		want  Effect
+		err   bool
+	}{
+		{"static", "static", Static, false},
+		{"breathing", "breathing", Breathing, false},
+		{"rainbow", "rainbow", Rainbow, false},
+		{"rainbow_wave", "rainbow_wave", RainbowWave, false},
+		{"snake", "snake", Snake, false},
+		{"gradient", "gradient", Gradient, false},
+		{"tail", "tail", Tail, false},
+		{"unknown", "flubber", 0, true},
+		{"empty", "", 0, true},
+	}
+
+	for _, tc := range cases {
+		t.Run(tc.name, func(t *testing.T) {
+			got, err := ParseEffect(tc.input)
+			if tc.err {
+				if err == nil {
+					t.Errorf("ParseEffect(%q) expected error, got nil", tc.input)
+				}
+				return
+			}
+			if err != nil {
+				t.Errorf("ParseEffect(%q) unexpected error: %v", tc.input, err)
+				return
+			}
+			if got != tc.want {
+				t.Errorf("ParseEffect(%q) = %d, want %d", tc.input, got, tc.want)
+			}
+		})
+	}
+}
+
+func TestEffectString(t *testing.T) {
+	cases := []struct {
+		effect Effect
+		want   string
+	}{
+		{Static, "static"},
+		{Breathing, "breathing"},
+		{Rainbow, "rainbow"},
+		{RainbowWave, "rainbow_wave"},
+		{Effect(99), "unknown"},
+	}
+
+	for _, tc := range cases {
+		t.Run(tc.want, func(t *testing.T) {
+			got := tc.effect.String()
+			if got != tc.want {
+				t.Errorf("Effect(%d).String() = %q, want %q", tc.effect, got, tc.want)
+			}
+		})
+	}
+}
+
+func TestParseHexColor(t *testing.T) {
+	cases := []struct {
+		name  string
+		input string
+		want  Color
+		err   bool
+	}{
+		{"red", "ff0000", Color{R: 255, G: 0, B: 0}, false},
+		{"green", "00ff00", Color{R: 0, G: 255, B: 0}, false},
+		{"blue", "0000ff", Color{R: 0, G: 0, B: 255}, false},
+		{"white", "ffffff", Color{R: 255, G: 255, B: 255}, false},
+		{"black", "000000", Color{R: 0, G: 0, B: 0}, false},
+		{"lowercase", "aabbcc", Color{R: 0xaa, G: 0xbb, B: 0xcc}, false},
+		{"uppercase", "AABBCC", Color{R: 0xaa, G: 0xbb, B: 0xcc}, false},
+		{"short", "ff000", Color{0, 0, 0}, true},
+		{"long", "ff00000", Color{0, 0, 0}, true},
+		{"empty", "", Color{0, 0, 0}, true},
+		{"invalid", "gggggg", Color{R: 0, G: 0, B: 0}, false},
+	}
+
+	for _, tc := range cases {
+		t.Run(tc.name, func(t *testing.T) {
+			got, err := ParseHexColor(tc.input)
+			if tc.err {
+				if err == nil {
+					t.Errorf("ParseHexColor(%q) expected error, got nil", tc.input)
+				}
+				return
+			}
+			if err != nil {
+				t.Errorf("ParseHexColor(%q) unexpected error: %v", tc.input, err)
+				return
+			}
+			if got != tc.want {
+				t.Errorf("ParseHexColor(%q) = %+v, want %+v", tc.input, got, tc.want)
+			}
+		})
+	}
+}
+
+func TestColorToHSV(t *testing.T) {
+	cases := []struct {
+		name  string
+		color Color
+		wantH uint8
+		wantS uint8
+		wantV uint8
+	}{
+		{"red", Color{R: 255, G: 0, B: 0}, 0, 255, 255},
+		{"green", Color{R: 0, G: 255, B: 0}, 85, 255, 255},
+		{"blue", Color{R: 0, G: 0, B: 255}, 170, 255, 255},
+		{"white", Color{R: 255, G: 255, B: 255}, 0, 0, 255},
+		{"black", Color{R: 0, G: 0, B: 0}, 0, 0, 0},
+		{"yellow", Color{R: 255, G: 255, B: 0}, 42, 255, 255},
+		{"cyan", Color{R: 0, G: 255, B: 255}, 127, 255, 255},
+		{"magenta", Color{R: 255, G: 0, B: 255}, 212, 255, 255},
+		{"gray", Color{R: 128, G: 128, B: 128}, 0, 0, 128},
+		{"dark red", Color{R: 128, G: 0, B: 0}, 0, 255, 128},
+		{"light red", Color{R: 255, G: 128, B: 128}, 0, 127, 255},
+	}
+
+	for _, tc := range cases {
+		t.Run(tc.name, func(t *testing.T) {
+			h, s, v := tc.color.HSV()
+			if h != tc.wantH {
+				t.Errorf("HSV(%q).H = %d, want %d", tc.name, h, tc.wantH)
+			}
+			if s != tc.wantS {
+				t.Errorf("HSV(%q).S = %d, want %d", tc.name, s, tc.wantS)
+			}
+			if v != tc.wantV {
+				t.Errorf("HSV(%q).V = %d, want %d", tc.name, v, tc.wantV)
+			}
+		})
+	}
+}

+ 85 - 0
internal/via/protocol.go

@@ -0,0 +1,85 @@
+package via
+
+import (
+	"fmt"
+
+	"github.com/wobkey/rgb/internal/device"
+	"github.com/wobkey/rgb/internal/hid"
+)
+
+// Protocol implements the VIA protocol for communication with a keyboard.
+type Protocol struct {
+	handle *hid.Device
+	kb     device.Keyboard
+}
+
+// New creates a new VIA protocol handler for a connected device.
+func New(dev device.Device) (*Protocol, error) {
+	h, err := hid.OpenPath(dev.Path)
+	if err != nil {
+		return nil, fmt.Errorf("open device %s: %w", dev.Path, err)
+	}
+
+	return &Protocol{
+		handle: h,
+		kb:     dev.Keyboard,
+	}, nil
+}
+
+// Close releases the device handle.
+func (p *Protocol) Close() error {
+	if p.handle != nil {
+		return p.handle.Close()
+	}
+	return nil
+}
+
+// ReportID is the VIA report ID.
+const ReportID = 0x52
+
+// Message defines VIA message types.
+type Message uint8
+
+const (
+	Enable Message = 0x01
+	Set    Message = 0x02
+	Get    Message = 0x03
+)
+
+// LEDType defines QMK LED subsystem types.
+type LEDType uint8
+
+const (
+	RGBLight LEDType = 0x01
+)
+
+// EnableHandshake sends the enable message to initialize the VIA connection.
+func (p *Protocol) EnableHandshake() error {
+	report := []byte{0x00, byte(Enable), 0x00, 0x00, 0x00}
+	_, err := p.handle.SendReport(report[0], report[1:])
+	return err
+}
+
+// SetValue sends a set value command for the given LED type and parameter.
+func (p *Protocol) SetValue(ledType LEDType, param uint8, value uint8) error {
+	report := []byte{0x00, byte(Set), byte(ledType), param, value}
+	_, err := p.handle.SendReport(report[0], report[1:])
+	return err
+}
+
+// GetValue sends a get value request and reads the response.
+func (p *Protocol) GetValue(ledType LEDType, param uint8) ([]byte, error) {
+	report := []byte{0x00, byte(Get), byte(ledType), param, 0x00}
+	_, err := p.handle.SendReport(report[0], report[1:])
+	if err != nil {
+		return nil, fmt.Errorf("send get request: %w", err)
+	}
+
+	buf := make([]byte, 64)
+	n, err := p.handle.Read(buf)
+	if err != nil {
+		return nil, fmt.Errorf("read response: %w", err)
+	}
+
+	return buf[:n], nil
+}

+ 18 - 0
keyboards.json

@@ -0,0 +1,18 @@
+[
+  {
+    "name": "Wobkey Rainy 75",
+    "vendorId": 26214,
+    "productId": 1,
+    "protocol": "via",
+    "viaVersion": 3,
+    "ledLayout": "rgblight"
+  },
+  {
+    "name": "Wobkey Impact 80",
+    "vendorId": 13968,
+    "productId": 12447,
+    "protocol": "via",
+    "viaVersion": 3,
+    "ledLayout": "rgblight"
+  }
+]