release: v0.7.0

Merge dev into main for Hermes-Relay v0.7.0 and relay-v0.7.0.
This commit is contained in:
Bailey Dixon
2026-05-19 20:05:19 -04:00
committed by GitHub
253 changed files with 56971 additions and 1913 deletions
+65
View File
@@ -0,0 +1,65 @@
name: CI dashboard plugin
on:
push:
branches: [main, dev]
paths:
- "plugin/dashboard/**"
- "scripts/check-relay-version-sync.py"
- ".github/workflows/ci-dashboard.yml"
pull_request:
branches: [main, dev]
paths:
- "plugin/dashboard/**"
- "scripts/check-relay-version-sync.py"
- ".github/workflows/ci-dashboard.yml"
permissions:
contents: read
concurrency:
group: ci-dashboard-${{ github.ref }}
cancel-in-progress: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
jobs:
build-and-test:
name: Build and test dashboard plugin
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v6
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version: "22"
cache: npm
cache-dependency-path: plugin/dashboard/package-lock.json
- name: Install dashboard deps
working-directory: plugin/dashboard
run: npm ci
- name: Build dashboard bundle
working-directory: plugin/dashboard
run: npm run build
- name: Setup Python
uses: actions/setup-python@v6
with:
python-version: "3.11"
- name: Verify relay-owned version metadata
run: python scripts/check-relay-version-sync.py
- name: Install dashboard API test deps
run: pip install -r relay_server/requirements.txt fastapi httpx pytest requests
- name: Run dashboard API tests
run: python -m unittest plugin.dashboard.test_plugin_api
- name: Verify dashboard bundle outputs
run: |
test -s plugin/dashboard/dist/index.js
test -s plugin/dashboard/dist/style.css
grep -q "hr-modal-card" plugin/dashboard/dist/style.css
+29 -1
View File
@@ -1,4 +1,4 @@
name: CI desktop CLI
name: CI desktop
on:
push:
@@ -86,3 +86,31 @@ jobs:
- name: --help
run: node bin/hermes-relay.js --help
tray-shell:
name: Tray shell checks
runs-on: windows-latest
defaults:
run:
working-directory: desktop
steps:
- uses: actions/checkout@v4
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version: '22'
cache: npm
cache-dependency-path: desktop/package-lock.json
- name: Setup Rust
uses: dtolnay/rust-toolchain@stable
- name: Install deps
run: npm ci
- name: Cargo check tray shell
run: npm run tray:check
- name: Test tray shell
run: npm run tray:test
+42 -11
View File
@@ -4,7 +4,7 @@
# Python-affecting paths so Android-only changes don't spin up the
# Python toolchain.
#
# Pipeline: syntax-check -> unit-tests
# Pipeline: syntax-check -> focused relay tests
name: CI — Relay
@@ -12,18 +12,42 @@ on:
push:
branches: [main, dev]
paths:
- "plugin/**"
- "plugin/__init__.py"
- "plugin/android_tool.py"
- "plugin/cli.py"
- "plugin/pair.py"
- "plugin/plugin.yaml"
- "plugin/dashboard/manifest.json"
- "plugin/dashboard/package.json"
- "plugin/dashboard/package-lock.json"
- "plugin/relay/**"
- "plugin/tools/**"
- "plugin/tests/**"
- "relay_server/**"
- "hermes_relay_bootstrap/**"
- "pyproject.toml"
- "scripts/check-relay-version-sync.py"
- "scripts/bump-relay-version.sh"
- ".github/workflows/ci-relay.yml"
pull_request:
branches: [main, dev]
paths:
- "plugin/**"
- "plugin/__init__.py"
- "plugin/android_tool.py"
- "plugin/cli.py"
- "plugin/pair.py"
- "plugin/plugin.yaml"
- "plugin/dashboard/manifest.json"
- "plugin/dashboard/package.json"
- "plugin/dashboard/package-lock.json"
- "plugin/relay/**"
- "plugin/tools/**"
- "plugin/tests/**"
- "relay_server/**"
- "hermes_relay_bootstrap/**"
- "pyproject.toml"
- "scripts/check-relay-version-sync.py"
- "scripts/bump-relay-version.sh"
- ".github/workflows/ci-relay.yml"
# Cancel in-progress runs for the same branch/PR, but let main and dev finish
@@ -38,6 +62,7 @@ jobs:
syntax-check:
name: Syntax check (Python)
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@v6
@@ -56,21 +81,27 @@ jobs:
python -m py_compile plugin/relay/channels/terminal.py
python -m py_compile plugin/relay/channels/chat.py
python -m py_compile plugin/relay/channels/bridge.py
python -m py_compile plugin/relay/voice.py
python -m py_compile plugin/relay/upstream_voice.py
- name: Syntax check (relay_server shim)
run: python -m py_compile relay_server/__init__.py relay_server/__main__.py
- name: Validate Relay version metadata
run: python scripts/check-relay-version-sync.py
# ──────────────────────────────────────────────
# Python Relay — unittest discover
# Python Relay — focused route/auth/session tests
#
# Tests are ADVISORY on dev (push or PR) so WIP commits don't block the
# merge queue. Strict on main — the dev → main release-merge PR surfaces
# any real failures before release.
# ──────────────────────────────────────────────
unit-tests:
name: Unit tests (Python)
name: Focused Relay tests (Python)
needs: syntax-check
runs-on: ubuntu-latest
timeout-minutes: 10
# Advisory on dev, strict on main. Evaluates to false (= strict) for
# pushes to main and PRs whose base branch is main; true (= advisory)
# for everything else (dev pushes, dev-targeted PRs, feature branches).
@@ -84,14 +115,14 @@ jobs:
with:
python-version: "3.11"
# conftest.py imports `pytest` and `responses` at collection time.
# `unittest discover` walks conftest.py like any other module, so both
# must be importable even though none of the tests themselves use
# pytest fixtures (they're all stdlib unittest.TestCase).
- name: Install dependencies
run: |
pip install -r relay_server/requirements.txt
pip install pytest responses
- name: Run unit tests
run: python -m unittest discover plugin/tests
- name: Run focused Relay tests
run: |
python -m pytest \
plugin/tests/test_relay_security.py \
plugin/tests/test_voice_routes.py \
plugin/tests/test_session_grants.py
+12 -1
View File
@@ -17,21 +17,32 @@ jobs:
# github.event.pull_request.user.login == 'external-contributor' ||
# github.event.pull_request.user.login == 'new-developer' ||
# github.event.pull_request.author_association == 'FIRST_TIME_CONTRIBUTOR'
runs-on: ubuntu-latest
timeout-minutes: 20
permissions:
contents: read
pull-requests: read
issues: read
id-token: write
env:
IS_RELEASE_PR: ${{ github.event.pull_request.base.ref == 'main' && github.event.pull_request.head.ref == 'dev' && startsWith(github.event.pull_request.title, 'release:') }}
steps:
- name: Skip aggregate release PR review
if: env.IS_RELEASE_PR == 'true'
run: |
echo "Skipping Claude Code Review for aggregate dev -> main release PR."
echo "Feature work is reviewed before it lands on dev; release PRs are gated by CI and release metadata checks."
- name: Checkout repository
if: env.IS_RELEASE_PR != 'true'
uses: actions/checkout@v4
with:
fetch-depth: 1
- name: Run Claude Code Review
if: env.IS_RELEASE_PR != 'true'
timeout-minutes: 15
id: claude-review
uses: anthropics/claude-code-action@v1
with:
+124 -33
View File
@@ -1,4 +1,4 @@
name: Release desktop CLI
name: Release desktop
on:
push:
@@ -8,8 +8,8 @@ permissions:
contents: write
jobs:
build-binaries:
name: Build cross-platform binaries via Bun compile
build-cli-binaries:
name: Build cross-platform CLI binaries via Bun compile
runs-on: ubuntu-latest
defaults:
run:
@@ -44,10 +44,8 @@ jobs:
- name: Prepare binary output dir
run: mkdir -p dist/bin
# NOTE: delegate to the package.json scripts so there's a single source
# of truth for Bun compile flags. Previously these steps inlined their
# own flag list, which silently diverged from `npm run build:bin:*` and
# made the "drop --bytecode" fix ineffective on desktop-v0.3.0-alpha.2.
# Keep the package.json scripts as the single source of truth for Bun
# compile flags so release and local smoke builds cannot diverge.
- name: Build Windows x64
run: npm run build:bin:win
@@ -66,20 +64,13 @@ jobs:
for f in dist/bin/hermes-relay-*; do
sz=$(stat -c%s "$f")
mb=$(( sz / 1024 / 1024 ))
echo " $f — ${mb} MB"
echo " $f - ${mb} MB"
if [ "$sz" -gt 157286400 ]; then
echo "FAIL: $f exceeds 150 MB — Bun likely shipped a debug build or we added a large dep."
echo "FAIL: $f exceeds 150 MB - Bun likely shipped a debug build or we added a large dep."
exit 1
fi
done
# Smoke-test the Linux binary (runner platform) before publishing.
# Catches the two failure modes we've burned alpha tags on:
# - startup segfault (exit != 0, no output)
# - silent exit 0 with zero output (main() never invoked)
# We can only smoke the Linux target without a cross-platform matrix;
# Windows/macOS smoke would need their own runners — tracked as an
# alpha.5+ hardening item.
- name: Smoke-test Linux binary
run: |
set -e
@@ -92,17 +83,109 @@ jobs:
echo "Raw output was: [$out]"
exit 1
fi
echo " smoke OK: $cmd → $(echo "$out" | head -1)"
echo " smoke OK: $cmd -> $(echo "$out" | head -1)"
done
- name: Generate SHA256SUMS
working-directory: desktop/dist/bin
- name: Upload CLI release assets
uses: actions/upload-artifact@v4
with:
name: desktop-cli-release
path: |
desktop/dist/bin/hermes-relay-win-x64.exe
desktop/dist/bin/hermes-relay-linux-x64
desktop/dist/bin/hermes-relay-darwin-x64
desktop/dist/bin/hermes-relay-darwin-arm64
retention-days: 7
build-windows-tray-installer:
name: Build Windows tray installer
runs-on: windows-latest
defaults:
run:
working-directory: desktop
steps:
- uses: actions/checkout@v4
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version: '22'
cache: npm
cache-dependency-path: desktop/package-lock.json
- name: Setup Bun
uses: oven-sh/setup-bun@v2
with:
bun-version: '1.3.x'
- name: Setup Rust
uses: dtolnay/rust-toolchain@stable
- name: Install deps
run: npm ci
- name: Type-check
run: npm run type-check
- name: Build dist/ (tsc)
run: npm run build
- name: Test tray shell
run: npm run tray:test
- name: Build tray installer
run: npm run tray:build
- name: Normalize installer asset name
shell: pwsh
run: |
sha256sum hermes-relay-* > SHA256SUMS.txt
cat SHA256SUMS.txt
New-Item -ItemType Directory -Force -Path dist/tray | Out-Null
$installer = Get-ChildItem -Path tray/src-tauri/target/release/bundle/nsis -Filter '*_x64-setup.exe' | Select-Object -First 1
if (-not $installer) { throw 'NSIS installer was not produced' }
Copy-Item -Force $installer.FullName dist/tray/hermes-relay-desktop-windows-x64-setup.exe
- name: Smoke-test tray exe launch
shell: pwsh
run: |
$home = Join-Path $env:RUNNER_TEMP 'hermes-tray-smoke-home'
New-Item -ItemType Directory -Force -Path $home | Out-Null
$env:USERPROFILE = $home
$env:HOME = $home
$proc = Start-Process -FilePath tray/src-tauri/target/release/hermes-relay-desktop.exe -WindowStyle Hidden -PassThru
Start-Sleep -Seconds 5
if ($proc.HasExited) { throw "tray app exited early with code $($proc.ExitCode)" }
Stop-Process -Id $proc.Id -Force
Write-Host "tray launch smoke OK pid=$($proc.Id)"
- name: Upload Windows tray release asset
uses: actions/upload-artifact@v4
with:
name: desktop-windows-tray-release
path: desktop/dist/tray/hermes-relay-desktop-windows-x64-setup.exe
retention-days: 7
publish-release:
name: Publish GitHub Release
runs-on: ubuntu-latest
needs:
- build-cli-binaries
- build-windows-tray-installer
steps:
- uses: actions/download-artifact@v4
with:
path: release-assets
- name: Generate SHA256SUMS
run: |
set -e
find release-assets -type f ! -name SHA256SUMS.txt -print0 \
| sort -z \
| xargs -0 sha256sum \
| sed -E 's#release-assets/[^/]+/##' > release-assets/SHA256SUMS.txt
cat release-assets/SHA256SUMS.txt
- name: Publish GitHub Release
uses: softprops/action-gh-release@v2
uses: softprops/action-gh-release@v3
with:
name: ${{ github.ref_name }}
tag_name: ${{ github.ref_name }}
@@ -110,18 +193,23 @@ jobs:
prerelease: ${{ contains(github.ref_name, 'alpha') || contains(github.ref_name, 'beta') || contains(github.ref_name, 'rc') }}
fail_on_unmatched_files: true
body: |
# Hermes-Relay Desktop CLI — ${{ github.ref_name }}
# Hermes-Relay Desktop - ${{ github.ref_name }}
**Experimental phase.** Binaries are unsigned — Windows SmartScreen and macOS Gatekeeper will warn on first launch. See the install scripts for the `Unblock-File` / `xattr -dr` escape hatches.
**Experimental phase.** Assets are unsigned - Windows SmartScreen and macOS Gatekeeper will warn on first launch. Windows now ships a tray installer as the primary desktop surface; CLI binaries remain available for terminal/headless use and for macOS/Linux.
## Install
**Windows (PowerShell):**
**Windows tray app (PowerShell):**
```powershell
irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
```
**macOS / Linux:**
**Windows CLI only:**
```powershell
$env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
```
**macOS / Linux CLI:**
```bash
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.sh | sh
```
@@ -130,17 +218,20 @@ jobs:
## Verify
```
```text
hermes-relay --version
hermes-relay pair --remote ws://<host>:8767
hermes-relay shell
```
See [Desktop CLI docs](https://codename-11.github.io/hermes-relay/desktop/) for full usage.
Open **Hermes Relay Desktop** from the Windows Start menu for tray pairing, devices, task log, settings, pause, and emergency stop.
See [Desktop docs](https://codename-11.github.io/hermes-relay/desktop/) for full usage.
files: |
desktop/dist/bin/hermes-relay-win-x64.exe
desktop/dist/bin/hermes-relay-linux-x64
desktop/dist/bin/hermes-relay-darwin-x64
desktop/dist/bin/hermes-relay-darwin-arm64
desktop/dist/bin/SHA256SUMS.txt
release-assets/desktop-cli-release/hermes-relay-win-x64.exe
release-assets/desktop-cli-release/hermes-relay-linux-x64
release-assets/desktop-cli-release/hermes-relay-darwin-x64
release-assets/desktop-cli-release/hermes-relay-darwin-arm64
release-assets/desktop-windows-tray-release/hermes-relay-desktop-windows-x64-setup.exe
release-assets/SHA256SUMS.txt
+117
View File
@@ -0,0 +1,117 @@
name: Release Relay server
on:
push:
tags:
- "relay-v*"
permissions:
contents: write
jobs:
validate:
name: Validate Relay release
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
version: ${{ steps.version.outputs.version }}
steps:
- uses: actions/checkout@v6
- name: Extract version from tag
id: version
run: echo "version=${GITHUB_REF#refs/tags/relay-v}" >> "$GITHUB_OUTPUT"
- name: Verify Relay version sync
run: python scripts/check-relay-version-sync.py --expect "$TAG_VERSION"
env:
TAG_VERSION: ${{ steps.version.outputs.version }}
test:
name: Test Relay package
needs: validate
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@v6
- name: Set up Python 3.11
uses: actions/setup-python@v6
with:
python-version: "3.11"
- name: Install test dependencies
run: |
pip install -r relay_server/requirements.txt
pip install pytest responses
- name: Syntax check
run: |
python -m py_compile plugin/relay/server.py
python -m py_compile plugin/relay/voice.py
python -m py_compile plugin/relay/upstream_voice.py
python -m py_compile plugin/relay/voice_auth.py
python -m py_compile plugin/tools/android_tool.py
python -m py_compile plugin/tools/desktop_tool.py
python -m py_compile relay_server/__init__.py relay_server/__main__.py
- name: Run focused Relay tests
run: |
python -m pytest \
plugin/tests/test_relay_security.py \
plugin/tests/test_voice_routes.py \
plugin/tests/test_session_grants.py
package:
name: Build and publish Relay package
needs: [validate, test]
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@v6
- name: Set up Python 3.11
uses: actions/setup-python@v6
with:
python-version: "3.11"
- name: Build wheel and sdist
run: |
pip install build
python -m build
- name: Generate checksums
run: |
cd dist
sha256sum * > SHA256SUMS.txt
cat SHA256SUMS.txt
- name: Publish GitHub Release
uses: softprops/action-gh-release@v3
with:
name: relay-v${{ needs.validate.outputs.version }}
tag_name: relay-v${{ needs.validate.outputs.version }}
prerelease: ${{ contains(needs.validate.outputs.version, '-') }}
fail_on_unmatched_files: true
body: |
# Hermes-Relay server/Python package - relay-v${{ needs.validate.outputs.version }}
This release contains the Relay server and Python plugin package.
Android app releases use `v*` tags. Desktop CLI releases use
`desktop-v*` tags.
## Install
```bash
pip install hermes-relay==${{ needs.validate.outputs.version }}
```
## Verify
```bash
python -m relay_server --help
```
files: |
dist/*.whl
dist/*.tar.gz
dist/SHA256SUMS.txt
+5 -3
View File
@@ -1,8 +1,9 @@
# Hermes-Relay — Release Pipeline
# Hermes-Relay — Android App Release Pipeline
#
# Triggered when a version tag (v*) is pushed.
# Validates the tag matches the app version in libs.versions.toml,
# runs CI checks, builds a release APK, and creates a GitHub Release.
# runs focused Android checks, builds release APK/AAB artifacts, and creates a
# GitHub Release. Relay server/Python package releases use relay-v* tags.
name: Release
@@ -77,6 +78,7 @@ jobs:
name: Build & Publish Release
needs: [validate, ci]
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@v6
@@ -131,7 +133,7 @@ jobs:
cat SHA256SUMS.txt
- name: Create GitHub Release
uses: softprops/action-gh-release@v2
uses: softprops/action-gh-release@v3
with:
name: v${{ needs.validate.outputs.version }}
body_path: RELEASE_NOTES.md
+10
View File
@@ -46,6 +46,13 @@ certs/
# Local tools
.subframe/
voice-lab-runs/
realtime-voice-runs/
realtime-agent-runs/
voice_rec_*.wav
# Ad-hoc debugging artifacts (logcat dumps, screenshots, UI XMLs)
.scratch/
# VitePress
user-docs/.vitepress/cache/
@@ -75,3 +82,6 @@ keystore.properties
# Desktop TUI smoke harness runtime artifacts
.smoke-relay.pid
.smoke-relay.log
# Generated tray frontend vendor assets copied from desktop/node_modules
desktop/tray/ui/vendor/
-386
View File
@@ -1,386 +0,0 @@
<!-- @subframe-version 0.15.1-beta -->
<!-- @subframe-managed -->
# hermes-android - SubFrame Project
This project is managed with **SubFrame**. AI assistants should follow the rules below to keep documentation up to date.
> **Note:** This file is named `AGENTS.md` to be AI-tool agnostic. CLAUDE.md and GEMINI.md contain a reference to this file.
---
## Core Working Principle
**Only do what the user asks.** Do not go beyond the scope of the request.
- Implement exactly what the user requested — nothing more, nothing less.
- Do not change business logic, flow, or architecture unless the user explicitly asks for it.
- If a user asks for a design change, only change the design. Do not refactor, restructure, or modify functionality alongside it.
- If you have additional suggestions or improvements, **present them as suggestions** to the user. Never implement them without approval.
- The user's request must be completed first. Additional ideas come after, as proposals.
---
## Relationship to Native AI Tools
SubFrame **enhances** native AI coding tools — it does not replace them.
**Claude Code** works exactly as normal. Built-in features (`/init`, `/commit`, `/review-pr`, `/compact`, `/memory`, CLAUDE.md) are fully supported. CLAUDE.md is Claude Code's native instruction file — users can add their own tool-specific instructions freely. SubFrame adds a small backlink reference pointing to this AGENTS.md file using HTML comment markers (`<!-- SUBFRAME:BEGIN -->` / `<!-- SUBFRAME:END -->`). SubFrame will never overwrite user content in CLAUDE.md.
**Gemini CLI** works exactly as normal. Built-in features (`/init`, `/model`, `/memory`, `/compress`, `/settings`, GEMINI.md) are fully supported. GEMINI.md is Gemini CLI's native instruction file — same backlink approach as CLAUDE.md. Users can add their own instructions freely and SubFrame won't overwrite them.
**Codex CLI** gets SubFrame context via a wrapper script at `.subframe/bin/codex` that injects AGENTS.md as an initial prompt.
**This file (AGENTS.md)** contains SubFrame-specific rules that apply across all tools:
- Sub-Task management (`.subframe/tasks/*.md`, index at `.subframe/tasks.json`)
- Codebase mapping (`.subframe/STRUCTURE.json`)
- Context preservation (`.subframe/PROJECT_NOTES.md`)
- Internal docs and changelog (`.subframe/docs-internal/`)
- Session notes and decision tracking
---
## Session Start
**Read these files at the start of each session:**
1. **`.subframe/STRUCTURE.json`** — Module map, file locations, architecture notes
2. **`.subframe/PROJECT_NOTES.md`** — Project vision, past decisions, session notes
3. **`.subframe/tasks.json`** — Sub-task index (pending, in-progress, completed)
This gives you full project context before making any changes. The session-start hook (if configured) automatically injects pending/in-progress sub-tasks into your context, but you should still read these files for deeper understanding.
### Concurrent Work & Worktrees
Before making changes, check whether other AI sessions or agent teams are already working on this repository. Signs of concurrent work include:
- In-progress sub-tasks you didn't start (check `.subframe/tasks.json`)
- Recent uncommitted changes in `git status` that aren't yours
- Lock files or active worktrees (`git worktree list`)
**If concurrent work is detected**, ask the user: "Another session appears to be working on this project. Should I use a git worktree to avoid conflicts?"
**Git worktrees** create an isolated copy of the repo on a separate branch, allowing parallel work without merge conflicts:
- Each worktree has its own working directory and branch
- Changes in one worktree don't affect others until merged
- Use worktrees when multiple agents or sessions work on different features simultaneously
**When to suggest a worktree:**
- Agent teams spawning multiple workers on the same repo
- User asks to work on a feature while another is in progress
- The session-start hook flags concurrent sessions
**When worktrees are NOT needed:**
- Single-session work with no concurrent agents
- Read-only exploration or research tasks
- Quick fixes that won't conflict with in-progress work
---
## Hooks (Automatic Awareness)
SubFrame can configure project-level hooks that automate sub-task awareness. These hooks fire automatically — no manual intervention needed.
| Hook | When it fires | What it does |
|------|---------------|--------------|
| **SessionStart** | Startup, resume, after compaction | Injects pending/in-progress sub-tasks into context |
| **UserPromptSubmit** | Each user prompt | Fuzzy-matches prompt against pending sub-tasks, suggests starting a match |
| **Stop** | When AI finishes responding | Reminds about in-progress sub-tasks; flags untracked work if source files changed |
| **PreToolUse** | Before tool execution | Project-specific guardrails (if configured) |
| **PostToolUse** | After tool execution | Project-specific follow-ups (if configured) |
These hooks ensure sub-task awareness even after context compaction. Hook configuration lives in `.claude/settings.json`.
---
## Skills (Slash Commands)
SubFrame provides optional slash commands for AI coding tools that support them (e.g., Claude Code):
| Skill | Purpose |
|-------|---------|
| `/sub-tasks` | Interactive sub-task management — list, start, complete, add, archive |
| `/sub-docs` | Sync all SubFrame documentation after feature work (changelog, CLAUDE.md, PROJECT_NOTES, STRUCTURE) |
| `/sub-audit` | Code review + documentation audit on recent changes |
| `/onboard` | Bootstrap SubFrame files from existing codebase context |
Skills are deployed to `.claude/skills/` and enhance the workflow — but direct file editing always works as a fallback. If your AI tool doesn't support skills, follow the manual instructions in each section below.
---
## Sub-Task Management
> **Terminology:** "Sub-Tasks" are SubFrame's project task tracking system. The name plays on "Sub" from SubFrame and disambiguates from Claude Code's internal todo tools. When the user says "sub-task", they mean this system.
### Sub-Task File Format
Each sub-task lives in its own markdown file at `.subframe/tasks/<id>.md` with YAML frontmatter:
```yaml
---
id: task-abc12345
title: Short and clear title (max 60 characters)
status: pending | in_progress | completed
priority: high | medium | low
category: feature | fix | refactor | docs | test | chore
description: AI's detailed explanation — what, how, which files affected
userRequest: User's original prompt/request — copy exactly
acceptanceCriteria: When is this task done? Concrete testable criteria
blockedBy: [] # task IDs this depends on
blocks: [] # task IDs that depend on this
createdAt: ISO timestamp
updatedAt: ISO timestamp
completedAt: ISO timestamp | null
---
## Notes
[YYYY-MM-DD] Session notes, alternatives considered, dependencies.
## Steps
- [x] Completed step
- [ ] Pending step
```
A generated index is kept at `.subframe/tasks.json` for hooks and quick lookups. After creating or modifying task `.md` files, regenerate the index by reading all `.subframe/tasks/*.md` files (excluding `archive/`) and building the JSON with tasks grouped by status.
### Sub-Task Recognition Rules
**These ARE SUB-TASKS:**
- When the user requests a feature or change
- Decisions like "Let's do this", "Let's add this", "Improve this"
- Deferred work: "We'll do this later", "Let's leave it for now"
- Gaps or improvement opportunities discovered while coding
- Situations requiring bug fixes
**These are NOT SUB-TASKS:**
- Error messages and debugging sessions
- Questions, explanations, information exchange
- Temporary experiments and tests
- Work already completed and closed
- Instant fixes (like typo fixes)
### Sub-Task Creation Flow
1. Detect sub-task patterns during conversation
2. **Check existing sub-tasks first** — read `.subframe/tasks.json` to avoid duplicates
3. Ask the user: "I identified these sub-tasks from our conversation, should I add them?"
4. If approved, create `.subframe/tasks/<id>.md` with all required frontmatter fields
5. Regenerate the `.subframe/tasks.json` index
### Sub-Task Content Rules
**title:** Short, action-oriented
- OK: "Add tasks button to terminal toolbar"
- Bad: "Tasks"
**description:** AI's detailed technical explanation
- What will be done, how, which files affected
- Minimum 2-3 sentences
**userRequest:** User's original words — copy verbatim for context preservation
**acceptanceCriteria:** Concrete, testable completion criteria
### Sub-Task Status Updates
**Before starting any work**, check `.subframe/tasks.json` for an existing sub-task that matches. If found, set it to `in_progress` — do not create a duplicate.
- `pending` → `in_progress` — immediately when you begin working (update `updatedAt`)
- `in_progress` → `completed` — when done and verified (set `completedAt`, update `updatedAt`)
- `completed` → `pending` — when reopening, add a note explaining why
- After commit: check and update the status of all related sub-tasks
- **Incomplete work:** If partially done at session end, leave as `in_progress` and add a notes entry
### Sub-Task Lifecycle
- If a sub-task grows beyond its original scope, split it — create new sub-tasks and reference the parent ID in notes
- Cross-reference relevant commit hashes or PR numbers in notes
- Update the description if the approach changes significantly
### Priority Guidelines
- **high** — Blocking other work or explicitly flagged as urgent by the user
- **medium** — Normal feature work and standard bug fixes
- **low** — Nice-to-have improvements, deferred items, minor polish
---
## .subframe/PROJECT_NOTES.md Rules
### When to Update?
- When an important architectural decision is made
- When a technology choice is made
- When an important problem is solved and the solution method is noteworthy
- When an approach is determined together with the user
### Format
Free format. Date + title is sufficient:
```markdown
### [YYYY-MM-DD] Topic title
Conversation/decision as is, with its context...
```
### Update Flow
- Update immediately after a decision is made
- You can add without asking the user (for important decisions)
- You can accumulate small decisions and add them in bulk
### Organization Rules
- Keep **"Project Vision"** at the top, then **"Session Notes"** in chronological order
- Notes should capture the **why** (decisions, trade-offs, alternatives rejected), not the **what** (code structure belongs in STRUCTURE.json)
- When the same topic spans multiple sessions, consolidate related notes under the original heading rather than creating duplicates
- When notes grow beyond ~500 lines, consider archiving older session notes or grouping by month
---
## Context Preservation (Automatic Note Taking)
SubFrame's core purpose is to prevent context loss. Capture important moments and ask the user.
### When to Ask?
Ask the user: **"Should I add this to .subframe/PROJECT_NOTES.md?"** when:
- A sub-task is successfully completed
- An important architectural/technical decision is made
- A bug is fixed and the solution method is noteworthy
- "Let's do this later" is said (also add as a sub-task)
- A new pattern or best practice is discovered
### Importance Threshold
**Would it take more than 5 minutes to re-derive or re-explain in a future session?** If yes, capture it.
**Always capture:** Architecture decisions, technology choices, approach changes, user preferences discovered during work.
**Never capture:** Routine debugging steps, simple config changes, typo fixes.
**Note failed approaches too** — a brief "We tried X, it didn't work because Y" prevents future re-exploration of dead ends.
### Completion Detection
Pay attention to these signals:
- User approval: "okay", "done", "it worked", "nice", "fixed", "yes"
- Moving from one topic to another
- User continuing after build/run succeeds
### How to Add?
1. **DON'T write a summary** — Add the conversation as is, with its context
2. **Add date** — In `### [YYYY-MM-DD] Title` format
3. **Add to Session Notes section** — At the end of PROJECT_NOTES.md
### When NOT to Ask
- For every small change (it becomes spam)
- Typo fixes, simple corrections
- If the user already said "no" or "not needed", don't ask again for that topic
### If User Says "No"
No problem, continue. The user can also say what they consider important themselves: "add this to notes"
---
## .subframe/STRUCTURE.json Rules
**This file is the map of the codebase.**
### When to Update?
- When a new file/folder is created
- When a file/folder is deleted or moved
- When module dependencies change
- When an IPC channel is added or changed
- When an important architectural pattern is discovered (architectureNotes)
### Full Schema
```json
{
"modules": {
"main/moduleName": {
"file": "src/main/moduleName.ts",
"description": "What this module does",
"exports": ["init", "loadData"],
"depends": ["fs", "path", "shared/ipcChannels"],
"functions": {
"init": { "line": 15 },
"loadData": { "line": 42 }
}
}
},
"ipcChannels": {
"CHANNEL_NAME": {
"direction": "renderer → main",
"handler": "main/moduleName"
}
},
"architectureNotes": {
"topicName": {
"issue": "Description of the pattern or concern",
"solution": "How it was resolved"
}
}
}
```
### Update Rules
- The pre-commit hook (if configured) auto-updates STRUCTURE.json when source files in `src/` are committed
- When deleting files, remove their entries from `modules` and update any `depends` arrays that referenced them
- When adding IPC channels, also add them to the `ipcChannels` section with `direction` and `handler`
- `architectureNotes` is for **structural patterns** (e.g., circular dependency workarounds, init ordering). Use PROJECT_NOTES.md for **decisions and session context**
- If function line numbers drift significantly after edits, re-run the pre-commit hook or update manually
---
## .subframe/docs-internal/ Directory
This directory holds project documentation that doesn't belong in the root:
| File | Purpose |
|------|---------|
| `changelog.md` | Track changes under `## [Unreleased]`, grouped by Added/Changed/Fixed/Removed |
| `*.md` (ADRs) | Architecture Decision Records for significant design choices |
**What goes here:** Changelog entries, architecture decision records, internal reference docs.
**What does NOT go here:** User-facing docs (those go in `docs/` or project root), task files (those go in `.subframe/tasks/`).
---
## .subframe/QUICKSTART.md Rules
### When to Update?
- When installation steps change
- When new requirements are added
- When important commands change
---
## Before Ending Work
After significant work (code changes, architecture decisions), verify SubFrame files are in sync:
1. **Sub-Tasks** — Was this work tracked? Check `.subframe/tasks.json` → create/complete as needed
2. **PROJECT_NOTES.md** — Any decisions worth preserving? Ask the user
3. **Changelog** — Does `.subframe/docs-internal/changelog.md` reflect the changes?
4. **STRUCTURE.json** — Source files changed? The pre-commit hook handles this automatically if configured; otherwise update manually
The stop hook (if configured) will flag untracked work automatically.
---
## General Rules
1. **Language:** Write documentation in English (except code examples)
2. **Date Format:** ISO 8601 (YYYY-MM-DDTHH:mm:ssZ)
3. **After Commit:** Check sub-tasks (`.subframe/tasks/*.md`) and `.subframe/STRUCTURE.json`
4. **Session Start:** Read STRUCTURE.json, PROJECT_NOTES.md, and tasks.json before making changes
5. **Don't Duplicate:** Always check existing sub-tasks before creating new ones
---
*This file was automatically created by SubFrame.*
*Creation date: 2026-04-14*
<!-- subframe-template-version: 1 -->
+62 -1
View File
@@ -6,6 +6,66 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/), and this
## [Unreleased]
## [0.7.0] - 2026-05-19
### Added
- **Profile-aware Hermes sessions and voice settings.** Android now treats Hermes profiles as first-class connection state: profile selection resolves against the active server, profile-specific chat sessions are persisted separately, default/Victor display is normalized, and per-profile voice provider/model/voice settings can be read and saved through relay-owned endpoints without depending on Hermes config mutations.
- **Realtime voice playground and provider lab.** The relay now includes standalone OpenAI/xAI/ElevenLabs-oriented voice lab tooling, provider adapters, provider option discovery routes, realtime playground routes, and generated WAV/JSONL artifact ignores for iterative voice quality testing outside production Hermes routes.
- **Streaming voice output routes.** Relay-owned `/voice/output/*`, realtime playground, profile voice config, and provider option endpoints support provider-neutral TTS rendering, dynamic voice/model option surfaces, and profile-scoped voice configuration for Android.
- **Experimental Android realtime voice overlay.** Android adds a richer voice overlay with tap-to-talk, continuous mode controls, optional system overlay mode, compact mode, realtime waveform visualization, playback controls, and an experimental badge around barge-in instead of treating all voice as experimental.
- **Experimental realtime Hermes voice-agent plan.** `docs/plans/2026-05-19-realtime-hermes-voice-agent.md` records the next architecture step: provider-native realtime speech with Hermes-brokered profiles, sessions, tools, confirmations, and transcript mirroring. The stable Hermes chat + voice-output path remains the default.
- **Desktop tray pairing and consent flow.** The desktop surface gained Tauri tray pairing, QR/consent affordances, sidecar preparation, and computer-action approval polish so desktop and Android pairing flows are closer to parity.
- **Shared relay/Quest scaffolding.** Experimental `relay-core`, `relay-ui`, and Quest prototype modules were added for shared pairing, terminal, transport, voice, and morphing-sphere work without changing the Android phone app's default route.
- **Desktop Chat tab with first-run route setup.** The Tauri tray dashboard now has a Chat tab inspired by the Hermes Desktop chat-first flow. It streams through the saved paired relay when `~/.hermes/remote-sessions.json` has an active session, or through a direct Hermes gateway/API URL when relay pairing is not available. The tab supports stop, retry, new chat, clear, current-session transcript history, and a setup panel that offers relay pairing or direct WebAPI configuration without saving the optional API key.
- **First-class desktop TUI tab.** The Tauri tray dashboard now gives the embedded xterm/PTY Hermes session its own sidebar tab instead of nesting it under Terminal / CLI. Terminal remains the external launcher, shim-state, and copyable-command surface, while plugin embeds route into the same TUI tab.
- **Desktop surface plugins.** The desktop CLI and Tauri tray now register built-in terminal surface plugins, starting with Herm (`herm-tui`) from `liftaris/herm`. Users can inspect plugin status, install or update Herm, launch a fresh dashboard, resume with `herm -c`, or embed the plugin in the tray's xterm/PTY surface with `bunx`/`npx` fallback when the `herm` binary is not installed.
- **Relay server release track.** Relay server and Python package releases now use `relay-v*` tags, validate relay-owned version metadata, build wheel/sdist artifacts, generate checksums, and publish through `.github/workflows/release-relay.yml`. This lets Relay server fixes ship independently from Android app `versionCode` bumps and desktop CLI alphas.
- **Dashboard plugin CI.** `.github/workflows/ci-dashboard.yml` builds the dashboard plugin, runs the dashboard API tests, and verifies the plugin-owned QR modal CSS markers are present in the built bundle.
- **Upstream integration sync reference.** `docs/upstream-integration-sync.md` now tracks which Hermes-Relay surfaces use upstream-supported extension points, which pieces are relay-owned compatibility layers, and what has to be checked before changing relay, Android, desktop, dashboard, bootstrap, or user-doc surfaces.
- **Relay version sync verifier.** `scripts/check-relay-version-sync.py` validates the relay package version against plugin metadata and dashboard metadata so release and dashboard surfaces cannot silently drift.
### Changed
- **Stable voice is now the main Android voice path.** Voice mode defaults to Hermes chat streaming plus relay-managed voice output, with realtime-provider work kept as a standalone lab/testbench and future experimental mode instead of replacing Hermes session/tool authority.
- **Realtime voice output uses balanced coalescing.** Normal assistant speech is batched into more natural chunks while tool/status speech stays immediate, reducing provider render resets and tone/volume variation during voice replies.
- **Voice settings are profile-scoped and option-aware.** Android can fetch provider/model/voice options from relay endpoints, show profile context in voice settings, save voice choices per Hermes profile, and expose advanced manual entry when provider metadata is incomplete.
- **Voice UI state is synchronized with chat state.** Voice mode now reuses more of the chat session/profile state, preserves live transcript and tool timeline visibility, and improves overlay exit/minimize behavior for hands-free use.
- **Release versioning is split by surface.** Android app releases remain on `v*` and use `gradle/libs.versions.toml`; Relay releases use `relay-v*` and keep `pyproject.toml`, `plugin/relay/__init__.py`, `plugin/plugin.yaml`, and dashboard plugin metadata in lockstep; desktop remains on `desktop-v*` and `desktop/package.json`. `scripts/bump-version.sh` is now a backward-compatible Android alias, with new explicit `scripts/bump-android-version.sh` and `scripts/bump-relay-version.sh` helpers.
- **Upstream voice imports are isolated.** Relay voice routes now call upstream Hermes STT/TTS helpers through `plugin.relay.upstream_voice`, keeping private upstream voice helper imports in one adapter module until Hermes exposes a stable HTTP voice API.
- **CI paths and release actions tightened.** Relay CI now watches Relay-owned paths instead of all `plugin/**`, validates Relay version metadata during syntax checks, uses explicit timeouts, and runs the focused route/auth/session test slice instead of broad test discovery. Release workflows now use `softprops/action-gh-release@v3`.
### Fixed
- **Profile switching no longer silently falls back to the wrong local API host.** Profile API URL resolution now handles per-profile Hermes API servers, default/Victor compatibility, and relay-managed profile metadata so selecting a profile does not try to create sessions against `localhost` from the phone.
- **Non-default profile names remain visible in chat.** Agent display metadata is normalized so selected profile names persist above finalized assistant messages instead of disappearing back to the default label after stream completion.
- **Voice waveform and playback state are better aligned to real audio.** The output waveform waits for audio playback, handles processing separately, and avoids returning to the microphone too early at the end of an assistant response.
- **Continuous voice mode no longer starts a session just because auto mode is enabled.** Auto/continuous remains a preference, while explicit voice start/stop controls decide when a voice session is active.
- **Android voice mode no longer 403s when paired over plain-LAN `ws://` with a Hermes API key saved.** Symptom: tap the mic in Voice mode → red banner *"Voice access expired — extend or re-pair with voice grants"* even though the Connections card shows API Server / Relay / Session all green. Root cause: `RelayVoiceClient` preferred the saved Hermes API key over the paired Relay session token; the relay's `_request_is_secure_enough_for_api_bearer` correctly rejects API-bearer auth on `/voice/*` over plaintext outside loopback/Tailscale, returning a generic 403 that the client flattened to "expired." Fix: invert bearer precedence so paired devices use the session token first (no transport guard — it's the credential the QR/pair handshake already established), with the API key as fallback for chat+voice-only installs that never paired. `describeHttpError` now also reads the server's text/plain response body when present so future 403s show the relay's actual reason instead of a one-size-fits-all string.
## [0.6.1] - 2026-05-06
### Added
@@ -1069,6 +1129,7 @@ MVP release — native Android companion app for Hermes agent with direct API ch
- **Dev scripts** — build, install, run, test, relay via scripts/dev.bat
- **ProGuard rules** — okhttp-sse, markdown renderer, intellij-markdown parser
[Unreleased]: https://github.com/Codename-11/hermes-relay/compare/v0.1.0...HEAD
[Unreleased]: https://github.com/Codename-11/hermes-relay/compare/v0.7.0...HEAD
[0.7.0]: https://github.com/Codename-11/hermes-relay/compare/v0.6.1...v0.7.0
[0.1.0]: https://github.com/Codename-11/hermes-relay/compare/v0.1.0-beta...v0.1.0
[0.1.0-beta]: https://github.com/Codename-11/hermes-relay/releases/tag/v0.1.0-beta
+19 -10
View File
@@ -36,7 +36,7 @@
| Surface | What | Status |
|---------|------|--------|
| **[Android app](#1a-android-app)** | Native phone control — chat, voice, the agent reads your screen and acts on it (tap, type, swipe), notification companion, multi-Connection. | Available — Google Play (Internal testing) + sideload APK |
| **[Desktop CLI](#1b-desktop-cli-experimental)** | Use a server-deployed Hermes from your laptop **like it's local** — same shell, same TUI, same `Win+Shift+S` → `Ctrl+A v` paste flow, same conversation continuity. The remote agent can also reach back through the relay and run tools on YOUR machine. | **Experimental** — `desktop-v0.3.0-alpha.14` (one-binary install, no Node required) |
| **[Desktop app + CLI](#1b-desktop-app--cli-experimental)** | Use a server-deployed Hermes from your laptop **like it's local**. Windows gets the native tray app first: pair, start/pause the daemon, view devices, task log, settings, overlay status, and emergency stop. The CLI remains the terminal/headless surface and powers macOS/Linux installs. Experimental computer-use tools are opt-in. | **Experimental** — `desktop-v0.3.0-alpha.18` (Windows tray installer + native CLI binaries, no Node required) |
Both share `~/.hermes/remote-sessions.json` and the same WSS relay. **Pair once from either, both work.**
@@ -68,17 +68,25 @@ Full walkthrough, including signing-certificate fingerprint: [Sideload guide](ht
**Staying up to date (sideload):** the app checks GitHub for a newer release on cold start (at most once every 6 hours) and shows a dismissable banner when you're behind. Tapping **Update** opens the next APK in your browser so Android's Downloads notification hands it to the system installer — no second app required. You can also trigger a check manually under **Settings → About → Updates**. Google Play installs get auto-updates through the Play Store and don't show this banner.
### 1b. Desktop CLI (experimental)
### 1b. Desktop app + CLI (experimental)
A single-binary thin client (`hermes-relay`) that talks to a server-deployed Hermes over WSS — same shell, same TUI, same `/paste` flow as a local install. The remote agent can also reach back through the relay and run `desktop_read_file`, `desktop_terminal`, `desktop_search_files`, `desktop_screenshot`, `desktop_clipboard_*`, `desktop_open_in_editor`, etc. **on your machine** while its brain stays on the host. One pair, two surfaces (with the Android app), no `ssh`.
The desktop surface talks to a server-deployed Hermes over WSS. On Windows, the default installer launches the native tray app with pairing, daemon control, devices, task log, settings, overlay status, pause, and emergency stop. The same release still ships the `hermes-relay` CLI for shell/TUI use, scripting, headless daemon mode, and macOS/Linux.
**Install** (Windows PowerShell):
The remote agent can also reach back through the relay and run `desktop_read_file`, `desktop_terminal`, `desktop_search_files`, `desktop_screenshot`, `desktop_clipboard_*`, `desktop_open_in_editor`, etc. **on your machine** while its brain stays on the host. One pair, two surfaces (with the Android app), no `ssh`.
**Install tray app** (Windows PowerShell):
```powershell
irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
```
**Install** (macOS / Linux):
**Install CLI only** (Windows PowerShell):
```powershell
$env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
```
**Install CLI** (macOS / Linux):
```bash
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.sh | sh
@@ -95,9 +103,9 @@ hermes-relay update # self-update via GitHub Releases
**Native paste workflow** (the killer demo): inside `hermes-relay shell`, hit `Win+Shift+S` to screenshot, then `Ctrl+A v` — the client reads your clipboard, ships the image to the server's inbox, and types `/paste` into the TUI for you. Identical UX to native local-Hermes paste. The same chord set works on macOS (`Cmd+Shift+4` → `Ctrl+A v`) and Linux (Wayland/X11 detected automatically).
**No Node required** — Bun-compiled native binaries (~60–110 MB per platform) installed via curl/irm. Version-aware install (`upgrading X → Y`), collision-safe `hermes` short alias, self-update via `hermes-relay update`. Binaries are **unsigned** during the experimental phase — SmartScreen/Gatekeeper warnings are expected; the install scripts show the one-line escape hatches. Code signing, multi-client server-side routing, and service installers (sc.exe / systemd / launchd) land with v1.0.
**No Node required** — the Windows tray installer bundles the compiled CLI sidecar; CLI-only installs use Bun-compiled native binaries (~60–110 MB per platform) via curl/irm. Version-aware install (`upgrading X → Y`), collision-safe `hermes` short alias for CLI installs, self-update via `hermes-relay update`. Assets are **unsigned** during the experimental phase — SmartScreen/Gatekeeper warnings are expected. Code signing, multi-client server-side routing, and service installers (sc.exe / systemd / launchd) land with v1.0.
- **Docs**: [Desktop CLI guide](https://codename-11.github.io/hermes-relay/desktop/) · [`desktop/README.md`](desktop/README.md)
- **Docs**: [Desktop guide](https://codename-11.github.io/hermes-relay/desktop/) · [`desktop/README.md`](desktop/README.md)
- **Release track**: tagged `desktop-v*`, [separate from Android](https://github.com/Codename-11/hermes-relay/releases?q=desktop)
- **AI-agent setup recipe**: `/hermes-relay-desktop-setup` (the agent can run `desktop_terminal` on your machine to diagnose install/pair issues live)
@@ -178,7 +186,7 @@ See the [changelog](CHANGELOG.md) for the full list.
- **Streaming chat** — Direct SSE to the Hermes API Server with real-time markdown rendering, session history, tool-call visualization, personality picker, searchable command palette (29+ gateway commands), file attachments, and send-while-streaming message queuing
- **Multi-Connection + agent profiles** — Pair with multiple Hermes servers and switch targets from the top bar; select an upstream-discovered agent profile to overlay model + `SOUL.md` on chat turns. Three-layer model: Connection (server) → Profile (agent directory) → Personality (prompt preset)
- **Voice mode** — Real-time voice conversation via the relay; the sphere listens with you and performs the agent's reply as it speaks. Uses your server's configured TTS/STT providers (Edge TTS, ElevenLabs, OpenAI, MiniMax, Mistral, NeuTTS / faster-whisper, Groq, OpenAI Whisper)
- **Voice mode** — Experimental server-mediated voice conversation via the relay; the sphere listens with you and performs the agent's reply as it speaks. Hermes owns chat, tool calls, and approvals, while relay voice output defaults to provider-neutral streaming TTS (`xai_tts` first) with realtime voice-agent providers kept as a separate lab mode.
- **Phone control (bridge)** — The agent can read what's on screen and act on it — tap, long-press, drag, swipe, scroll, type, and press system keys — plus take screenshots, read/write the clipboard, and control system-wide media playback. Gesture reliability is hardened for dim/idle screens, and a smarter tap-fallback cascade handles apps where labels sit inside non-clickable wrappers
- **Screen understanding** — Filtered accessibility-tree search, per-node property lookups with stable IDs, cheap screen-hash change detection, and multi-window reads (system overlays, popups, notification shade) so the agent can reason about UI without guessing
- **Workflow automation** — Batched macro execution for multi-step flows, real-time accessibility event streaming for "wait until something happens" waits, and a raw-Intent escape hatch for apps that expose deep-link actions
@@ -193,7 +201,7 @@ See the [changelog](CHANGELOG.md) for the full list.
- **Shell mode (default)** — bare `hermes-relay` pipes the host's actual `hermes` Ink TUI through a PTY in tmux. Same banner, same skin, same slash commands as a local install. `Ctrl+A .` detaches (preserves tmux), `Ctrl+A k` kills, `Ctrl+A v` pastes a clipboard image, `Ctrl+A ?` re-prints chord help, `Ctrl+A Ctrl+A` literal.
- **Chat mode** — REPL or one-shot or piped stdin. `--json` emits `GatewayEvent`s per line for `jq` / automation. REPL slash commands `/paste` (clipboard), `/screenshot` (multi-monitor by default; `primary` / `1` / `2` to narrow), `/image <path>` attach the next message.
- **Local tool routing** — agent calls `desktop_read_file`, `desktop_write_file`, `desktop_terminal`, `desktop_search_files`, `desktop_patch`, `desktop_clipboard_read/write`, `desktop_screenshot`, `desktop_open_in_editor` — all run on YOUR machine over the same WSS relay. One-time per-URL consent gate; `--no-tools` kill-switch; non-TTY stdin fails closed; agent-proposed patches render as colored diffs with `y/n/e/r` interactive approval.
- **Local tool routing** — agent calls `desktop_read_file`, `desktop_write_file`, `desktop_terminal`, `desktop_search_files`, `desktop_patch`, `desktop_clipboard_read/write`, `desktop_screenshot`, `desktop_open_in_editor` — all run on YOUR machine over the same WSS relay. One-time per-URL consent gate; `--no-tools` kill-switch; non-TTY stdin fails closed; agent-proposed patches render as colored diffs with `y/n/e/r` interactive approval. Experimental `desktop_computer_*` control tools require `--experimental-computer-use` / `HERMES_RELAY_EXPERIMENTAL_COMPUTER_USE=1`, task-scoped grants, and visible local approval.
- **Daemon mode** — `hermes-relay daemon` runs the tool router headless so the agent can reach you even when no shell is open. JSON-line lifecycle logs by default, auto-human on TTY. Fails closed on missing consent.
- **Self-update** — `hermes-relay update` polls GitHub Releases (SemVer-max picker, prerelease-aware), verifies SHA256, atomic-swaps the binary on POSIX (running daemon keeps inode), cooperative `.new.exe` swap on Windows.
- **Multi-endpoint pairing + reconnect-on-drop + TOFU cert pinning** — same as the Android app. One QR carries LAN + Tailscale + public; client races candidates in priority order, re-probes on every network change.
@@ -239,7 +247,8 @@ Chat from the Android app connects directly to the Hermes API Server with the He
| [API Reference](https://codename-11.github.io/hermes-relay/reference/api.html) | Hermes API endpoints used by both surfaces |
| [Specification](docs/spec.md) | Full spec — protocol, UI, phases, dependencies |
| [Architecture Decisions](docs/decisions.md) | ADRs — framework, channels, auth, terminal |
| [Changelog](CHANGELOG.md) | Release history (Android `v*` and desktop `desktop-v*`) |
| [Upstream Integration Sync](docs/upstream-integration-sync.md) | Supported Hermes extension points vs relay-owned compatibility layers |
| [Changelog](CHANGELOG.md) | Release history (Android `v*`, Relay `relay-v*`, and desktop `desktop-v*`) |
---
+149 -62
View File
@@ -3,7 +3,7 @@
> The full recipe for cutting a new release. Read this end-to-end before
> tagging your first release.
## Versioning
## Release Tracks And Versioning
Hermes-Relay follows [SemVer](https://semver.org/): `MAJOR.MINOR.PATCH`,
with optional prerelease identifiers.
@@ -13,6 +13,20 @@ with optional prerelease identifiers.
- `PATCH` — bug fixes, backwards compatible
- Prerelease suffixes: `-alpha`, `-beta`, `-rc.N` (e.g. `0.2.0-beta.1`)
Hermes-Relay now ships three independently versioned surfaces:
| Surface | Tag prefix | Version source | Bump script | Release workflow |
|---|---|---|---|---|
| Android app | `v*` | `gradle/libs.versions.toml` | `scripts/bump-android-version.sh` | `.github/workflows/release.yml` |
| Relay server / Python package | `relay-v*` | `pyproject.toml` plus checked plugin/dashboard metadata | `scripts/bump-relay-version.sh` | `.github/workflows/release-relay.yml` |
| Desktop CLI | `desktop-v*` | `desktop/package.json` | `npm version` or manual package bump | `.github/workflows/release-desktop.yml` |
This split is intentional. The Relay server now carries features for both
Android and desktop, so Relay fixes can ship without forcing an Android app
versionCode bump, and desktop CLI alphas can continue on their own cadence.
### Android app versioning
**Source of truth:** `gradle/libs.versions.toml`
```toml
@@ -44,30 +58,44 @@ Never decrement `appVersionCode` — Play Console rejects any upload whose
code is lower than or equal to a previous upload on the same track. Confirm
current values with `scripts\dev.bat version`.
### The three version sources (MUST stay in lockstep)
There are **three** places the version lives, and they MUST all match on
every release commit. Drift is silent and painful — we chased a "why does
/health say 0.2.0" bug for hours on 2026-04-12 because `pyproject.toml`
had drifted to `0.5.0` speculatively and `plugin/relay/__init__.py` was
still at a stale `0.2.0`.
| File | Line | Written by |
|---|---|---|
| `gradle/libs.versions.toml` | `appVersionName = "…"` | You (canonical) |
| `pyproject.toml` | `version = "…"` | You (Python package) |
| `plugin/relay/__init__.py` | `__version__ = "…"` | You (runtime, reported by `/health`) |
**Always bump them atomically via `scripts/bump-version.sh`**:
Always bump Android releases via:
```bash
bash scripts/bump-version.sh 0.3.0
bash scripts/bump-android-version.sh 0.6.2
```
The script validates SemVer, bumps `appVersionCode` monotonically, rewrites
all three files, runs a post-bump sanity grep, prints the diff, and tells
you the next steps. It deliberately does NOT commit, tag, or touch
`CHANGELOG.md` / `RELEASE_NOTES.md` — those need human prose.
`scripts/bump-version.sh` remains as a backward-compatible alias for the
Android script.
### Relay server / Python package versioning
Relay version metadata lives in these relay-owned files and must stay in
lockstep:
| File | Line | Purpose |
|---|---|---|
| `pyproject.toml` | `version = "..."` | Python package metadata |
| `plugin/relay/__init__.py` | `__version__ = "..."` | runtime version reported by `/health` |
| `plugin/plugin.yaml` | `version: ...` | Hermes plugin metadata |
| `plugin/dashboard/manifest.json` | `"version": "..."` | Hermes dashboard plugin metadata |
| `plugin/dashboard/package.json` | `"version": "..."` | dashboard build/package metadata |
| `plugin/dashboard/package-lock.json` | `"version": "..."` | locked dashboard package metadata |
Always bump Relay releases via:
```bash
bash scripts/bump-relay-version.sh 0.6.2
```
Check the current metadata with:
```bash
python scripts/check-relay-version-sync.py
```
The `relay-v*` release workflow validates the tag against the same metadata,
runs Relay tests, builds a wheel and sdist, generates checksums, and publishes
a GitHub Release with the package artifacts.
## Branching policy
@@ -125,16 +153,17 @@ Squash merges lose that detail and are **not** the house style.
### Version bumps happen at release-prep on `dev`, NOT on feature branches
Feature branches **never** touch `gradle/libs.versions.toml`,
`pyproject.toml`, or `plugin/relay/__init__.py`. If two feature branches
both bumped the version, they'd collide on `appVersionCode` (which must
be monotonic) and you'd hit a merge conflict for no good reason.
relay-owned version metadata, or `desktop/package.json`.
If two feature branches both bumped a release version, they'd collide on
version files and, for Android, on `appVersionCode` (which must be
monotonic).
The version-bump commit lives on `dev`, created via
`scripts/bump-version.sh`, as the last commit of the release-prep work.
It's a dedicated commit with the message `release: vX.Y.Z` that also
lands the CHANGELOG and RELEASE_NOTES updates. A release PR then merges
`dev` → `main` with `--no-ff`, and the `v<version>` tag is cut from the
resulting `main` tip.
Version-bump commits live on `dev` as the last commit of release-prep
work. Android commits use `release: vX.Y.Z`; Relay commits use
`release(relay): relay-vX.Y.Z`; desktop commits use the existing
`release: desktop-vX.Y.Z` convention. A release PR then merges `dev` →
`main` with `--no-ff`, and the matching tag is cut from the resulting
`main` tip.
### Branch protection
@@ -261,7 +290,8 @@ to Play Console. Manual UI uploads work without this.
### 4. GitHub Actions secrets
In the repo: **Settings > Secrets and variables > Actions > New repository
secret.** Add all four (see the table in "Required GitHub Secrets" below).
secret.** Add all four (see the table in "Required Android Release Secrets"
below).
If `HERMES_KEYSTORE_BASE64` is missing, CI release builds fall back to
debug signing and print a warning in the workflow summary — those
@@ -294,15 +324,14 @@ the unstable build.
## Release Process
### 1. Bump the version (atomic across all three sources)
### 1. Bump the Android app version
Use `scripts/bump-version.sh` — it rewrites `libs.versions.toml`,
`pyproject.toml`, AND `plugin/relay/__init__.py::__version__` in lockstep,
increments `appVersionCode` monotonically, and runs a sanity check. Don't
edit the files by hand; drift is silent and painful.
Use `scripts/bump-android-version.sh`. It rewrites
`gradle/libs.versions.toml`, increments `appVersionCode` monotonically,
and runs a sanity check. Don't edit the Android version files by hand.
```bash
bash scripts/bump-version.sh 0.3.0
bash scripts/bump-android-version.sh 0.6.2
```
Confirm the bump:
@@ -311,8 +340,8 @@ Confirm the bump:
scripts\dev.bat version
```
The script's diff output should show exactly three files changed and all
three carrying the new version string.
The script's diff output should show `gradle/libs.versions.toml` carrying
the new app version and a higher `appVersionCode`.
### 2. Update release notes and changelog
@@ -374,28 +403,54 @@ resulting merge commit on `main`:
git checkout dev
git pull --ff-only origin dev
git add gradle/libs.versions.toml pyproject.toml plugin/relay/__init__.py \
RELEASE_NOTES.md CHANGELOG.md \
git add gradle/libs.versions.toml RELEASE_NOTES.md CHANGELOG.md \
app/src/main/assets/whats_new.txt docs/play-store-listing.md
git commit -m "release: v0.3.0"
git commit -m "release: v0.6.2"
git push origin dev
# Open the release PR (dev -> main) and merge with --no-ff.
# After merge, tag from the new main tip:
git checkout main
git pull --ff-only origin main
git tag v0.3.0
git push origin v0.3.0
git tag v0.6.2
git push origin v0.6.2
```
Pushing a tag matching `v*` triggers `.github/workflows/release.yml`,
which builds, signs, checksums, and creates a GitHub Release. Watch the
run under the **Actions** tab.
> **Why all three files in the commit?** See "The three version sources"
> above — `bump-version.sh` rewrites them atomically, so they must be
> staged + committed atomically too. Missing one creates the same drift
> the script was built to prevent.
Relay/Python version files are intentionally not part of an Android app
release unless the Relay package itself is also being released.
### Relay server / Python package release
Use this when Relay server behavior changes independently of Android app
delivery, for example desktop channel support, bridge routes, pairing
server fixes, voice auth, or packaging changes.
```bash
git checkout dev
git pull --ff-only origin dev
bash scripts/bump-relay-version.sh 0.6.2
git add pyproject.toml plugin/relay/__init__.py plugin/plugin.yaml plugin/dashboard/manifest.json plugin/dashboard/package.json plugin/dashboard/package-lock.json CHANGELOG.md
git commit -m "release(relay): relay-v0.6.2"
git push origin dev
# Open the release PR (dev -> main) and merge with --no-ff.
# After merge, tag from the new main tip:
git checkout main
git pull --ff-only origin main
git tag relay-v0.6.2
git push origin relay-v0.6.2
```
Pushing `relay-v*` triggers `.github/workflows/release-relay.yml`, which
validates all relay-owned version metadata with
`scripts/check-relay-version-sync.py`, runs Relay tests, builds a wheel and
sdist, generates `SHA256SUMS.txt`, and creates a GitHub Release for the Relay
package.
### 5. Upload to Play Console
@@ -462,14 +517,20 @@ Promote via the Play Console UI or `gradlew promoteReleaseArtifact`.
## CI Behavior
Android, Relay, dashboard, and desktop now have separate CI/release lanes.
This keeps a dashboard CSS fix from running the full Relay suite, and keeps
Relay server changes from forcing an Android app `versionCode` bump.
On every push of a tag matching `v*`, `.github/workflows/release.yml`:
1. Validates the tag matches `appVersionName` in
`gradle/libs.versions.toml` (mismatches fail the workflow).
2. Runs `./gradlew assembleDebug` and `./gradlew test`.
2. Runs the Android debug build and the stable sideload pairing/connection
regression slice with explicit timeouts.
3. Decodes `HERMES_KEYSTORE_BASE64` into `$RUNNER_TEMP/release.keystore`
and exports `HERMES_KEYSTORE_PATH` (skipped if the secret is unset).
4. Builds both artifacts: `./gradlew bundleRelease assembleRelease`.
4. Builds both Android release artifacts:
`./gradlew bundleRelease assembleRelease`.
5. Generates `SHA256SUMS.txt` covering both.
6. Creates a GitHub Release named `v<version>` with `RELEASE_NOTES.md` as
the body. Attaches the APK, AAB, and `SHA256SUMS.txt`. Tags any version
@@ -478,7 +539,25 @@ On every push of a tag matching `v*`, `.github/workflows/release.yml`:
succeeded. If `HERMES_KEYSTORE_BASE64` is missing, the summary warns
that the artifacts are debug-signed and unsuitable for Play Store.
## Required GitHub Secrets
On every push of a tag matching `relay-v*`,
`.github/workflows/release-relay.yml`:
1. Validates the tag matches all relay-owned version metadata checked by
`scripts/check-relay-version-sync.py`.
2. Runs Relay syntax checks and the focused route/auth/session test slice.
3. Builds the Python wheel and sdist with `python -m build`.
4. Generates `dist/SHA256SUMS.txt`.
5. Creates a GitHub Release named `relay-v<version>` with the wheel,
sdist, and checksum file attached.
On every push of a tag matching `desktop-v*`,
`.github/workflows/release-desktop.yml` builds and publishes the desktop
CLI binaries. Dashboard-only changes are covered by
`.github/workflows/ci-dashboard.yml`, which builds the dashboard plugin,
runs the dashboard API tests, and verifies the modal CSS markers are present
in the built bundle.
## Required Android Release Secrets
| Secret | Purpose | How to populate |
|-----------------------------|-------------------------------------|--------------------------------------------------|
@@ -490,24 +569,32 @@ On every push of a tag matching `v*`, `.github/workflows/release.yml`:
## Hotfix Recipe
When production has a bug and you need to ship a fix without picking up
unreleased work from `dev`:
unreleased work from `dev`, branch from the affected release tag and only
bump the version source for the surface you are shipping.
1. `git checkout -b fix/short-name v0.1.0` — branch from the released
tag (not from `main` or `dev`).
For an Android app hotfix:
1. `git checkout -b fix/short-name v0.6.1` — branch from the released
Android tag (not from `main` or `dev`).
2. Apply the fix, add a test, commit.
3. Bump `appVersionName` and `appVersionCode` in
`gradle/libs.versions.toml` (and the other two version sources via
`scripts/bump-version.sh`).
4. Update `RELEASE_NOTES.md` and `CHANGELOG.md`.
3. Run `bash scripts/bump-android-version.sh 0.6.2` to update
`gradle/libs.versions.toml`.
4. Update `RELEASE_NOTES.md`, `CHANGELOG.md`, in-app What's New, and Play
listing notes as needed.
5. Open a PR from `fix/short-name` into `main`, merge with `--no-ff`.
6. `git tag v0.1.1` from the new `main` tip and `git push origin v0.1.1`
— CI builds and publishes.
6. `git tag v0.6.2` from the new `main` tip and `git push origin v0.6.2`
so Android release CI builds and publishes.
7. Upload to Play Console as normal.
8. Merge `main` back into `dev` (`git checkout dev && git merge --no-ff main`)
so `dev` picks up the hotfix and the version bumps. Without this,
`dev`'s `appVersionCode` lags behind `main` and the next release
so `dev` picks up the hotfix and the versionCode bump. Without this,
`dev`'s `appVersionCode` lags behind `main` and the next app release
bump collides.
For a Relay server hotfix, branch from the affected `relay-v*` tag, apply
the fix, run `bash scripts/bump-relay-version.sh <next-version>`, merge to
`main`, and tag `relay-v<next-version>`. Do not touch
`gradle/libs.versions.toml` unless an Android app release is also shipping.
## Troubleshooting
**`Tag version (X) does not match appVersionName (Y)` in CI validate step**
+45 -33
View File
@@ -1,22 +1,22 @@
# Hermes-Relay v0.6.1
# Hermes-Relay v0.7.0
**Release Date:** May 6, 2026
**Since v0.6.0:** API-key voice auth, durable pairing/session recovery, route failover hardening, dashboard pairing UI polish, and full Android bridge media/MMS handoff support.
**Release Date:** May 19, 2026
**Since v0.6.1:** profile-aware chat/voice state, relay-owned voice provider settings, realtime voice lab/testbench routes, Android voice overlay polish, and Relay package voice-provider support.
v0.6.1 is a compatibility and bridge-contract release. It keeps the v0.6.0 multi-connection model, then tightens the real-world paths we just exercised: LAN/Tailscale pairing, API-key voice mode, stale session recovery, and Android phone bridge tools.
v0.7.0 is a minor release for the profile and voice workstream. The stable Android path is still Hermes chat streaming plus relay-managed voice output, while realtime provider work remains isolated as a lab/testbench and planned experimental mode.
---
## Download
v0.6.1 ships in two build flavors. APK and AAB filenames are version-tagged:
v0.7.0 ships in two Android build flavors. APK and AAB filenames are version-tagged:
| Flavor | File | Who it's for |
|---|---|---|
| sideload | `hermes-relay-0.6.1-sideload-release.apk` | Recommended for full bridge/device-control features, including share/MMS/file handoff. Installs as `com.axiomlabs.hermesrelay.sideload`. |
| Google Play | `hermes-relay-0.6.1-googlePlay-release.aab` | Conservative Play-track build for chat and voice without sideload-only bridge-control surfaces. |
| googlePlay APK | `hermes-relay-0.6.1-googlePlay-release.apk` | Parity/testing artifact. |
| sideload AAB | `hermes-relay-0.6.1-sideload-release.aab` | Parity/testing artifact. |
| sideload | `hermes-relay-0.7.0-sideload-release.apk` | Recommended for full bridge/device-control features, voice overlay testing, and profile-aware relay features. Installs as `com.axiomlabs.hermesrelay.sideload`. |
| Google Play | `hermes-relay-0.7.0-googlePlay-release.aab` | Conservative Play-track build for chat, profiles, and voice without sideload-only bridge-control surfaces. |
| googlePlay APK | `hermes-relay-0.7.0-googlePlay-release.apk` | Parity/testing artifact. |
| sideload AAB | `hermes-relay-0.7.0-sideload-release.aab` | Parity/testing artifact. |
Verify integrity with `SHA256SUMS.txt` from the same release. See the [Sideload guide](https://codename-11.github.io/hermes-relay/guide/getting-started.html#sideload-apk) for install steps.
@@ -24,41 +24,53 @@ Verify integrity with `SHA256SUMS.txt` from the same release. See the [Sideload
## Highlights
### Voice auth and pairing durability
### Profile-aware Hermes use
- Voice routes now accept either Relay session voice grants or a saved Hermes API bearer token for `/voice/config`, `/voice/transcribe`, and `/voice/synthesize`.
- Chat and voice can work from manual API URL + API key setup without a full QR pairing. Bridge, terminal, Android control, media, clipboard, and pair-session features still use the Relay session path.
- Relay sessions can recover through trusted device refresh after server restarts/updates, reducing forced re-pairing.
- Pairing and endpoint resolution were hardened for LAN/Tailscale transitions, including route normalization and clearer VPN fallback behavior.
- Profile selection now resolves against the active server and keeps profile-specific chat sessions separate.
- Default/Victor display is normalized so the selected profile name stays visible through streamed and finalized messages.
- Session drawer and voice settings can reflect the active profile instead of treating every connection as one shared default context.
- Profile API URL resolution handles per-profile Hermes API servers and avoids phone-side `localhost` fallbacks when a remote profile is selected.
### Android bridge media, files, and MMS handoff
### Voice settings and output quality
- Added `android_share_media` for text, files, relay media tokens, screenshots, and attachment lists through Android's native share sheet.
- Added `android_send_mms` for MMS compose/share handoff with body text and attachments.
- Added the missing `/return_to_hermes` relay route and `android_return_to_hermes` tool so docs, tools, relay HTTP, and the phone command contract match.
- `android_send_sms` remains text-only and now returns structured states such as `sent`, `blocked`, `awaiting_confirmation`, `timeout`, and `failed`.
- The active plugin import now uses `plugin.tools.android_tool` as the single source of truth, preventing Hermes tool exposure drift.
- Relay now owns profile voice configuration endpoints instead of depending on Hermes config edits for realtime voice settings.
- Android can fetch provider/model/voice option metadata, save per-profile voice choices, and fall back to advanced manual entry when a provider cannot expose a complete option list.
- Voice output uses balanced coalescing: assistant speech is grouped into natural chunks, while tool/status speech remains immediate.
- Waveform and playback state are better aligned to real audio output, reducing premature mic return and output-state jitter.
- Barge-in remains explicitly experimental, with known self-capture limitations documented in settings.
### Dashboard and operator flow
### Realtime provider lab and Relay package
- Dashboard-minted pairing QRs and the QR modal styling were corrected so the plugin owns its dialog layout instead of inheriting host dashboard card styling.
- The relay CLI can toggle insecure LAN API-key voice auth at runtime with `hermes relay insecure-api-key status|on|off`.
- Docs now spell out the bridge HTTP routes, SMS schema, `current_app` limitations, share/MMS handoff behavior, sideload-only restrictions, and direct HTTP fallback route contract.
- Added standalone voice lab CLI/TUI tooling, provider adapters, waveform/playback support, evaluation helpers, and generated WAV/JSONL artifact ignores for OpenAI, xAI, ElevenLabs, and stub testing.
- Added relay routes for streaming voice output, realtime playground calls, provider options, and profile voice config.
- Added a plan for the next experimental Realtime Hermes Voice Agent mode, where providers handle speech but Hermes remains the authority for profiles, sessions, memory, tools, confirmations, and transcript history.
### Android voice UI polish
- Voice mode includes better tap-to-talk, continuous-mode, overlay, compact-mode, and state-display behavior.
- Continuous mode no longer starts a session solely because the preference is enabled; voice sessions start and stop through explicit controls.
- Voice overlay state is closer to chat state, including live transcript/tool timeline surfaces without forcing an exit and reload.
### Included groundwork
- Desktop tray pairing and consent-flow improvements are included from the dev branch.
- Experimental shared `relay-core`, `relay-ui`, and Quest prototype modules are included for future shared pairing/terminal/voice work. They do not change the Android phone app's default flow.
---
## Verification
- Relay/tool regression slice: `134 passed`
- Android Kotlin compile: `:app:compileSideloadDebugKotlin` passed
- Android Kotlin compile: `:app:compileGooglePlayDebugKotlin` passed
- Dashboard build passed before deployment
- Remote staging deploy restarted `hermes-relay` and `hermes-gateway`, both active
- Relay version metadata: `python scripts/check-relay-version-sync.py --expect 0.7.0` passed.
- Android version metadata: `scripts\dev.bat version` reported `Hermes-Relay v0.7.0 (versionCode 9)`.
- Relay route/auth/session/provider slice: 99 pytest tests passed.
- Voice lab provider/tooling slice: 31 pytest tests passed.
- Android sideload and Google Play Kotlin compile passed.
- Focused Android voice/profile unit slice and release-CI unit slice passed.
## Post-install smoke
- Install the sideload APK over the existing sideload app with `adb install -r`.
- Existing pairing should survive a normal same-flavor update. Re-pair only if you uninstall app data, switch flavor/applicationId, revoke the device, or the server session store was intentionally cleared.
- Confirm chat works over the selected route, then test voice with the saved Hermes API key.
- For bridge media: try `android_share_media` or `POST /share_media` with a relay media token or host file path. The phone should show an on-device confirmation and then Android's native share UI.
- For MMS: `android_send_mms` opens the composer with the attachment; it does not silently send MMS.
- Existing pairing should survive a same-flavor update. Re-pair only if you uninstall app data, switch flavor/applicationId, revoke the device, or intentionally clear the server session store.
- Confirm the profile selector shows the expected server default and named profiles, then create/switch a chat while watching that the agent name remains stable.
- In Voice settings, confirm the selected profile's provider/model/voice options load, save a per-profile voice, and run a short voice test.
- In Voice mode, test tap-to-talk first, then continuous mode. Leave barge-in off unless you are explicitly testing the experimental self-capture path.
+4 -1
View File
@@ -30,7 +30,9 @@ Release tags: `desktop-v*` (separate cadence from Android `v*`). Curl-installed
**Active — `desktop-v0.3.0-alpha.7` (native image paste):** Plan at [`docs/plans/2026-04-23-desktop-alpha-7-native-paste.md`](docs/plans/2026-04-23-desktop-alpha-7-native-paste.md). Two-repo workstream: client slash commands `/paste` (clipboard), `/screenshot` (primary display), `/image <path>` (file) land in `hermes-relay chat`, each echoes a one-line feedback and attaches the image to the next `prompt.submit` so the vision-capable model sees it in the same turn — parity with Claude Desktop's paste UX minus OS-level Ctrl+V (terminals don't pipe image bytes to stdin). Client half is new `desktop/src/chatAttach.ts` + slash-command branches in `desktop/src/commands/chat.ts`. Server half is ONE new `@method("image.attach.bytes")` on the fork's `tui_gateway/server.py` (branch `feat/image-attach-bytes` → merged to `axiom`); the fork's existing `_enrich_with_attached_images` already handles multimodal payload plumbing and session-scoped image state, so this release is almost entirely about bridging client-captured bytes to server-side state that's been there for months. Relay channel unchanged — `tui` is a transparent RPC forwarder. Graceful fallback when hermes-host hasn't been updated yet: client catches `method not found`, prints a pointer at the axiom rollout, REPL stays alive.
**Active — experimental desktop computer-use MVP:** Plan at [`docs/plans/desktop-computer-use-mvp.md`](docs/plans/desktop-computer-use-mvp.md). Current slice extends the existing desktop tool channel with `desktop_computer_*` contracts for status, screenshot observe mode, grant request/cancel, and Windows-only action execution behind desktop-tool consent and a visible, task-scoped grant approval. It does not implement unrestricted or silent mouse/keyboard automation and does not depend on npm publication.
**Active — desktop control / computer-use:** Enhanced plan at [`docs/plans/desktop-control-computer-use-enhanced.md`](docs/plans/desktop-control-computer-use-enhanced.md); earlier MVP implementation record at [`docs/plans/desktop-computer-use-mvp.md`](docs/plans/desktop-computer-use-mvp.md). Windows now has the first Tauri tray/overlay app as the primary Easy/Standard install surface: pair, start/pause daemon, Devices/Revoke, Task Log, Settings, overlay status chip, emergency stop, and bundled CLI sidecar. The existing CLI and daemon remain the primary advanced/headless surface. `desktop_computer_*` schemas are registered on the normal desktop tool channel but advertised only behind the explicit experimental computer-use flag. Host input still requires desktop-tool consent plus a visible, task-scoped assist/control grant; there is no unrestricted or silent mouse/keyboard automation.
**Desktop control UX direction:** Tauri v2 (Rust + static web UI) is the native shell for the polished Easy-tier experience: tray icon, always-visible overlay chip, task log, settings, and one-click pause/emergency stop. Easy tier pairs once, shows a connected/observing chip, and exposes Devices / Revoke / Task Log / Settings / Emergency Stop from the tray. Standard tier adds full tray management; Advanced tier remains CLI + daemon + JSON policy (`~/.hermes/desktop-control.json`) for operators. The default policy baseline blocks password managers, credential prompts, banking/payment/crypto surfaces, OS security/admin settings, and private-key/token material until locally overridden.
**Deferred to alpha.8 / alpha.9 / v1.0:**
@@ -40,6 +42,7 @@ Release tags: `desktop-v*` (separate cadence from Android `v*`). Curl-installed
- **Environment-variable passthrough** — security-sensitive; needs per-var prompt UX + threat model before shipping.
- **Global hotkey to summon a prompt** — OS-specific helper installers; out of scope for binary-only release.
- **Watch mode** (`hermes-relay daemon --watch`) — needs a DSL and clear safety bounds; own feature branch.
- **Native assist/control grant modal hardening** — the tray-managed daemon now has a local grant bridge and Grant Requests view. Next pass should polish native modal behavior, notification routing, and multi-client grant ownership.
- **Kitty / iTerm2 inline image protocols for paste feedback** — would show a thumbnail of the attached image directly in the terminal after `/paste` instead of a plain text line. Most terminals don't support them; the slash-command feedback line works anywhere. Revisit if users request it.
**Earlier alpha.2–alpha.5 workstreams (now in-flight / done — see DEVLOG 2026-04-23 entries for specifics):**
+49
View File
@@ -6,6 +6,55 @@ For shipped work, see `DEVLOG.md`. For architectural decisions, see `docs/decisi
---
## Hands-free agentic voice backlog
Goal: make Hermes usable for hands-free work without leaving the operator blind
to tool state, safety prompts, or the current task.
- **Waveform output-start sync** — current input waveform timing feels good, but
the agent-output waveform can unfold and begin movement before audible speech
starts. Split "preparing audio" from "speaking audio" in the visual layer, or
gate the unfolded Speaking waveform on the first real playback frame/audio
amplitude. Processing can stay as the folded circular spinner until output is
actually audible.
- **Voice command layer** — reserve local commands that bypass normal agent
routing: "pause", "resume", "stop talking", "cancel", "repeat that", "open
overlay", "return to Hermes", and "new chat". These should work while the
agent is thinking, speaking, or using tools.
- **Spoken tool progress** — when Hermes uses tools, voice mode should speak
short status updates such as "I'm checking the relay logs" or "I found an
error" without waiting for final assistant text. Long tool calls should emit
periodic, low-noise progress updates.
- **Realtime tool timeline parity** — the voice overlay should render the same
live thinking blocks, streaming assistant text, and tool call progress as the
normal chat surface without requiring exit/reload.
- **Hands-free confirmation flow** — risky actions need first-class spoken and
visual confirmation: "yes", "no", "cancel", "confirm", plus a visible and
audible countdown for destructive actions.
- **Voice session memory/status** — add a compact "where are we?" summary for
the current voice task: active objective, last tool result, pending next step,
and whether the agent is waiting on the user.
- **Mode presets** — add presets such as Hands-free, Low latency, Careful tool
mode, and Quiet/visual-only. Hands-free should favor Continuous listening,
spoken tool progress, confirmations, and overlay availability.
- **Barge-in hardening** — keep barge-in experimental until echo/self-recording
is solved. The target path is proper AEC, playback-ducking, and a rule that
output audio can never become a user turn.
- **Audio quality guardrails** — normalize output volume across realtime and
fallback TTS providers, keep pronunciation hints/profile voice tuning, and
measure provider-specific delay, chunk gaps, and tail clipping.
- **Voice engine selector** — implemented as an opt-in experimental Realtime
Agent engine in `docs/plans/2026-05-19-realtime-hermes-voice-agent.md`.
Follow-up work is provider-native turn-taking, richer confirmation handling,
and quality/latency evaluation before promotion beyond Experimental.
- **Realtime-native Hermes bridge prototype** — first relay-brokered slice
implemented in `docs/plans/2026-05-19-realtime-hermes-voice-agent.md`.
Remaining work: let OpenAI/xAI realtime sessions own more of the live speech
turn while still proxying every tool, confirmation, memory, and Android bridge
action through Hermes/relay safety.
---
## Research / open questions
### Proper Hermes plugin / skill / tool distribution
+14 -13
View File
@@ -1,16 +1,17 @@
v0.6.1 - Voice auth, route recovery, and media bridge
v0.7.0 - Profiles, voice settings, and realtime voice polish
Voice and pairing
* Voice mode can now use your saved Hermes API key directly.
* Relay sessions recover better after server restarts and updates.
* LAN/Tailscale pairing and route fallback are more reliable.
Profiles
* Profile switching now keeps profile-specific chat/session state separate.
* Default/Victor display is normalized so the selected agent name stays visible.
* Profile API URL handling is safer for remote Hermes profile servers.
Android bridge
* Added share-sheet support for files, screenshots, relay media, and attachments.
* Added MMS compose handoff with attachments.
* SMS now reports clearer sent, blocked, timeout, and failed states.
* Return-to-Hermes and bridge route docs now match the actual relay API.
Voice
* Voice settings can show and save provider/model/voice choices per profile.
* Voice output uses balanced coalescing for smoother replies.
* Waveform and playback state better follow real audio output.
* Continuous mode no longer starts a session just because auto mode is enabled.
* Barge-in is now clearly marked experimental with known self-capture caveats.
Update note
* Same-flavor installs should keep pairing. Re-pair only after uninstalling data,
switching app flavor, revoking the device, or clearing the server session store.
Relay
* Added relay-owned profile voice config, provider options, streaming voice output,
and realtime voice playground endpoints for OpenAI/xAI/ElevenLabs testing.
@@ -8,6 +8,7 @@ import android.media.MediaRecorder
import android.media.audiofx.AcousticEchoCanceler
import android.media.audiofx.NoiseSuppressor
import android.util.Log
import kotlinx.coroutines.CancellationException
import kotlinx.coroutines.CoroutineDispatcher
import kotlinx.coroutines.CoroutineScope
import kotlinx.coroutines.Dispatchers
@@ -175,11 +176,26 @@ class BargeInListener internal constructor(
_aecAttached.value = false
readerJob = scope.launch(readerDispatcher) {
try {
audioSource.start()
try {
audioSource.start()
} catch (t: CancellationException) {
throw t
} catch (t: Throwable) {
Log.w(TAG, "AudioFrameSource.start failed: ${t.message}")
return@launch
}
Log.i(TAG, "Barge-in AudioRecord reader started")
maybeAttachEffects()
while (isActive) {
val read = audioSource.read(frameBuffer, VadEngine.FRAME_SIZE_SAMPLES)
val read = try {
audioSource.read(frameBuffer, VadEngine.FRAME_SIZE_SAMPLES)
} catch (t: CancellationException) {
throw t
} catch (t: Throwable) {
Log.w(TAG, "AudioFrameSource.read failed; stopping reader: ${t.message}")
break
}
if (read <= 0) {
// Negative values are AudioRecord error codes; 0 means
// no data yet. Either way, yield briefly and retry
@@ -196,7 +212,16 @@ class BargeInListener internal constructor(
continue
}
val result = vadEngine.analyze(frameBuffer)
if (!isActive) break
val result = try {
vadEngine.analyze(frameBuffer)
} catch (t: CancellationException) {
throw t
} catch (t: Throwable) {
Log.w(TAG, "VadEngine.analyze failed; stopping reader: ${t.message}")
break
}
if (result.probability > 0f) {
_maybeSpeech.tryEmit(Unit)
}
@@ -228,9 +253,14 @@ class BargeInListener internal constructor(
* actual release happens in the reader coroutine's `finally` block, which
* is typically a single frame later.
*/
fun stop() {
readerJob?.cancel()
fun stop(): Job? {
val job = readerJob
if (job?.isActive == true) {
Log.i(TAG, "Stopping barge-in AudioRecord reader")
}
job?.cancel()
readerJob = null
return job
}
private suspend fun maybeAttachEffects() {
@@ -253,6 +283,7 @@ class BargeInListener internal constructor(
created.enabled = true
aec = created
_aecAttached.value = true
Log.i(TAG, "AcousticEchoCanceler attached to session=$sessionId")
} else {
Log.i(TAG, "AcousticEchoCanceler.create returned null; continuing without")
}
@@ -0,0 +1,153 @@
package com.hermesandroid.relay.audio
import android.media.AudioAttributes
import android.media.AudioFormat
import android.media.AudioManager
import android.media.AudioTrack
import android.os.Build
import android.util.Log
import kotlinx.coroutines.flow.MutableStateFlow
import kotlinx.coroutines.flow.StateFlow
import kotlinx.coroutines.flow.asStateFlow
import kotlin.math.max
import kotlin.math.sqrt
/**
* Small streaming PCM sink for the realtime voice dev testbench.
*
* The relay sends mono 16-bit little-endian PCM chunks over the websocket. This
* writes them directly to an AudioTrack so the Android Studio dev build can
* hear provider output without waiting for an encoded file.
*/
class RealtimePcmPlayer {
private var audioTrack: AudioTrack? = null
private var currentSampleRate: Int = 0
private var currentVolume: Float = 1f
private val _amplitude = MutableStateFlow(0f)
val amplitude: StateFlow<Float> = _amplitude.asStateFlow()
val isActive: Boolean
get() = audioTrack != null
val audioSessionId: Int
get() = audioTrack?.audioSessionId ?: 0
fun write(pcm: ByteArray, sampleRate: Int): Float {
if (pcm.isEmpty()) return 0f
val level = computePcm16LeRms(pcm)
val track = ensureTrack(sampleRate)
try {
val written = track.write(pcm, 0, pcm.size)
if (written > 0) {
_amplitude.value = level
}
} catch (e: Exception) {
Log.w(TAG, "PCM write failed: ${e.message}")
return 0f
}
return level
}
fun stop() {
audioTrack?.let { track ->
Log.i(TAG, "Stopping streaming PCM playback")
try { track.pause() } catch (_: Exception) { }
try { track.flush() } catch (_: Exception) { }
try { track.release() } catch (_: Exception) { }
}
audioTrack = null
currentSampleRate = 0
_amplitude.value = 0f
}
fun setVolume(volume: Float) {
val clamped = volume.coerceIn(0f, 1f)
currentVolume = clamped
try { audioTrack?.setVolume(clamped) } catch (_: Exception) { }
}
fun duck() {
setVolume(0.3f)
}
fun unduck() {
setVolume(1f)
}
private fun ensureTrack(sampleRate: Int): AudioTrack {
val existing = audioTrack
if (existing != null && currentSampleRate == sampleRate) {
return existing
}
stop()
val minBuffer = AudioTrack.getMinBufferSize(
sampleRate,
AudioFormat.CHANNEL_OUT_MONO,
AudioFormat.ENCODING_PCM_16BIT,
).coerceAtLeast(sampleRate / 10 * 2)
val bufferSize = max(minBuffer, sampleRate / 2 * 2)
val format = AudioFormat.Builder()
.setEncoding(AudioFormat.ENCODING_PCM_16BIT)
.setSampleRate(sampleRate)
.setChannelMask(AudioFormat.CHANNEL_OUT_MONO)
.build()
val track = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) {
AudioTrack.Builder()
.setAudioAttributes(
AudioAttributes.Builder()
.setUsage(AudioAttributes.USAGE_MEDIA)
.setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
.build()
)
.setAudioFormat(format)
.setTransferMode(AudioTrack.MODE_STREAM)
.setBufferSizeInBytes(bufferSize)
.build()
} else {
@Suppress("DEPRECATION")
AudioTrack(
AudioManager.STREAM_MUSIC,
sampleRate,
AudioFormat.CHANNEL_OUT_MONO,
AudioFormat.ENCODING_PCM_16BIT,
bufferSize,
AudioTrack.MODE_STREAM,
)
}
track.play()
try { track.setVolume(currentVolume) } catch (_: Exception) { }
audioTrack = track
currentSampleRate = sampleRate
Log.i(TAG, "Started streaming PCM playback at ${sampleRate}Hz session=${track.audioSessionId}")
return track
}
private fun computePcm16LeRms(pcm: ByteArray): Float {
val usable = pcm.size - (pcm.size % 2)
if (usable <= 0) return 0f
var sumSquares = 0.0
var samples = 0
var index = 0
while (index < usable) {
val low = pcm[index].toInt() and 0xff
val high = pcm[index + 1].toInt()
val sample = ((high shl 8) or low).toShort().toInt()
val normalized = sample / Short.MAX_VALUE.toDouble()
sumSquares += normalized * normalized
samples++
index += 2
}
if (samples == 0) return 0f
val rms = sqrt(sumSquares / samples)
val lifted = sqrt((rms / 0.28).coerceIn(0.0, 1.0))
return if (lifted.isNaN() || lifted.isInfinite()) 0f else lifted.toFloat().coerceIn(0f, 1f)
}
companion object {
private const val TAG = "RealtimePcmPlayer"
}
}
@@ -0,0 +1,63 @@
package com.hermesandroid.relay.audio
import android.annotation.SuppressLint
import android.media.AudioFormat
import android.media.AudioRecord
import android.media.MediaRecorder
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.withContext
import java.io.ByteArrayOutputStream
import kotlin.math.min
/**
* Captures a short mono 16-bit PCM sample for realtime voice test runs.
*/
class RealtimePcmRecorder(
private val sampleRate: Int = 16_000,
) {
@SuppressLint("MissingPermission")
suspend fun capture(durationMs: Long = 800): ByteArray = withContext(Dispatchers.IO) {
val minBuffer = AudioRecord.getMinBufferSize(
sampleRate,
AudioFormat.CHANNEL_IN_MONO,
AudioFormat.ENCODING_PCM_16BIT,
).coerceAtLeast(sampleRate / 10 * 2)
val targetBytes = ((sampleRate * durationMs) / 1000L * 2L)
.toInt()
.coerceAtLeast(minBuffer)
val recorder = AudioRecord.Builder()
.setAudioSource(MediaRecorder.AudioSource.MIC)
.setAudioFormat(
AudioFormat.Builder()
.setEncoding(AudioFormat.ENCODING_PCM_16BIT)
.setSampleRate(sampleRate)
.setChannelMask(AudioFormat.CHANNEL_IN_MONO)
.build()
)
.setBufferSizeInBytes(minBuffer)
.build()
val out = ByteArrayOutputStream(targetBytes)
val buffer = ByteArray(minBuffer)
try {
recorder.startRecording()
while (out.size() < targetBytes) {
val read = recorder.read(
buffer,
0,
min(buffer.size, targetBytes - out.size()),
)
if (read > 0) {
out.write(buffer, 0, read)
} else {
break
}
}
} finally {
try { recorder.stop() } catch (_: Exception) { }
recorder.release()
}
out.toByteArray()
}
}
@@ -2,60 +2,47 @@ package com.hermesandroid.relay.audio
import android.annotation.SuppressLint
import android.content.Context
import android.media.AudioFormat
import android.media.AudioRecord
import android.media.MediaRecorder
import android.os.Build
import android.util.Log
import kotlinx.coroutines.CoroutineScope
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.Job
import kotlinx.coroutines.delay
import kotlinx.coroutines.flow.MutableStateFlow
import kotlinx.coroutines.flow.StateFlow
import kotlinx.coroutines.flow.asStateFlow
import kotlinx.coroutines.isActive
import kotlinx.coroutines.launch
import java.io.ByteArrayOutputStream
import java.io.File
import java.io.IOException
import java.util.concurrent.CountDownLatch
import java.util.concurrent.TimeUnit
import java.util.concurrent.atomic.AtomicBoolean
import kotlin.math.sqrt
/**
* Captures the user's voice into an `.m4a` (AAC-in-MP4) file for V2a voice
* mode. The relay's `/voice/transcribe` endpoint feeds this to whisper-1 via
* OpenAI, which accepts m4a/mp4 natively.
* Captures the user's voice as 16 kHz mono PCM and writes a `.wav` file for
* the relay STT endpoint. The raw PCM is retained for the server-mediated
* `/voice/realtime/{session}` path so the main voice UI can send the same utterance
* through the realtime websocket without opening a second microphone stream.
*
* A live [amplitude] flow is exposed for the UI (MorphingSphere + meter) —
* driven by polling `MediaRecorder.maxAmplitude` every ~16 ms. The polling
* coroutine runs on the caller-supplied [scope] so it dies with the owning
* ViewModel.
*
* One recorder instance owns at most one active recording at a time. Calling
* [startRecording] again while a recording is in flight will stop the
* previous one first. [stopRecording] is safe to call when nothing is
* running (it just returns the last file, or throws if there never was one).
* A live [amplitude] flow is exposed for the UI (MorphingSphere + meter). The
* value is computed from the same PCM frames that are written to disk, which
* keeps legacy STT fallback and realtime voice testing on a single capture
* path.
*/
class VoiceRecorder(
private val context: Context,
private val scope: CoroutineScope,
@Suppress("UNUSED_PARAMETER") private val scope: kotlinx.coroutines.CoroutineScope,
) {
companion object {
private const val TAG = "VoiceRecorder"
private const val SAMPLE_RATE = 16_000
private const val BIT_RATE = 64_000
private const val AMPLITUDE_POLL_MS = 16L
private const val BYTES_PER_SAMPLE = 2
private const val CHANNEL_COUNT = 1
private const val MAX_AMPLITUDE_SHORT = 32_767f
private const val MAX_PCM_BYTES = 25 * 1024 * 1024
// Perceptual amplitude mapping constants. Raw PCM peak values from
// MediaRecorder.maxAmplitude for a phone at arm's length:
// silence / ambient : 100..500 (≤0.015 of max)
// quiet speech : 500..3000 (0.015..0.09)
// normal speech : 3000..8000 (0.09..0.24)
// loud speech : 8000..18000 (0.24..0.55)
// shout / clipping : 18000..32767 (0.55..1.0)
//
// Linear 0..1 puts normal conversation between 0.09 and 0.24 — the
// meter barely moves. Subtract a noise floor, rescale into the
// speech-ceiling window, then apply a sqrt curve so quiet speech
// still registers visually without drowning loud speech at the top.
// Keep the perceptual curve from the previous MediaRecorder-backed
// implementation so the on-screen meter feels the same.
private const val NOISE_FLOOR = 0.01f
private const val SPEECH_CEILING = 0.35f
}
@@ -63,162 +50,231 @@ class VoiceRecorder(
private val _amplitude = MutableStateFlow(0f)
val amplitude: StateFlow<Float> = _amplitude.asStateFlow()
private var mediaRecorder: MediaRecorder? = null
val sampleRate: Int get() = SAMPLE_RATE
private val bufferLock = Any()
private val stopRequested = AtomicBoolean(false)
private var audioRecord: AudioRecord? = null
private var currentOutputFile: File? = null
private var pollJob: Job? = null
private var readThread: Thread? = null
private var readDone: CountDownLatch? = null
private var pcmBuffer = ByteArrayOutputStream(SAMPLE_RATE * BYTES_PER_SAMPLE * 4)
private var lastPcmBytes: ByteArray = ByteArray(0)
/**
* Begin a new recording. Returns the output [File] that will receive the
* audio once [stopRecording] is called. Throws on permission failure or
* encoder init failure — callers should catch and surface to the UI.
* Begin a new recording. Returns the output [File] that will contain WAV
* audio once [stopRecording] is called.
*/
@SuppressLint("MissingPermission")
fun startRecording(): File {
// Defensive: if a recording is somehow still running, tear it down
// before starting a new one. MediaRecorder transitions are strict.
if (mediaRecorder != null) {
Log.w(TAG, "startRecording called while another recording is in flight — stopping it first")
if (audioRecord != null) {
Log.w(TAG, "startRecording called while another recording is in flight; stopping it first")
try {
stopRecording()
} catch (_: Exception) {
// Swallow — we're about to overwrite state anyway.
releaseRecorder()
}
}
val outFile = File(context.cacheDir, "voice_rec_${System.currentTimeMillis()}.m4a")
currentOutputFile = outFile
val minBuffer = AudioRecord.getMinBufferSize(
SAMPLE_RATE,
AudioFormat.CHANNEL_IN_MONO,
AudioFormat.ENCODING_PCM_16BIT,
).coerceAtLeast(SAMPLE_RATE / 10 * BYTES_PER_SAMPLE)
val recorder = buildRecorder()
try {
recorder.setAudioSource(MediaRecorder.AudioSource.MIC)
recorder.setOutputFormat(MediaRecorder.OutputFormat.MPEG_4)
recorder.setAudioEncoder(MediaRecorder.AudioEncoder.AAC)
recorder.setAudioSamplingRate(SAMPLE_RATE)
recorder.setAudioEncodingBitRate(BIT_RATE)
recorder.setAudioChannels(1)
recorder.setOutputFile(outFile.absolutePath)
recorder.prepare()
recorder.start()
} catch (e: IllegalStateException) {
Log.e(TAG, "MediaRecorder failed to start: ${e.message}")
try {
recorder.reset()
} catch (_: Exception) { /* ignore */ }
val outFile = File(context.cacheDir, "voice_rec_${System.currentTimeMillis()}.wav")
currentOutputFile = outFile
synchronized(bufferLock) {
pcmBuffer = ByteArrayOutputStream(SAMPLE_RATE * BYTES_PER_SAMPLE * 4)
lastPcmBytes = ByteArray(0)
}
stopRequested.set(false)
_amplitude.value = 0f
val recorder = AudioRecord.Builder()
.setAudioSource(MediaRecorder.AudioSource.MIC)
.setAudioFormat(
AudioFormat.Builder()
.setEncoding(AudioFormat.ENCODING_PCM_16BIT)
.setSampleRate(SAMPLE_RATE)
.setChannelMask(AudioFormat.CHANNEL_IN_MONO)
.build()
)
.setBufferSizeInBytes(minBuffer * 2)
.build()
if (recorder.state != AudioRecord.STATE_INITIALIZED) {
recorder.release()
mediaRecorder = null
currentOutputFile = null
throw e
throw IllegalStateException("AudioRecord failed to initialize")
}
try {
recorder.startRecording()
} catch (e: Exception) {
Log.e(TAG, "MediaRecorder setup failed: ${e.message}")
try {
recorder.reset()
} catch (_: Exception) { /* ignore */ }
recorder.release()
mediaRecorder = null
currentOutputFile = null
throw e
}
mediaRecorder = recorder
startAmplitudePolling()
audioRecord = recorder
val done = CountDownLatch(1)
readDone = done
readThread = Thread(
{
readPcmLoop(recorder, minBuffer)
done.countDown()
},
"HermesVoiceRecorder",
).also { it.start() }
return outFile
}
/**
* Stop the active recording, flush the encoder, and return the completed
* output [File]. Safe to call when nothing is recording — in that case
* it returns the last file produced, or throws if there never was one.
* Stop the active recording, write the WAV container, and return it.
*/
fun stopRecording(): File {
val file = currentOutputFile
?: throw IllegalStateException("stopRecording called with no active recording")
stopAmplitudePolling()
val recorder = mediaRecorder
if (recorder != null) {
val record = audioRecord
stopRequested.set(true)
if (record != null) {
try {
recorder.stop()
record.stop()
} catch (e: IllegalStateException) {
// MediaRecorder.stop throws if called before any audio was
// captured (sub-300ms recordings). Treat as recoverable —
// the output file may be 0 bytes but the caller can check.
Log.w(TAG, "MediaRecorder.stop threw — recording may be empty: ${e.message}")
} catch (e: RuntimeException) {
Log.w(TAG, "MediaRecorder.stop runtime error: ${e.message}")
} finally {
releaseRecorder()
Log.w(TAG, "AudioRecord.stop threw; recording may be empty: ${e.message}")
}
}
readDone?.await(1, TimeUnit.SECONDS)
releaseRecorder()
val pcm = synchronized(bufferLock) {
pcmBuffer.toByteArray().also { lastPcmBytes = it }
}
writeWav(file, pcm)
_amplitude.value = 0f
return file
}
/**
* True if a recording is currently active. Cheap — just checks whether
* we have a live [MediaRecorder] reference.
*/
fun isRecording(): Boolean = mediaRecorder != null
fun isRecording(): Boolean = audioRecord != null && !stopRequested.get()
fun lastPcmBytes(): ByteArray = synchronized(bufferLock) {
lastPcmBytes.copyOf()
}
/**
* Release any recorder resources without returning a file. Safe fallback
* for error paths where the output file is known-invalid.
* Release any recorder resources without returning a file.
*/
fun cancel() {
stopAmplitudePolling()
mediaRecorder?.let { r ->
try {
r.stop()
} catch (_: Exception) { /* ignore */ }
stopRequested.set(true)
audioRecord?.let { record ->
try { record.stop() } catch (_: Exception) { }
}
readDone?.await(500, TimeUnit.MILLISECONDS)
releaseRecorder()
currentOutputFile?.let { f ->
try { f.delete() } catch (_: Exception) { /* ignore */ }
currentOutputFile?.let { file ->
try { file.delete() } catch (_: Exception) { }
}
currentOutputFile = null
synchronized(bufferLock) {
pcmBuffer.reset()
lastPcmBytes = ByteArray(0)
}
_amplitude.value = 0f
}
private fun releaseRecorder() {
mediaRecorder?.let { r ->
try { r.reset() } catch (_: Exception) { /* ignore */ }
try { r.release() } catch (_: Exception) { /* ignore */ }
}
mediaRecorder = null
}
@Suppress("DEPRECATION")
private fun buildRecorder(): MediaRecorder =
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) {
MediaRecorder(context)
} else {
MediaRecorder()
}
private fun startAmplitudePolling() {
pollJob?.cancel()
pollJob = scope.launch(Dispatchers.Default) {
while (isActive) {
val recorder = mediaRecorder ?: break
val raw = try {
recorder.maxAmplitude
} catch (e: IllegalStateException) {
// Recorder torn down under us — exit quietly.
break
private fun readPcmLoop(record: AudioRecord, minBuffer: Int) {
val buffer = ByteArray(minBuffer)
while (!stopRequested.get()) {
val read = try {
record.read(buffer, 0, buffer.size)
} catch (e: Exception) {
Log.w(TAG, "AudioRecord.read failed: ${e.message}")
break
}
if (read > 0) {
synchronized(bufferLock) {
if (pcmBuffer.size() + read <= MAX_PCM_BYTES) {
pcmBuffer.write(buffer, 0, read)
} else {
stopRequested.set(true)
}
}
val raw01 = (raw.toFloat() / MAX_AMPLITUDE_SHORT).coerceIn(0f, 1f)
val floored = ((raw01 - NOISE_FLOOR) / (SPEECH_CEILING - NOISE_FLOOR))
.coerceIn(0f, 1f)
_amplitude.value = sqrt(floored)
delay(AMPLITUDE_POLL_MS)
updateAmplitude(buffer, read)
}
}
}
private fun stopAmplitudePolling() {
pollJob?.cancel()
pollJob = null
private fun updateAmplitude(buffer: ByteArray, read: Int) {
var peak = 0
var index = 0
val usable = read - (read % BYTES_PER_SAMPLE)
while (index < usable) {
val low = buffer[index].toInt() and 0xff
val high = buffer[index + 1].toInt()
val sample = (high shl 8) or low
val abs = kotlin.math.abs(sample.coerceIn(Short.MIN_VALUE.toInt(), Short.MAX_VALUE.toInt()))
if (abs > peak) peak = abs
index += BYTES_PER_SAMPLE
}
val raw01 = (peak.toFloat() / MAX_AMPLITUDE_SHORT).coerceIn(0f, 1f)
val floored = ((raw01 - NOISE_FLOOR) / (SPEECH_CEILING - NOISE_FLOOR))
.coerceIn(0f, 1f)
_amplitude.value = sqrt(floored)
}
private fun releaseRecorder() {
audioRecord?.let { record ->
try { record.release() } catch (_: Exception) { }
}
audioRecord = null
readThread = null
readDone = null
}
private fun writeWav(file: File, pcm: ByteArray) {
try {
file.outputStream().use { out ->
out.write(wavHeader(pcm.size))
out.write(pcm)
}
} catch (e: IOException) {
throw IOException("Failed to write WAV recording: ${e.message}", e)
}
}
private fun wavHeader(pcmBytes: Int): ByteArray {
val totalDataLen = pcmBytes + 36
val byteRate = SAMPLE_RATE * CHANNEL_COUNT * BYTES_PER_SAMPLE
return ByteArray(44).also { header ->
fun ascii(offset: Int, value: String) {
value.encodeToByteArray().copyInto(header, offset)
}
fun leInt(offset: Int, value: Int) {
header[offset] = (value and 0xff).toByte()
header[offset + 1] = ((value shr 8) and 0xff).toByte()
header[offset + 2] = ((value shr 16) and 0xff).toByte()
header[offset + 3] = ((value shr 24) and 0xff).toByte()
}
fun leShort(offset: Int, value: Int) {
header[offset] = (value and 0xff).toByte()
header[offset + 1] = ((value shr 8) and 0xff).toByte()
}
ascii(0, "RIFF")
leInt(4, totalDataLen)
ascii(8, "WAVE")
ascii(12, "fmt ")
leInt(16, 16)
leShort(20, 1)
leShort(22, CHANNEL_COUNT)
leInt(24, SAMPLE_RATE)
leInt(28, byteRate)
leShort(32, CHANNEL_COUNT * BYTES_PER_SAMPLE)
leShort(34, 16)
ascii(36, "data")
leInt(40, pcmBytes)
}
}
}
@@ -147,6 +147,9 @@ class AuthManager(
* metadata) are optional on the wire. Missing / malformed values
* fall back to `false` / `false` / `0` so older relays stay
* compatible and bad server data can't crash the pairing handshake.
* - `api_server_*` metadata is optional. When present, it lets the
* client route chat through a profile's isolated Hermes API
* server without exposing that profile server's key.
*/
fun parseAgentProfiles(array: JsonArray): List<Profile> {
return array.mapNotNull { entry ->
@@ -164,6 +167,16 @@ class AuthManager(
?.jsonPrimitive?.booleanOrNull ?: false
val skillCount = obj["skill_count"]
?.jsonPrimitive?.intOrNull ?: 0
val apiServerEnabled = obj["api_server_enabled"]
?.jsonPrimitive?.booleanOrNull ?: false
val apiServerUrl = obj["api_server_url"]
?.jsonPrimitive?.contentOrNull
val apiServerHost = obj["api_server_host"]
?.jsonPrimitive?.contentOrNull
val apiServerPort = obj["api_server_port"]
?.jsonPrimitive?.intOrNull
val apiServerKeyPresent = obj["api_server_key_present"]
?.jsonPrimitive?.booleanOrNull ?: false
Profile(
name = name,
model = model,
@@ -172,6 +185,11 @@ class AuthManager(
gatewayRunning = gatewayRunning,
hasSoul = hasSoul,
skillCount = skillCount,
apiServerEnabled = apiServerEnabled,
apiServerUrl = apiServerUrl,
apiServerHost = apiServerHost,
apiServerPort = apiServerPort,
apiServerKeyPresent = apiServerKeyPresent,
)
}
}
@@ -0,0 +1,83 @@
package com.hermesandroid.relay.data
/**
* Shared profile/personality display and request identity helpers.
*
* A null profile name is the app's explicit "Server default" state. The
* relay also advertises the root Hermes config as a synthetic profile named
* "default"; for request/session identity that row is an alias of server
* default so it does not split chat, voice, or session scope.
*/
object AgentDisplay {
const val SERVER_DEFAULT_PROFILE_KEY: String = "__server_default__"
fun effectiveProfile(
selectedProfile: Profile?,
profiles: List<Profile>,
): Profile? = selectedProfile
?: profiles.firstOrNull { it.name.equals("default", ignoreCase = true) }
fun profileDisplayName(profile: Profile?): String? {
if (profile == null) return null
return when {
profile.description.isNotBlank() -> profile.description.trim()
profile.name.isNotBlank() -> titleCase(profile.name.trim())
else -> null
}
}
fun agentName(
profile: Profile?,
selectedPersonality: String,
defaultPersonality: String,
connectionLabel: String?,
): String {
profileDisplayName(profile)?.let { return it }
val personalityName = if (
selectedPersonality == "default" &&
defaultPersonality.isNotBlank()
) {
defaultPersonality
} else {
selectedPersonality
}
return when {
personalityName.isNotBlank() && personalityName != "default" ->
titleCase(personalityName.trim())
!connectionLabel.isNullOrBlank() -> connectionLabel.trim()
else -> "Hermes"
}
}
fun personalityLabel(
selectedPersonality: String,
defaultPersonality: String,
): String = when {
selectedPersonality != "default" && selectedPersonality.isNotBlank() ->
titleCase(selectedPersonality.trim())
defaultPersonality.isNotBlank() -> titleCase(defaultPersonality.trim())
else -> "Default"
}
fun isServerDefaultAlias(profileName: String?): Boolean =
profileName?.trim()?.equals("default", ignoreCase = true) == true
fun normalizeSelection(profile: Profile?): Profile? =
if (isServerDefaultAlias(profile?.name)) null else profile
fun profileRequestName(profileName: String?): String? =
profileName
?.trim()
?.takeIf { it.isNotEmpty() && !isServerDefaultAlias(it) }
fun profileSessionKey(profileName: String?): String =
profileRequestName(profileName) ?: SERVER_DEFAULT_PROFILE_KEY
fun profileContextKey(connectionId: String?, profileName: String?): String =
"${connectionId.orEmpty()}::${profileSessionKey(profileName)}"
private fun titleCase(value: String): String =
value.replaceFirstChar { it.uppercase() }
}
@@ -7,6 +7,7 @@ import androidx.datastore.preferences.core.booleanPreferencesKey
import androidx.datastore.preferences.core.edit
import androidx.datastore.preferences.core.stringPreferencesKey
import kotlinx.coroutines.flow.Flow
import kotlinx.coroutines.flow.distinctUntilChanged
import kotlinx.coroutines.flow.map
/**
@@ -87,15 +88,17 @@ class BargeInPreferencesRepository(
booleanPreferencesKey("barge_in_resume_after_interruption")
}
val flow: Flow<BargeInPreferences> = dataStore.data.map { prefs ->
BargeInPreferences(
enabled = prefs[KEY_ENABLED] ?: DEFAULT_ENABLED,
sensitivity = prefs[KEY_SENSITIVITY]?.let { decodeSensitivity(it) }
?: DEFAULT_SENSITIVITY,
resumeAfterInterruption = prefs[KEY_RESUME_AFTER_INTERRUPTION]
?: DEFAULT_RESUME_AFTER_INTERRUPTION,
)
}
val flow: Flow<BargeInPreferences> = dataStore.data
.map { prefs ->
BargeInPreferences(
enabled = prefs[KEY_ENABLED] ?: DEFAULT_ENABLED,
sensitivity = prefs[KEY_SENSITIVITY]?.let { decodeSensitivity(it) }
?: DEFAULT_SENSITIVITY,
resumeAfterInterruption = prefs[KEY_RESUME_AFTER_INTERRUPTION]
?: DEFAULT_RESUME_AFTER_INTERRUPTION,
)
}
.distinctUntilChanged()
suspend fun setEnabled(value: Boolean) {
dataStore.edit { it[KEY_ENABLED] = value }
@@ -13,13 +13,16 @@ import kotlinx.serialization.Serializable
* and added [systemMessage], sourced from each profile's `SOUL.md`.
*
* A Profile is a NAMED AGENT CONFIG within a Connection. Switching profile
* changes:
* - which model the phone asks the server to use on the next chat
* request (via [model]);
* - which system message the phone sends for that request (via
* [systemMessage], when non-blank).
* changes the active agent identity for the Android chat surface:
* - which profile API server the phone routes chat/session calls to when
* the relay advertises [apiServerUrl];
* - which profile name the phone sends to the server for new sessions and
* chat turns when isolated routing is not available;
* - which model and system message the phone can send as compatibility
* fallback (via [model] and [systemMessage]);
* - which profile-scoped session id Android resumes for local chat context.
*
* It does not change the server, sessions, or memory.
* It does not mutate the server's configured default profile.
*
* Wire shape uses snake_case (`system_message`), this class uses camelCase
* (`systemMessage`) — translated via [SerialName].
@@ -39,6 +42,12 @@ import kotlinx.serialization.Serializable
* All three default to safe zero-values and are optional on the wire, so
* older relays without the fields deserialize cleanly as
* `gatewayRunning = false, hasSoul = false, skillCount = 0`.
*
* **Hermes profile API metadata.** A relay can advertise an isolated
* profile API server without exposing its secret. When [apiServerUrl] is
* present, Android routes chat/session traffic to that URL and reuses the
* active connection's stored API key. Operators that use distinct API keys
* per profile should pair those profile API servers as separate connections.
*/
@Serializable
data class Profile(
@@ -53,4 +62,17 @@ data class Profile(
val hasSoul: Boolean = false,
@SerialName("skill_count")
val skillCount: Int = 0,
)
@SerialName("api_server_enabled")
val apiServerEnabled: Boolean = false,
@SerialName("api_server_url")
val apiServerUrl: String? = null,
@SerialName("api_server_host")
val apiServerHost: String? = null,
@SerialName("api_server_port")
val apiServerPort: Int? = null,
@SerialName("api_server_key_present")
val apiServerKeyPresent: Boolean = false,
) {
val hasIsolatedApi: Boolean
get() = !apiServerUrl.isNullOrBlank()
}
@@ -0,0 +1,69 @@
package com.hermesandroid.relay.data
import android.content.Context
import androidx.datastore.core.DataStore
import androidx.datastore.preferences.core.Preferences
import androidx.datastore.preferences.core.edit
import androidx.datastore.preferences.core.stringPreferencesKey
import androidx.datastore.preferences.preferencesDataStore
import kotlinx.coroutines.flow.Flow
import kotlinx.coroutines.flow.map
/**
* Per-connection, per-Hermes-profile last active chat session.
*
* This is intentionally separate from [ProfileSelectionStore]. Selection says
* which agent is active; this store says which chat session belongs to that
* agent on that connection. Null profile name is the explicit Server default
* context.
*/
class ProfileSessionStore(
private val dataStore: DataStore<Preferences>,
) {
constructor(context: Context) : this(context.profileSessionsDataStore)
companion object {
private const val PREFIX = "profile_session__"
private fun keyName(connectionId: String, profileName: String?): String =
"$PREFIX${connectionId}__${AgentDisplay.profileSessionKey(profileName)}"
private fun keyFor(connectionId: String, profileName: String?) =
stringPreferencesKey(keyName(connectionId, profileName))
private fun connectionPrefix(connectionId: String): String =
"$PREFIX${connectionId}__"
}
suspend fun setSessionId(
connectionId: String,
profileName: String?,
sessionId: String?,
) {
dataStore.edit { prefs ->
val key = keyFor(connectionId, profileName)
if (sessionId.isNullOrBlank()) {
prefs.remove(key)
} else {
prefs[key] = sessionId
}
}
}
fun sessionIdFlow(connectionId: String, profileName: String?): Flow<String?> {
val key = keyFor(connectionId, profileName)
return dataStore.data.map { prefs -> prefs[key] }
}
suspend fun clearConnection(connectionId: String) {
val prefix = connectionPrefix(connectionId)
dataStore.edit { prefs ->
prefs.asMap().keys
.filter { it.name.startsWith(prefix) }
.forEach { prefs.remove(it) }
}
}
}
internal val Context.profileSessionsDataStore: DataStore<Preferences>
by preferencesDataStore(name = "profile_sessions")
@@ -1,11 +1,14 @@
package com.hermesandroid.relay.data
import android.content.Context
import androidx.datastore.core.DataStore
import androidx.datastore.preferences.core.booleanPreferencesKey
import androidx.datastore.preferences.core.edit
import androidx.datastore.preferences.core.longPreferencesKey
import androidx.datastore.preferences.core.Preferences
import androidx.datastore.preferences.core.stringPreferencesKey
import kotlinx.coroutines.flow.Flow
import kotlinx.coroutines.flow.distinctUntilChanged
import kotlinx.coroutines.flow.map
/**
@@ -20,48 +23,72 @@ import kotlinx.coroutines.flow.map
* (V1 doesn't accept a language param).
*/
data class VoiceSettings(
val engineMode: String = VoiceEngineMode.HermesVoiceOutput.storageValue,
val interactionMode: String = "tap",
val silenceThresholdMs: Long = 3000L,
val autoTts: Boolean = false,
val language: String = "",
)
class VoicePreferencesRepository(private val context: Context) {
enum class VoiceEngineMode(val storageValue: String) {
HermesVoiceOutput("hermes_voice_output"),
RealtimeAgent("realtime_agent");
companion object {
fun fromStorage(value: String?): VoiceEngineMode =
values().firstOrNull { it.storageValue == value } ?: HermesVoiceOutput
}
}
class VoicePreferencesRepository(private val dataStore: DataStore<Preferences>) {
constructor(context: Context) : this(context.relayDataStore)
companion object {
private val KEY_ENGINE_MODE = stringPreferencesKey("voice_engine_mode")
private val KEY_INTERACTION_MODE = stringPreferencesKey("voice_interaction_mode")
private val KEY_SILENCE_THRESHOLD_MS = longPreferencesKey("voice_silence_threshold_ms")
private val KEY_AUTO_TTS = booleanPreferencesKey("voice_auto_tts")
private val KEY_LANGUAGE = stringPreferencesKey("voice_language")
const val DEFAULT_ENGINE_MODE = "hermes_voice_output"
const val DEFAULT_INTERACTION_MODE = "tap"
const val DEFAULT_SILENCE_THRESHOLD_MS = 3000L
const val DEFAULT_AUTO_TTS = false
const val DEFAULT_LANGUAGE = ""
}
val settings: Flow<VoiceSettings> = context.relayDataStore.data.map { prefs ->
VoiceSettings(
interactionMode = prefs[KEY_INTERACTION_MODE] ?: DEFAULT_INTERACTION_MODE,
silenceThresholdMs = prefs[KEY_SILENCE_THRESHOLD_MS] ?: DEFAULT_SILENCE_THRESHOLD_MS,
autoTts = prefs[KEY_AUTO_TTS] ?: DEFAULT_AUTO_TTS,
language = prefs[KEY_LANGUAGE] ?: DEFAULT_LANGUAGE,
)
val settings: Flow<VoiceSettings> = dataStore.data
.map { prefs ->
VoiceSettings(
engineMode = VoiceEngineMode.fromStorage(
prefs[KEY_ENGINE_MODE] ?: DEFAULT_ENGINE_MODE,
).storageValue,
interactionMode = prefs[KEY_INTERACTION_MODE] ?: DEFAULT_INTERACTION_MODE,
silenceThresholdMs = prefs[KEY_SILENCE_THRESHOLD_MS] ?: DEFAULT_SILENCE_THRESHOLD_MS,
autoTts = prefs[KEY_AUTO_TTS] ?: DEFAULT_AUTO_TTS,
language = prefs[KEY_LANGUAGE] ?: DEFAULT_LANGUAGE,
)
}
.distinctUntilChanged()
suspend fun setEngineMode(mode: VoiceEngineMode) {
dataStore.edit { it[KEY_ENGINE_MODE] = mode.storageValue }
}
suspend fun setInteractionMode(mode: String) {
context.relayDataStore.edit { it[KEY_INTERACTION_MODE] = mode }
dataStore.edit { it[KEY_INTERACTION_MODE] = mode }
}
suspend fun setSilenceThresholdMs(ms: Long) {
context.relayDataStore.edit { it[KEY_SILENCE_THRESHOLD_MS] = ms.coerceAtLeast(500L) }
dataStore.edit { it[KEY_SILENCE_THRESHOLD_MS] = ms.coerceAtLeast(500L) }
}
suspend fun setAutoTts(enabled: Boolean) {
context.relayDataStore.edit { it[KEY_AUTO_TTS] = enabled }
dataStore.edit { it[KEY_AUTO_TTS] = enabled }
}
suspend fun setLanguage(language: String) {
context.relayDataStore.edit { it[KEY_LANGUAGE] = language }
dataStore.edit { it[KEY_LANGUAGE] = language }
}
}
@@ -179,6 +179,9 @@ class ConnectionManager(
}
fun setInsecureMode(enabled: Boolean) {
if (_insecureMode.value == enabled) {
return
}
_insecureMode.value = enabled
if (enabled) {
Log.w(TAG, "⚠ INSECURE MODE ENABLED — ws:// connections allowed. Do NOT use in production.")
@@ -3,6 +3,7 @@ package com.hermesandroid.relay.network
import android.os.Handler
import android.os.Looper
import android.util.Log
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.AppAnalytics
import com.hermesandroid.relay.network.models.CreateSessionRequest
import com.hermesandroid.relay.network.models.HermesSseEvent
@@ -21,10 +22,9 @@ import kotlinx.serialization.encodeToString
import kotlinx.serialization.json.Json
import kotlinx.serialization.json.JsonArray
import kotlinx.serialization.json.JsonObject
import kotlinx.serialization.json.addJsonObject
import kotlinx.serialization.json.buildJsonObject
import kotlinx.serialization.json.put
import kotlinx.serialization.json.putJsonArray
import kotlinx.serialization.json.JsonPrimitive
import kotlinx.serialization.json.contentOrNull
import kotlinx.serialization.json.decodeFromJsonElement
import okhttp3.MediaType.Companion.toMediaType
import okhttp3.OkHttpClient
import okhttp3.Request
@@ -57,16 +57,16 @@ enum class ChatMode {
* The Android client uses this to pick the best chat path automatically when
* `streamingEndpoint = "auto"`. The bootstrap-injected vanilla-upstream case
* is the interesting one: `sessionsApi=true` (we injected it) but
* `sessionsChatStream=false` (we deliberately didn't inject the chat
* handler — runs is better). The auto-resolver picks `runs` for chat in that
* case while still using sessions endpoints for browse/rename/delete.
* `sessionsChatStream=false` (the chat handler is absent). The auto-resolver
* now picks OpenAI-compatible chat completions for that case because the route
* returns an SSE stream, while `/v1/runs` may be an async JSON run-start API.
*/
data class ServerCapabilities(
/** `/api/sessions` (CRUD) — true on fork, upstream-merged, OR bootstrap-injected. */
val sessionsApi: Boolean,
/** `/api/sessions/{id}/chat/stream` (SSE) — true ONLY on fork or upstream-merged. */
val sessionsChatStream: Boolean,
/** `/v1/runs` (structured-event SSE) — standard upstream chat path. */
/** `/v1/runs` (structured-event SSE) — true only when explicitly advertised as SSE-compatible. */
val runs: Boolean,
/** `/v1/chat/completions` — OpenAI-compatible fallback. */
val portable: Boolean,
@@ -76,6 +76,7 @@ data class ServerCapabilities(
/** Resolve `streamingEndpoint = "auto"` to the best concrete choice. */
fun preferredChatEndpoint(): String = when {
sessionsChatStream -> "sessions"
portable -> "completions"
runs -> "runs"
else -> "sessions" // last-resort: try sessions, will surface a clear error
}
@@ -252,9 +253,19 @@ class HermesApiClient(
suspend fun listSessions(limit: Int = 50): List<SessionItem> =
listSessionsResult(limit).getOrElse { emptyList() }
suspend fun createSessionResult(title: String? = null): Result<SessionItem> = withContext(Dispatchers.IO) {
suspend fun createSessionResult(
title: String? = null,
profileName: String? = null,
model: String? = null,
): Result<SessionItem> = withContext(Dispatchers.IO) {
try {
val reqBody = json.encodeToString(CreateSessionRequest(title = title))
val reqBody = json.encodeToString(
CreateSessionRequest(
title = title,
model = model,
profile = AgentDisplay.profileRequestName(profileName),
),
)
val request = authRequest("$baseUrl/api/sessions")
.post(reqBody.toRequestBody(JSON_MEDIA))
.build()
@@ -279,8 +290,12 @@ class HermesApiClient(
}
}
suspend fun createSession(title: String? = null): SessionItem? =
createSessionResult(title).getOrNull()
suspend fun createSession(
title: String? = null,
profileName: String? = null,
model: String? = null,
): SessionItem? =
createSessionResult(title, profileName, model).getOrNull()
suspend fun deleteSession(sessionId: String): Boolean = withContext(Dispatchers.IO) {
try {
@@ -387,10 +402,22 @@ class HermesApiClient(
key to ((value as? kotlinx.serialization.json.JsonPrimitive)?.content ?: "")
} ?: emptyMap()
// Default personality: config.display.personality
// Default display identity. Upstream Hermes currently uses
// config.display.personality for the active persona and often
// mirrors the same identity through skin. Accept name-style
// aliases too so older or profile-specific configs don't make
// Android fall back to the literal "Hermes" label.
val display = config["display"] as? JsonObject
val defaultName = (display?.get("personality") as? kotlinx.serialization.json.JsonPrimitive)
?.content ?: ""
val defaultPersonality = display.stringField("personality")
val defaultName = firstNonBlank(
defaultPersonality.takeUnless { it.equals("default", ignoreCase = true) },
display.stringField("agent_name"),
display.stringField("assistant_name"),
display.stringField("display_name"),
display.stringField("name"),
display.stringField("skin"),
defaultPersonality,
)
PersonalityConfig(
names = prompts.keys.toList(),
@@ -453,39 +480,23 @@ class HermesApiClient(
onComplete: () -> Unit,
onUsage: (UsageInfo?) -> Unit,
onError: (String) -> Unit,
modelOverride: String? = null
modelOverride: String? = null,
profileName: String? = null,
): EventSource {
val requestPayload = buildJsonObject {
put("message", message)
if (!systemMessage.isNullOrBlank()) {
put("system_message", systemMessage)
}
if (!modelOverride.isNullOrBlank()) {
// Agent-profile model override — top-level `model` mirrors
// the OpenAI Chat Completions shape and is what the
// upstream sessions handler looks at when deciding which
// model to route the turn through.
put("model", modelOverride)
Log.d(TAG, "sendChatStream: modelOverride=$modelOverride (profile pick)")
}
if (!attachments.isNullOrEmpty()) {
putJsonArray("attachments") {
attachments.forEach { att ->
addJsonObject {
put("contentType", att.contentType)
put("content", att.content)
}
}
}
}
if (voiceIntentMessages != null && voiceIntentMessages.isNotEmpty()) {
// Nest the synthesized OpenAI-format pairs under `messages`.
// Stays additive — the upstream sessions handler reads
// `message` for the live turn and treats `messages` as
// history context to seed the LLM with.
put("messages", voiceIntentMessages)
}
if (!modelOverride.isNullOrBlank()) {
Log.d(TAG, "sendChatStream: modelOverride=$modelOverride (profile pick)")
}
AgentDisplay.profileRequestName(profileName)?.let {
Log.d(TAG, "sendChatStream: profile=$it")
}
val requestPayload = buildSessionChatStreamPayload(
message = message,
systemMessage = systemMessage,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
modelOverride = modelOverride,
profileName = profileName,
)
val requestBody = json.encodeToString(JsonObject.serializer(), requestPayload)
val request = authRequest("$baseUrl/api/sessions/$sessionId/chat/stream")
@@ -681,6 +692,180 @@ class HermesApiClient(
return sseFactory.newEventSource(request, listener)
}
// --- OpenAI-compatible chat streaming via /v1/chat/completions ---
/**
* Stream chat through the OpenAI-compatible chat completions endpoint.
*
* This is the portable SSE fallback for servers that expose
* `/v1/chat/completions` but where `/v1/runs` is an async JSON run-start
* API rather than an EventSource-compatible stream.
*/
fun sendChatCompletionsStream(
message: String,
model: String? = null,
systemMessage: String? = null,
attachments: List<com.hermesandroid.relay.data.Attachment>? = null,
voiceIntentMessages: JsonArray? = null,
onSessionId: (String) -> Unit,
onMessageStarted: (String) -> Unit,
onTextDelta: (String) -> Unit,
onThinkingDelta: (String) -> Unit,
onToolCallStart: (String, String) -> Unit,
onToolCallDone: (String, String?) -> Unit,
onToolCallFailed: (String, String?) -> Unit,
onTurnComplete: () -> Unit,
onComplete: () -> Unit,
onUsage: (UsageInfo?) -> Unit,
onError: (String) -> Unit,
modelOverride: String? = null,
profileName: String? = null,
): EventSource {
if (!modelOverride.isNullOrBlank()) {
Log.d(TAG, "sendChatCompletionsStream: modelOverride=$modelOverride (profile pick, was model=$model)")
}
AgentDisplay.profileRequestName(profileName)?.let {
Log.d(TAG, "sendChatCompletionsStream: profile=$it")
}
val requestPayload = buildChatCompletionsStreamPayload(
message = message,
model = model,
systemMessage = systemMessage,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
modelOverride = modelOverride,
profileName = profileName,
)
val requestBody = json.encodeToString(JsonObject.serializer(), requestPayload)
val request = authRequest("$baseUrl/v1/chat/completions")
.header("Accept", "text/event-stream")
.post(requestBody.toRequestBody(JSON_MEDIA))
.build()
val completeCalled = AtomicBoolean(false)
val messageStarted = AtomicBoolean(false)
val listener = object : EventSourceListener() {
override fun onEvent(
eventSource: EventSource,
id: String?,
type: String?,
data: String
) {
if (data == "[DONE]") {
if (completeCalled.compareAndSet(false, true)) {
mainHandler.post { onComplete() }
}
return
}
try {
val event = json.decodeFromString<JsonObject>(data)
openAiErrorMessage(event)?.let { msg ->
if (completeCalled.compareAndSet(false, true)) {
mainHandler.post { onError(msg) }
}
return
}
openAiUsage(event)?.let { usage ->
mainHandler.post { onUsage(usage) }
}
if (messageStarted.compareAndSet(false, true)) {
openAiMessageId(event)?.let { messageId ->
mainHandler.post { onMessageStarted(messageId) }
}
}
openAiReasoningDelta(event)?.let { reasoning ->
if (reasoning.isNotEmpty()) {
mainHandler.post { onThinkingDelta(reasoning) }
}
}
openAiTextDelta(event)?.let { delta ->
if (delta.isNotEmpty()) {
mainHandler.post { onTextDelta(delta) }
}
}
val finishReason = openAiFinishReason(event)
if (!finishReason.isNullOrBlank() && completeCalled.compareAndSet(false, true)) {
mainHandler.post { onComplete() }
}
} catch (e: Exception) {
Log.w(TAG, "Unparseable chat completion SSE event ($type): ${e.message}\nRaw: $data")
}
}
override fun onFailure(
eventSource: EventSource,
t: Throwable?,
response: Response?
) {
if (completeCalled.compareAndSet(false, true)) {
val msg = when {
response != null && !response.isSuccessful ->
"API error ${response.code}: ${response.message}"
t is IOException -> "Connection failed: ${t.message}"
t != null -> "Stream error: ${t.message}"
else -> "Unknown stream error"
}
mainHandler.post { onError(msg) }
}
}
override fun onClosed(eventSource: EventSource) {
if (completeCalled.compareAndSet(false, true)) {
mainHandler.post { onComplete() }
}
}
}
return sseFactory.newEventSource(request, listener)
}
private fun openAiChoice(event: JsonObject): JsonObject? =
(event["choices"] as? JsonArray)
?.firstOrNull()
?.let { it as? JsonObject }
private fun openAiDelta(event: JsonObject): JsonObject? =
openAiChoice(event)?.get("delta") as? JsonObject
private fun openAiTextDelta(event: JsonObject): String? =
(openAiDelta(event)?.get("content") as? JsonPrimitive)?.contentOrNull
private fun openAiReasoningDelta(event: JsonObject): String? {
val delta = openAiDelta(event) ?: return null
return (delta["reasoning_content"] as? JsonPrimitive)?.contentOrNull
?: (delta["reasoning"] as? JsonPrimitive)?.contentOrNull
?: (delta["thinking"] as? JsonPrimitive)?.contentOrNull
}
private fun openAiFinishReason(event: JsonObject): String? =
(openAiChoice(event)?.get("finish_reason") as? JsonPrimitive)?.contentOrNull
private fun openAiMessageId(event: JsonObject): String? =
(event["id"] as? JsonPrimitive)?.contentOrNull
private fun openAiErrorMessage(event: JsonObject): String? {
val error = event["error"] ?: return null
return when (error) {
is JsonPrimitive -> error.contentOrNull
is JsonObject -> (error["message"] as? JsonPrimitive)?.contentOrNull
?: (error["error"] as? JsonPrimitive)?.contentOrNull
else -> null
}
}
private fun openAiUsage(event: JsonObject): UsageInfo? =
(event["usage"] as? JsonObject)?.let { usage ->
runCatching { json.decodeFromJsonElement<UsageInfo>(usage) }.getOrNull()
}
// --- Run streaming via /v1/runs ---
/**
@@ -715,44 +900,24 @@ class HermesApiClient(
onComplete: () -> Unit,
onUsage: (UsageInfo?) -> Unit,
onError: (String) -> Unit,
modelOverride: String? = null
modelOverride: String? = null,
profileName: String? = null,
): EventSource {
// Resolution: modelOverride (profile pick) > model (caller default) > "default".
// Blank strings are treated as null so callers can't accidentally
// ship `"model": ""` over the wire.
val resolvedModel = when {
!modelOverride.isNullOrBlank() -> modelOverride
!model.isNullOrBlank() -> model
else -> "default"
}
if (!modelOverride.isNullOrBlank()) {
Log.d(TAG, "sendRunStream: modelOverride=$modelOverride (profile pick, was model=$model)")
}
val requestPayload = buildJsonObject {
put("model", resolvedModel)
put("input", message)
put("stream", true)
if (!systemMessage.isNullOrBlank()) {
put("system_message", systemMessage)
}
if (!attachments.isNullOrEmpty()) {
putJsonArray("attachments") {
attachments.forEach { att ->
addJsonObject {
put("contentType", att.contentType)
put("content", att.content)
}
}
}
}
if (voiceIntentMessages != null && voiceIntentMessages.isNotEmpty()) {
// /v1/runs is OpenAI Responses-shaped — accepts an
// additional `messages` field for context-priming the
// run. Mirror the chat-stream branch so both endpoints
// ingest the synthetic voice-intent history identically.
put("messages", voiceIntentMessages)
}
AgentDisplay.profileRequestName(profileName)?.let {
Log.d(TAG, "sendRunStream: profile=$it")
}
val requestPayload = buildRunStreamPayload(
message = message,
model = model,
systemMessage = systemMessage,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
modelOverride = modelOverride,
profileName = profileName,
)
val requestBody = json.encodeToString(JsonObject.serializer(), requestPayload)
val request = authRequest("$baseUrl/v1/runs")
@@ -970,8 +1135,9 @@ class HermesApiClient(
* presence. The handler only accepts POST, so HEAD returns 405
* (Method Not Allowed) when the route is registered. 404 means
* the route doesn't exist at all.
* 4. `HEAD /v1/runs` — runs endpoint presence (same 405-vs-404 logic).
* 5. `HEAD /v1/models` — OpenAI-compat reachability.
* 4. `HEAD /v1/chat/completions` — OpenAI-compatible SSE fallback.
* 5. `HEAD /v1/runs` with `Accept: text/event-stream` — accepted only
* when the response explicitly advertises event-stream compatibility.
*
* **Why HEAD instead of OPTIONS:** The hermes-agent gateway runs CORS
* middleware (`security_headers_middleware`) that intercepts OPTIONS
@@ -981,10 +1147,13 @@ class HermesApiClient(
* for present, 404 for missing). Verified empirically against the
* production hermes-agent gateway on 2026-04-12.
*
* **Success criterion:** any HTTP response code that isn't 404 means
* the route is registered. We accept 200, 204, 401, 403, 405, 415,
* etc. as positive — even quirky middleware responses count, because
* the alternative (404) is the only signal that means "no such path."
* **Route presence criterion:** for sessions and completions, any HTTP
* response code that isn't 404 means the route is registered. We accept
* 200, 204, 401, 403, 405, 415, etc. as positive because the alternative
* (404) is the only signal that means "no such path." `/v1/runs` is
* stricter: route presence alone is not enough because async runs can
* return `202 application/json`; auto only uses it if event-stream support
* is explicitly advertised.
*
* Network errors (connection refused, DNS failure, etc.) count as
* "missing" since we can't differentiate from a server-down case.
@@ -1010,10 +1179,31 @@ class HermesApiClient(
false
}
fun Response.advertisesEventStream(): Boolean {
val contentType = header("Content-Type").orEmpty()
val streamMode = header("X-Hermes-Stream-Mode").orEmpty()
val runStreaming = header("X-Hermes-Run-Streaming").orEmpty()
return contentType.contains("text/event-stream", ignoreCase = true) ||
streamMode.equals("sse", ignoreCase = true) ||
runStreaming.equals("sse", ignoreCase = true)
}
fun routeExplicitlySupportsEventStream(path: String): Boolean = try {
val req = authRequest("$baseUrl$path")
.head()
.header("Accept", "text/event-stream")
.build()
client.newCall(req).execute().use { response ->
response.code != 404 && response.advertisesEventStream()
}
} catch (_: Exception) {
false
}
val sessionsApi = routeExists("/api/sessions?limit=1")
val sessionsChatStream = routeExists("/api/sessions/probe/chat/stream")
val runs = routeExists("/v1/runs")
val portable = routeExists("/v1/models")
val portable = routeExists("/v1/chat/completions")
val runs = routeExplicitlySupportsEventStream("/v1/runs")
ServerCapabilities(
sessionsApi = sessionsApi,
@@ -1055,4 +1245,10 @@ class HermesApiClient(
}
return IOException(message)
}
private fun JsonObject?.stringField(name: String): String =
((this?.get(name) as? JsonPrimitive)?.contentOrNull ?: "").trim()
private fun firstNonBlank(vararg values: String?): String =
values.firstOrNull { !it.isNullOrBlank() }.orEmpty()
}
@@ -0,0 +1,145 @@
package com.hermesandroid.relay.network
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.Attachment
import kotlinx.serialization.json.JsonArray
import kotlinx.serialization.json.JsonObject
import kotlinx.serialization.json.add
import kotlinx.serialization.json.addJsonObject
import kotlinx.serialization.json.buildJsonArray
import kotlinx.serialization.json.buildJsonObject
import kotlinx.serialization.json.put
import kotlinx.serialization.json.putJsonArray
import kotlinx.serialization.json.putJsonObject
internal fun buildSessionChatStreamPayload(
message: String,
systemMessage: String? = null,
attachments: List<Attachment>? = null,
voiceIntentMessages: JsonArray? = null,
modelOverride: String? = null,
profileName: String? = null,
): JsonObject = buildJsonObject {
put("message", message)
if (!systemMessage.isNullOrBlank()) {
put("system_message", systemMessage)
}
if (!modelOverride.isNullOrBlank()) {
put("model", modelOverride)
}
AgentDisplay.profileRequestName(profileName)?.let { put("profile", it) }
if (!attachments.isNullOrEmpty()) {
putJsonArray("attachments") {
attachments.forEach { att ->
addJsonObject {
put("contentType", att.contentType)
put("content", att.content)
}
}
}
}
if (voiceIntentMessages != null && voiceIntentMessages.isNotEmpty()) {
put("messages", voiceIntentMessages)
}
}
internal fun buildRunStreamPayload(
message: String,
model: String? = null,
systemMessage: String? = null,
attachments: List<Attachment>? = null,
voiceIntentMessages: JsonArray? = null,
modelOverride: String? = null,
profileName: String? = null,
): JsonObject {
val resolvedModel = when {
!modelOverride.isNullOrBlank() -> modelOverride
!model.isNullOrBlank() -> model
else -> "default"
}
return buildJsonObject {
put("model", resolvedModel)
put("input", message)
put("stream", true)
if (!systemMessage.isNullOrBlank()) {
put("system_message", systemMessage)
}
AgentDisplay.profileRequestName(profileName)?.let { put("profile", it) }
if (!attachments.isNullOrEmpty()) {
putJsonArray("attachments") {
attachments.forEach { att ->
addJsonObject {
put("contentType", att.contentType)
put("content", att.content)
}
}
}
}
if (voiceIntentMessages != null && voiceIntentMessages.isNotEmpty()) {
put("messages", voiceIntentMessages)
}
}
}
internal fun buildChatCompletionsStreamPayload(
message: String,
model: String? = null,
systemMessage: String? = null,
attachments: List<Attachment>? = null,
voiceIntentMessages: JsonArray? = null,
modelOverride: String? = null,
profileName: String? = null,
): JsonObject {
val resolvedModel = when {
!modelOverride.isNullOrBlank() -> modelOverride
!model.isNullOrBlank() -> model
else -> "default"
}
return buildJsonObject {
put("model", resolvedModel)
put("stream", true)
AgentDisplay.profileRequestName(profileName)?.let { put("profile", it) }
putJsonArray("messages") {
if (!systemMessage.isNullOrBlank()) {
addJsonObject {
put("role", "system")
put("content", systemMessage)
}
}
if (voiceIntentMessages != null && voiceIntentMessages.isNotEmpty()) {
voiceIntentMessages.forEach { add(it) }
}
addJsonObject {
put("role", "user")
if (!attachments.isNullOrEmpty() && attachments.any { it.isImage }) {
put("content", buildJsonArray {
addJsonObject {
put("type", "text")
put("text", message)
}
attachments.filter { it.isImage }.forEach { att ->
addJsonObject {
put("type", "image_url")
putJsonObject("image_url") {
put("url", "data:${att.contentType};base64,${att.content}")
}
}
}
})
} else {
put("content", message)
}
}
}
if (!attachments.isNullOrEmpty() && attachments.any { !it.isImage }) {
putJsonArray("attachments") {
attachments.filter { !it.isImage }.forEach { att ->
addJsonObject {
put("contentType", att.contentType)
put("content", att.content)
}
}
}
}
}
}
@@ -0,0 +1,52 @@
package com.hermesandroid.relay.network
import java.net.URI
/**
* Resolves profile-scoped Hermes API URLs for phone use.
*
* Relays and Hermes gateways often bind profile API servers to loopback or
* 0.0.0.0 on the host machine. Those addresses are correct for the relay
* process but wrong on Android, where 127.0.0.1 means the phone. When the
* active connection uses a reachable LAN/Tailscale host, replace loopback
* profile hosts with that same host while preserving the profile port.
*/
object ProfileApiUrlResolver {
fun normalize(url: String?): String? =
url?.trim()?.takeIf { it.isNotBlank() }?.trimEnd('/')
fun resolveForConnection(profileApiUrl: String?, baseApiUrl: String?): String? {
val profile = normalize(profileApiUrl) ?: return null
val base = normalize(baseApiUrl) ?: return profile
val profileUri = runCatching { URI(profile) }.getOrNull() ?: return profile
val profileHost = profileUri.host?.takeIf { it.isNotBlank() } ?: return profile
if (!isLocalBindHost(profileHost)) return profile
val baseUri = runCatching { URI(base) }.getOrNull() ?: return profile
val baseHost = baseUri.host?.takeIf { it.isNotBlank() } ?: return profile
if (isLocalBindHost(baseHost)) return profile
val scheme = baseUri.scheme?.takeIf { it.isNotBlank() }
?: profileUri.scheme?.takeIf { it.isNotBlank() }
?: return profile
val hostPart = if (baseHost.contains(":") && !baseHost.startsWith("[")) {
"[$baseHost]"
} else {
baseHost
}
val portPart = profileUri.port.takeIf { it != -1 }?.let { ":$it" }.orEmpty()
val pathPart = profileUri.rawPath?.takeIf { it.isNotBlank() && it != "/" }.orEmpty()
val queryPart = profileUri.rawQuery?.let { "?$it" }.orEmpty()
val fragmentPart = profileUri.rawFragment?.let { "#$it" }.orEmpty()
return "$scheme://$hostPart$portPart$pathPart$queryPart$fragmentPart".trimEnd('/')
}
private fun isLocalBindHost(host: String): Boolean {
return when (host.lowercase().trim('[', ']')) {
"localhost", "127.0.0.1", "0.0.0.0", "::1", "::" -> true
else -> false
}
}
}
File diff suppressed because it is too large Load Diff
@@ -554,6 +554,31 @@ class ChatHandler {
dispatchedCardMarkers.clear()
}
/**
* Repair assistant labels after late-arriving agent config. History can
* load before GET /api/config returns, leaving default-profile messages
* with the generic "Hermes" label. Keep local phone/voice action trace
* labels intact because those bubbles do not represent the server agent.
*/
fun relabelGenericAssistantMessages(agentName: String?) {
val trimmed = agentName?.trim()?.takeIf { it.isNotBlank() } ?: return
_messages.update { list ->
list.map { message ->
if (
message.role == MessageRole.ASSISTANT &&
!message.id.startsWith("voice-intent-") &&
message.agentName != "Voice action" &&
message.agentName != "Phone action" &&
(message.agentName.isNullOrBlank() || message.agentName == "Hermes")
) {
message.copy(agentName = trimmed)
} else {
message
}
}
}
}
/**
* Replace a placeholder message's ID with the server-assigned ID.
* Only acts on empty, streaming messages to avoid renaming completed turns
@@ -667,6 +692,7 @@ class ChatHandler {
isStreaming = false,
toolCalls = toolCalls,
cards = extractedCards,
agentName = if (role == MessageRole.ASSISTANT) activeAgentName else null,
)
}
@@ -879,6 +905,10 @@ class ChatHandler {
}
}
fun clearSessions() {
_sessions.value = emptyList()
}
/**
* Remove a session from the local list (optimistic delete).
*/
@@ -112,7 +112,8 @@ data class SessionItem(
@Serializable
data class CreateSessionRequest(
val title: String? = null,
val model: String? = null
val model: String? = null,
val profile: String? = null,
)
@Serializable
@@ -73,6 +73,7 @@ import com.hermesandroid.relay.ui.components.UpdateBanner
import com.hermesandroid.relay.update.UpdateCheckResult
import com.hermesandroid.relay.viewmodel.UpdateViewModel
import com.hermesandroid.relay.ui.components.WhatsNewDialog
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.BridgePreferencesRepository
import com.hermesandroid.relay.data.BridgeSafetyPreferencesRepository
import com.hermesandroid.relay.data.BuildFlavor
@@ -95,6 +96,7 @@ import com.hermesandroid.relay.ui.screens.MediaSettingsScreen
import com.hermesandroid.relay.ui.screens.PairedDevicesScreen
import com.hermesandroid.relay.ui.screens.ConnectionsSettingsScreen
import com.hermesandroid.relay.ui.screens.ProfileInspectorScreen
import com.hermesandroid.relay.ui.screens.RealtimeVoiceTestScreen
import com.hermesandroid.relay.ui.screens.SettingsScreen
import com.hermesandroid.relay.ui.screens.TerminalScreen
import com.hermesandroid.relay.ui.screens.NotificationCompanionSettingsScreen
@@ -109,6 +111,7 @@ import com.hermesandroid.relay.viewmodel.VoiceViewModel
import com.hermesandroid.relay.audio.VoicePlayer
import com.hermesandroid.relay.audio.VoiceRecorder
import com.hermesandroid.relay.audio.VoiceSfxPlayer
import com.hermesandroid.relay.audio.RealtimePcmPlayer
import com.hermesandroid.relay.network.RelayVoiceClient
import com.hermesandroid.relay.auth.AuthState
import androidx.lifecycle.viewModelScope
@@ -223,6 +226,7 @@ sealed class Screen(
data object AppearanceSettings : Screen("settings/appearance", "Appearance", Icons.Filled.Settings)
data object Analytics : Screen("settings/analytics", "Analytics", Icons.Filled.Settings)
data object DeveloperSettings : Screen("settings/developer", "Developer", Icons.Filled.Settings)
data object RealtimeVoiceTest : Screen("settings/developer/realtime_voice", "Realtime voice", Icons.Filled.Settings)
data object About : Screen("settings/about", "About", Icons.Filled.Settings)
// Profile Inspector — full-screen read-only viewer with 4 tabs
@@ -350,10 +354,11 @@ fun RelayApp() {
}
}
// Initialize ChatViewModel reactively when API client becomes available
val apiClient by connectionViewModel.apiClient.collectAsState()
// Initialize ChatViewModel reactively when the chat-routed API client becomes available
val chatApiClient by connectionViewModel.chatApiClient.collectAsState()
val lastSessionId by connectionViewModel.lastSessionId.collectAsState()
var sessionResumed by remember { mutableStateOf(false) }
val selectedProfile by connectionViewModel.selectedProfile.collectAsState()
val activeConnectionId by connectionViewModel.activeConnectionId.collectAsState()
val mediaContext = androidx.compose.ui.platform.LocalContext.current
@@ -369,6 +374,9 @@ fun RelayApp() {
.connectTimeout(15, java.util.concurrent.TimeUnit.SECONDS)
.build(),
relayUrlProvider = { connectionViewModel.effectiveRelayUrl.value },
profileNameProvider = {
AgentDisplay.profileRequestName(connectionViewModel.selectedProfile.value?.name)
},
sessionTokenProvider = {
(connectionViewModel.authState.value as? AuthState.Paired)?.token
},
@@ -398,6 +406,7 @@ fun RelayApp() {
// VoiceSfxPlayer is internally crash-proof — failed AudioTrack builds
// become null-tracks that no-op — so we don't need an outer try/catch.
val voiceSfxPlayer = remember { VoiceSfxPlayer(mediaContext) }
val realtimePcmPlayer = remember { RealtimePcmPlayer() }
LaunchedEffect(Unit) {
val recorder = VoiceRecorder(mediaContext, voiceViewModel.viewModelScope)
val player = VoicePlayer(mediaContext)
@@ -406,6 +415,7 @@ fun RelayApp() {
chatViewModel = chatViewModel,
recorder = recorder,
player = player,
realtimePcmPlayer = realtimePcmPlayer,
sfxPlayer = voiceSfxPlayer,
// === PHASE3-voice-intents-localdispatch ===
// Wire the local in-process dispatcher so voice intents go
@@ -431,6 +441,15 @@ fun RelayApp() {
// app restarts. VoicePreferencesRepository is the same repo
// VoiceSettingsScreen reads/writes.
voicePreferences = com.hermesandroid.relay.data.VoicePreferencesRepository(mediaContext),
bargeInPreferences = com.hermesandroid.relay.data.BargeInPreferencesRepository(mediaContext),
vadEngineFactory = { com.hermesandroid.relay.audio.VadEngine(mediaContext) },
bargeInListenerFactory = { vad, audioSessionIdProvider ->
com.hermesandroid.relay.audio.BargeInListener.create(
mediaContext,
vad,
audioSessionIdProvider,
)
},
)
}
@@ -463,8 +482,8 @@ fun RelayApp() {
chatViewModel.observeConnectionSwitches(connectionViewModel.connectionSwitchEvents)
}
LaunchedEffect(apiClient) {
apiClient?.let { client ->
LaunchedEffect(chatApiClient) {
chatApiClient?.let { client ->
chatViewModel.initialize(client, connectionViewModel.chatHandler)
chatViewModel.updateApiClient(client)
@@ -485,20 +504,38 @@ fun RelayApp() {
chatViewModel.setSelectedProfileProvider {
connectionViewModel.selectedProfile.value
}
chatViewModel.setEffectiveProfileProvider {
AgentDisplay.effectiveProfile(
selectedProfile = connectionViewModel.selectedProfile.value,
profiles = connectionViewModel.agentProfiles.value,
)
}
// Wire session persistence callback
chatViewModel.onSessionChanged = { sessionId ->
connectionViewModel.saveLastSessionId(sessionId)
}
// Resume last session on first connection
if (!sessionResumed && lastSessionId != null) {
chatViewModel.resumeSession(lastSessionId!!)
sessionResumed = true
}
}
}
LaunchedEffect(chatApiClient, activeConnectionId, selectedProfile?.name, lastSessionId) {
if (chatApiClient == null) return@LaunchedEffect
chatViewModel.switchProfileContext(
contextKey = AgentDisplay.profileContextKey(
connectionId = activeConnectionId,
profileName = selectedProfile?.name,
),
sessionId = lastSessionId,
)
chatViewModel.refreshSessions()
}
LaunchedEffect(selectedProfile?.name) {
voiceViewModel.onProfileChanged(
AgentDisplay.profileRequestName(selectedProfile?.name)
)
}
// === PHASE3-status: sync granular phone-status settings to chat ===
val appContextEnabled by connectionViewModel.appContextEnabled.collectAsState()
val appContextBridgeState by connectionViewModel.appContextBridgeState.collectAsState()
@@ -530,8 +567,8 @@ fun RelayApp() {
// Sync streaming endpoint preference to chat. Resolves "auto" against the
// current server capabilities so vanilla upstream + bootstrap-injected
// sessions API picks /v1/runs for chat (which has live tool events)
// while still using /api/sessions/* for browse/rename/delete.
// sessions API picks /v1/chat/completions for portable SSE chat while
// still using /api/sessions/* for browse/rename/delete.
val streamingEndpoint by connectionViewModel.streamingEndpoint.collectAsState()
val serverCapabilities by connectionViewModel.serverCapabilities.collectAsState()
LaunchedEffect(streamingEndpoint, serverCapabilities) {
@@ -893,6 +930,7 @@ fun RelayApp() {
chatViewModel = chatViewModel,
connectionViewModel = connectionViewModel,
voiceViewModel = voiceViewModel,
voiceClient = voiceClient,
maxBubbleWidth = maxBubbleWidth,
openAgentSheetOnEntry = openAgentSheetArg,
onAgentSheetArgConsumed = {
@@ -995,6 +1033,7 @@ fun RelayApp() {
VoiceSettingsScreen(
voiceViewModel = voiceViewModel,
voiceClient = voiceClient,
selectedProfile = selectedProfile,
onBack = { navController.popBackStack() }
)
}
@@ -1199,7 +1238,16 @@ fun RelayApp() {
composable(Screen.DeveloperSettings.route) {
DeveloperSettingsScreen(
connectionViewModel = connectionViewModel,
onBack = { navController.popBackStack() }
onBack = { navController.popBackStack() },
onNavigateToRealtimeVoice = {
navController.navigate(Screen.RealtimeVoiceTest.route)
},
)
}
composable(Screen.RealtimeVoiceTest.route) {
RealtimeVoiceTestScreen(
voiceClient = voiceClient,
onBack = { navController.popBackStack() },
)
}
composable(Screen.About.route) {
@@ -60,6 +60,7 @@ import androidx.compose.ui.text.font.FontFamily
import androidx.compose.ui.text.font.FontWeight
import androidx.compose.ui.unit.dp
import com.hermesandroid.relay.auth.AuthState
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.AppAnalytics
import com.hermesandroid.relay.data.FeatureFlags
import com.hermesandroid.relay.data.Profile
@@ -668,12 +669,16 @@ fun AgentInfoSheet(
) {
// ---- Header: avatar + agent name + live status ----
AgentSheetHeader(
selectedProfile = selectedProfile,
profile = AgentDisplay.effectiveProfile(
selectedProfile = selectedProfile,
profiles = agentProfiles,
),
selectedPersonality = selectedPersonality,
defaultPersonality = defaultPersonality,
serverModelName = serverModelName,
apiServerReachable = apiServerReachable,
chatMode = chatMode,
isCustomized = selectedProfile != null || selectedPersonality != "default",
)
HorizontalDivider()
@@ -689,6 +694,10 @@ fun AgentInfoSheet(
// important when one of the profiles is literally named
// "default" so "Server default" vs "default" would be
// otherwise indistinguishable.
val serverDefaultProfile = agentProfiles
.firstOrNull { AgentDisplay.isServerDefaultAlias(it.name) }
val selectableProfiles = agentProfiles
.filterNot { AgentDisplay.isServerDefaultAlias(it.name) }
val apparentActiveProfile = agentProfiles
.firstOrNull { it.gatewayRunning }
@@ -698,25 +707,78 @@ fun AgentInfoSheet(
hint = "Overlay an agent's model + SOUL",
)
// "Server default" row — clears the profile override
// so the request uses whatever the server's own
// config.yaml picks. Renamed from "Default" so a
// profile literally named "default" doesn't collide
// with this row's label.
val defaultDotColor = serverDefaultProfile?.let { profile ->
if (profile.gatewayRunning) {
MaterialTheme.colorScheme.primary
} else {
MaterialTheme.colorScheme.onSurfaceVariant.copy(alpha = 0.3f)
}
}
val defaultDotA11y = serverDefaultProfile?.let { profile ->
if (profile.gatewayRunning) "Gateway running" else "Gateway idle"
}
val defaultRunning = serverDefaultProfile?.let { profile ->
if (profile.gatewayRunning) " \u2022 Running" else " \u2022 Idle"
}.orEmpty()
val defaultDisplay = serverDefaultProfile?.let { profile ->
AgentDisplay.profileDisplayName(profile)
?: profile.name.replaceFirstChar { it.uppercase() }
}
val defaultSecondary = serverDefaultProfile?.let { profile ->
listOfNotNull(
defaultDisplay,
profile.model.takeIf { it.isNotBlank() }?.plus(defaultRunning),
).joinToString(" \u2022 ")
} ?: "Use this connection's default profile"
val soulBg = MaterialTheme.colorScheme.primaryContainer
val soulFg = MaterialTheme.colorScheme.onPrimaryContainer
val skillsBg = MaterialTheme.colorScheme.surfaceVariant
val skillsFg = MaterialTheme.colorScheme.onSurfaceVariant
// "Server default" is the single selectable state for
// the root Hermes config. If the relay advertises a
// synthetic profile named "default" (usually displayed
// as Victor), fold its metadata into this row instead of
// creating a second chat/voice/session scope.
ProfileRadioRow(
primary = "Server default",
secondary = "Use this connection's default profile",
secondary = defaultSecondary,
tertiary = "Uses this connection's default profile",
selected = selectedProfile == null,
enabled = !isStreaming,
leadingDotColor = defaultDotColor,
leadingDotContentDescription = defaultDotA11y,
secondaryTrailing = serverDefaultProfile?.let { profile ->
if (profile.hasSoul || profile.skillCount > 0) {
{
if (profile.skillCount > 0) {
ProfileMetadataBadge(
text = "${profile.skillCount} skills",
background = skillsBg,
contentColor = skillsFg,
)
}
if (profile.hasSoul) {
ProfileMetadataBadge(
text = "SOUL",
background = soulBg,
contentColor = soulFg,
)
}
}
} else {
null
}
},
onSelect = {
if (selectedProfile != null) {
connectionViewModel.selectProfile(null)
toast("Using default model")
toast("Using Server default")
}
},
)
agentProfiles.forEach { profile ->
selectableProfiles.forEach { profile ->
// v0.7.0 runtime metadata indicators:
// - leadingDotColor: green when this profile's
// gateway is the live one, grey otherwise.
@@ -741,10 +803,6 @@ fun AgentInfoSheet(
} else {
" \u2022 Idle"
}
val soulBg = MaterialTheme.colorScheme.primaryContainer
val soulFg = MaterialTheme.colorScheme.onPrimaryContainer
val skillsBg = MaterialTheme.colorScheme.surfaceVariant
val skillsFg = MaterialTheme.colorScheme.onSurfaceVariant
val isApparentActive =
apparentActiveProfile?.name == profile.name
// Emphasize the description as the primary label
@@ -1269,20 +1327,21 @@ private fun ProfileMetadataBadge(
@Composable
private fun AgentSheetHeader(
selectedProfile: Profile?,
profile: Profile?,
selectedPersonality: String,
defaultPersonality: String,
serverModelName: String,
apiServerReachable: Boolean,
chatMode: ChatMode,
isCustomized: Boolean,
) {
val agentName = when {
selectedProfile != null -> selectedProfile.name.replaceFirstChar { it.uppercase() }
selectedPersonality != "default" -> selectedPersonality.replaceFirstChar { it.uppercase() }
defaultPersonality.isNotBlank() -> defaultPersonality.replaceFirstChar { it.uppercase() }
else -> "Hermes"
}
val modelLabel = selectedProfile?.model ?: serverModelName
val agentName = AgentDisplay.agentName(
profile = profile,
selectedPersonality = selectedPersonality,
defaultPersonality = defaultPersonality,
connectionLabel = null,
)
val modelLabel = profile?.model ?: serverModelName
val isConnecting = !apiServerReachable && chatMode != ChatMode.DISCONNECTED
val statusText = when {
apiServerReachable -> "Connected"
@@ -1294,8 +1353,6 @@ private fun AgentSheetHeader(
isConnecting -> MaterialTheme.colorScheme.tertiary
else -> MaterialTheme.colorScheme.error
}
val customized = selectedProfile != null || selectedPersonality != "default"
Row(
verticalAlignment = Alignment.CenterVertically,
horizontalArrangement = Arrangement.spacedBy(12.dp),
@@ -1304,7 +1361,7 @@ private fun AgentSheetHeader(
modifier = Modifier
.size(48.dp)
.then(
if (customized) {
if (isCustomized) {
Modifier.border(
width = 2.dp,
color = MaterialTheme.colorScheme.primary,
@@ -44,6 +44,8 @@ import java.util.Locale
fun SessionDrawerContent(
sessions: List<ChatSession>,
currentSessionId: String?,
scopeTitle: String = "Sessions",
scopeSubtitle: String? = null,
onNewChat: () -> Unit,
onSelectSession: (String) -> Unit,
onDeleteSession: (String) -> Unit,
@@ -56,9 +58,19 @@ fun SessionDrawerContent(
Column(modifier = Modifier.padding(16.dp)) {
// Header
Text(
text = "Sessions",
text = scopeTitle,
style = MaterialTheme.typography.titleLarge
)
scopeSubtitle?.takeIf { it.isNotBlank() }?.let { subtitle ->
Spacer(modifier = Modifier.height(2.dp))
Text(
text = subtitle,
style = MaterialTheme.typography.bodySmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
maxLines = 2,
overflow = TextOverflow.Ellipsis,
)
}
Spacer(modifier = Modifier.height(8.dp))
@@ -682,6 +682,14 @@ private fun VoiceSection(stats: VoiceStats) {
label = "Avg latency (last ${stats.recentTtsLatenciesMs.size.coerceAtLeast(0)})",
value = if (stats.recentTtsLatenciesMs.isEmpty()) "—" else formatMsWithSeconds(avgTtsLatency),
)
KeyValueRow(
label = "Response chunks",
value = "${stats.currentResponseTtsChunks}",
)
KeyValueRow(
label = "Last chunk gap",
value = if (stats.currentResponseTtsChunks <= 1) "—" else formatMsWithSeconds(stats.lastTtsChunkGapMs),
)
KeyValueRow(
label = "Received",
value = formatByteCount(stats.ttsBytesReceived),
@@ -400,7 +400,7 @@ internal fun buildTimelineEvents(
timestampMs = System.currentTimeMillis(),
kind = TimelineEventKind.VoiceTurn,
title = "voice · ${voiceStats.lastTranscript.take(40)}",
details = "stt ${voiceStats.lastSttLatencyMs} ms · tts calls ${voiceStats.ttsCallCount}",
details = "stt ${voiceStats.lastSttLatencyMs} ms · chunks ${voiceStats.currentResponseTtsChunks} · tts calls ${voiceStats.ttsCallCount}",
),
)
}
File diff suppressed because it is too large Load Diff
@@ -1,6 +1,7 @@
package com.hermesandroid.relay.ui.components
import androidx.compose.animation.animateColorAsState
import androidx.compose.animation.core.animateFloatAsState
import androidx.compose.animation.core.tween
import androidx.compose.foundation.Canvas
import androidx.compose.foundation.background
@@ -20,6 +21,7 @@ import androidx.compose.runtime.setValue
import androidx.compose.runtime.withFrameNanos
import androidx.compose.ui.Modifier
import androidx.compose.ui.geometry.Offset
import androidx.compose.ui.geometry.Size
import androidx.compose.ui.graphics.BlendMode
import androidx.compose.ui.graphics.Brush
import androidx.compose.ui.graphics.Color
@@ -32,6 +34,7 @@ import androidx.compose.ui.unit.dp
import com.hermesandroid.relay.viewmodel.VoiceState
import kotlin.math.PI
import kotlin.math.max
import kotlin.math.min
import kotlin.math.sin
/**
@@ -113,6 +116,7 @@ private const val EDGE_FADE_FRACTION = 0.12f
fun VoiceWaveform(
amplitude: Float,
state: VoiceState,
outputAudioActive: Boolean = false,
modifier: Modifier = Modifier,
) {
// No downstream smoothing. The VoiceViewModel already runs an
@@ -160,6 +164,16 @@ fun VoiceWaveform(
// velocity scales with the current amplitude so the wave visibly surges
// when the user speaks instead of running on its own fixed clock.
val phases = rememberAmplitudeDrivenPhases(phaseDurationsMs, displayAmplitude)
val waitingForOutputAudio = state == VoiceState.Speaking && !outputAudioActive
val processing = state == VoiceState.Transcribing ||
state == VoiceState.Thinking ||
waitingForOutputAudio
val waveformUnfold by animateFloatAsState(
targetValue = if (processing) 0f else 1f,
animationSpec = tween(durationMillis = 360),
label = "waveformUnfold",
)
val spinnerPhase = rememberProcessingSpinnerPhase(processing)
Canvas(
modifier = modifier
@@ -173,9 +187,38 @@ fun VoiceWaveform(
val centerY = height / 2f
val peakPixels = height * PEAK_FRACTION
val strokePx = STROKE_WIDTH_DP.dp.toPx()
val centerX = width / 2f
val spinnerAlpha = 1f - waveformUnfold
if (spinnerAlpha > 0.01f) {
val spinnerSize = min(width, height) * 0.64f
val radius = spinnerSize / 2f
val topLeft = Offset(centerX - radius, centerY - radius)
val arcSize = Size(spinnerSize, spinnerSize)
val sweep = 82f + displayAmplitude * 64f
drawArc(
color = primaryColor.copy(alpha = primaryColor.alpha * spinnerAlpha * 0.82f),
startAngle = spinnerPhase,
sweepAngle = sweep,
useCenter = false,
topLeft = topLeft,
size = arcSize,
style = Stroke(width = strokePx * 1.35f, cap = StrokeCap.Round),
)
drawArc(
color = secondaryColor.copy(alpha = secondaryColor.alpha * spinnerAlpha * 0.58f),
startAngle = spinnerPhase + 152f,
sweepAngle = 42f + displayAmplitude * 36f,
useCenter = false,
topLeft = topLeft,
size = arcSize,
style = Stroke(width = strokePx * 1.05f, cap = StrokeCap.Round),
)
}
// Sample every SAMPLE_STEP_PX pixels across the canvas. steps+1 vertices.
val steps = max(2, (width / SAMPLE_STEP_PX).toInt())
if (waveformUnfold <= 0.01f) return@Canvas
// Offscreen layer so we can mask the stroke alpha with DstIn and the
// mask never bleeds onto whatever's underneath the waveform. Without
@@ -199,12 +242,13 @@ fun VoiceWaveform(
// true silence (amplitude = 0).
val rawEnvelope = displayAmplitude * layerScale * peakPixels
val minEnvelope = MIN_ENVELOPE * peakPixels * layerScale
val envelope = max(rawEnvelope, minEnvelope)
val envelope = max(rawEnvelope, minEnvelope) * waveformUnfold
val path = Path()
path.moveTo(0f, centerY)
path.moveTo(centerX - (centerX * waveformUnfold), centerY)
for (i in 0..steps) {
val x = i * (width / steps)
val foldedX = centerX + (x - centerX) * waveformUnfold
val t = x / width
val angle = (t * freq * 2f * PI).toFloat() + phase
// Geometric tuck-in: sin(πt) goes 0 → 1 → 0 across the
@@ -214,15 +258,15 @@ fun VoiceWaveform(
// without any hard cut at the canvas boundary.
val taper = sin(PI.toFloat() * t)
val y = centerY + sin(angle) * envelope * taper
if (i == 0) path.moveTo(x, y) else path.lineTo(x, y)
if (i == 0) path.moveTo(foldedX, y) else path.lineTo(foldedX, y)
}
drawPath(
path = path,
brush = Brush.horizontalGradient(
listOf(
primaryColor.copy(alpha = primaryColor.alpha * layerAlpha),
secondaryColor.copy(alpha = secondaryColor.alpha * layerAlpha),
primaryColor.copy(alpha = primaryColor.alpha * layerAlpha * waveformUnfold),
secondaryColor.copy(alpha = secondaryColor.alpha * layerAlpha * waveformUnfold),
),
),
style = Stroke(width = strokePx, cap = StrokeCap.Round),
@@ -302,6 +346,29 @@ private fun rememberAmplitudeDrivenPhases(
return phases
}
@Composable
private fun rememberProcessingSpinnerPhase(active: Boolean): Float {
val activeRef = rememberUpdatedState(active)
var phase by remember { mutableStateOf(0f) }
LaunchedEffect(Unit) {
var prevNanos = 0L
while (true) {
withFrameNanos { nanos ->
if (prevNanos == 0L) {
prevNanos = nanos
return@withFrameNanos
}
val dtSec = (nanos - prevNanos) / 1_000_000_000f
prevNanos = nanos
if (activeRef.value) {
phase = (phase + dtSec * 220f) % 360f
}
}
}
}
return phase
}
// ── Previews ─────────────────────────────────────────────────────────
@Preview(
@@ -59,6 +59,7 @@ import androidx.compose.material3.TopAppBar
import androidx.compose.material3.TopAppBarDefaults
import androidx.compose.material3.rememberDrawerState
import androidx.compose.runtime.Composable
import androidx.compose.runtime.DisposableEffect
import androidx.compose.runtime.LaunchedEffect
import androidx.compose.runtime.collectAsState
import androidx.compose.runtime.derivedStateOf
@@ -89,6 +90,8 @@ import com.hermesandroid.relay.ui.theme.purpleGlow
import com.hermesandroid.relay.ui.theme.radialNavyBackground
import com.hermesandroid.relay.network.ChatMode
import com.hermesandroid.relay.network.ConnectivityObserver
import com.hermesandroid.relay.network.RelayVoiceClient
import com.hermesandroid.relay.network.VoiceOutputConfig
import androidx.compose.animation.AnimatedContent
import androidx.compose.animation.core.animateFloatAsState
import androidx.compose.animation.core.tween
@@ -115,6 +118,7 @@ import androidx.compose.material.icons.filled.Add
import androidx.compose.material.icons.filled.Close
import androidx.compose.material.icons.filled.Description
import androidx.compose.ui.platform.LocalContext
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.Attachment
import com.hermesandroid.relay.data.displayLabel
import com.hermesandroid.relay.ui.components.AgentInfoSheet
@@ -136,6 +140,12 @@ import com.hermesandroid.relay.ui.showHumanError
import com.hermesandroid.relay.viewmodel.ChatViewModel
import com.hermesandroid.relay.viewmodel.ConnectionViewModel
import com.hermesandroid.relay.viewmodel.VoiceViewModel
import com.hermesandroid.relay.voice.VoiceOverlayHost
import com.hermesandroid.relay.voice.VoiceOverlaySession
import com.hermesandroid.relay.voice.openHermesFromOverlay
import androidx.lifecycle.Lifecycle
import androidx.lifecycle.LifecycleEventObserver
import androidx.lifecycle.compose.LocalLifecycleOwner
import kotlinx.coroutines.delay
import kotlinx.coroutines.launch
@@ -165,6 +175,7 @@ fun ChatScreen(
chatViewModel: ChatViewModel,
connectionViewModel: ConnectionViewModel,
voiceViewModel: VoiceViewModel,
voiceClient: RelayVoiceClient? = null,
maxBubbleWidth: Dp = 300.dp,
// Deep-link nudge from Settings → Active Agent card: when `true`, the
// AgentInfoSheet auto-opens on first composition and [onAgentSheetArgConsumed]
@@ -179,8 +190,9 @@ fun ChatScreen(
onNavigateToConnections: () -> Unit = {},
) {
val voiceUiState by voiceViewModel.uiState.collectAsState()
var voiceCompactMode by remember { mutableStateOf(false) }
val chatAlpha by animateFloatAsState(
targetValue = if (voiceUiState.voiceMode) 0.4f else 1f,
targetValue = if (voiceUiState.voiceMode && !voiceCompactMode) 0.4f else 1f,
animationSpec = tween(300),
label = "chatAlpha",
)
@@ -198,8 +210,11 @@ fun ChatScreen(
// request; on grant, latch the pending-enter and fire enterVoiceMode() in
// the callback. Denial shows an inline banner above the input.
val context = LocalContext.current
val lifecycleOwner = LocalLifecycleOwner.current
val voiceOverlayHost = remember { VoiceOverlayHost.install(context) }
var pendingVoiceEnter by remember { mutableStateOf(false) }
var micPermissionDenied by remember { mutableStateOf(false) }
var pendingVoiceOverlayPermission by remember { mutableStateOf(false) }
val micPermissionLauncher = rememberLauncherForActivityResult(
contract = ActivityResultContracts.RequestPermission(),
) { granted ->
@@ -231,6 +246,7 @@ fun ChatScreen(
val messages by chatViewModel.messages.collectAsState()
val isStreaming by chatViewModel.isStreaming.collectAsState()
var voiceOutputConfig by remember { mutableStateOf<VoiceOutputConfig?>(null) }
val chatReady by connectionViewModel.chatReady.collectAsState()
// Voice mode's /voice/transcribe and /voice/synthesize calls both go
// over the relay, but voice can authenticate with the saved Hermes API
@@ -333,6 +349,107 @@ fun ChatScreen(
val haptic = LocalHapticFeedback.current
val snackbarHostState = remember { SnackbarHostState() }
val showVoiceSystemOverlay: () -> Unit = {
if (!voiceOverlayHost.hasOverlayPermission()) {
pendingVoiceOverlayPermission = true
runCatching {
val intent = Intent(
Settings.ACTION_MANAGE_OVERLAY_PERMISSION,
Uri.parse("package:${context.packageName}"),
).apply { addFlags(Intent.FLAG_ACTIVITY_NEW_TASK) }
context.startActivity(intent)
}
scope.launch {
snackbarHostState.showSnackbar(
message = "Enable Display over other apps, then return to start Voice Overlay.",
duration = SnackbarDuration.Short,
)
}
} else {
pendingVoiceOverlayPermission = false
val shown = voiceOverlayHost.show(
VoiceOverlaySession(
uiState = voiceViewModel.uiState,
provider = voiceOutputConfig?.default_provider,
model = voiceOutputConfig?.default_model,
voice = voiceOutputConfig?.default_voice,
profileName = selectedProfile?.description?.takeIf { it.isNotBlank() }
?: selectedProfile?.name,
configScope = voiceOutputConfig?.configScope,
outputEnabled = voiceOutputConfig?.enabled,
fallbackEnabled = voiceOutputConfig?.fallback_enabled,
onStartListening = { voiceViewModel.startListening() },
onStopListening = { voiceViewModel.stopListening() },
onInterrupt = { voiceViewModel.interruptSpeaking() },
onPauseAutoMode = { voiceViewModel.pauseContinuousMode() },
onReturnToHermes = {
openHermesFromOverlay(context)
voiceOverlayHost.hide()
},
onDismissOverlay = { voiceOverlayHost.hide() },
onExit = {
voiceOverlayHost.hide()
voiceViewModel.exitVoiceMode()
},
),
)
if (!shown) {
scope.launch {
snackbarHostState.showSnackbar(
message = "Voice Overlay could not be started.",
duration = SnackbarDuration.Short,
)
}
}
}
}
LaunchedEffect(voiceUiState.voiceMode) {
if (!voiceUiState.voiceMode) {
voiceOverlayHost.hide()
pendingVoiceOverlayPermission = false
voiceCompactMode = false
}
}
DisposableEffect(
lifecycleOwner,
pendingVoiceOverlayPermission,
voiceUiState.voiceMode,
voiceOutputConfig,
) {
val observer = LifecycleEventObserver { _, event ->
if (event == Lifecycle.Event.ON_RESUME && pendingVoiceOverlayPermission) {
when {
voiceOverlayHost.hasOverlayPermission() && voiceUiState.voiceMode -> {
showVoiceSystemOverlay()
}
!voiceOverlayHost.hasOverlayPermission() -> {
pendingVoiceOverlayPermission = false
scope.launch {
snackbarHostState.showSnackbar(
message = "Voice Overlay permission was not granted.",
duration = SnackbarDuration.Short,
)
}
}
else -> pendingVoiceOverlayPermission = false
}
}
}
lifecycleOwner.lifecycle.addObserver(observer)
onDispose { lifecycleOwner.lifecycle.removeObserver(observer) }
}
LaunchedEffect(voiceClient, voiceUiState.voiceMode, selectedProfile?.name) {
if (!voiceUiState.voiceMode) return@LaunchedEffect
val client = voiceClient ?: return@LaunchedEffect
val result = client.getVoiceOutputConfig()
if (result.isSuccess) {
voiceOutputConfig = result.getOrNull()
}
}
// File picker for attachments (any file type)
val filePickerLauncher = rememberLauncherForActivityResult(
contract = ActivityResultContracts.OpenMultipleDocuments()
@@ -607,8 +724,10 @@ fun ChatScreen(
// loading and a "default" entry shows up.
val effectiveProfile by remember(selectedProfile, agentProfiles) {
derivedStateOf {
selectedProfile
?: agentProfiles.firstOrNull { it.name.equals("default", ignoreCase = true) }
AgentDisplay.effectiveProfile(
selectedProfile = selectedProfile,
profiles = agentProfiles,
)
}
}
@@ -629,34 +748,37 @@ fun ChatScreen(
val agentDisplayName by remember {
derivedStateOf {
val profile = effectiveProfile
val fromProfile = when {
profile == null -> null
profile.description.isNotBlank() -> profile.description
profile.name.isNotBlank() -> profile.name.replaceFirstChar { it.uppercase() }
else -> null
}
fromProfile ?: run {
val personalityName = if (selectedPersonality == "default" && defaultPersonality.isNotBlank()) {
defaultPersonality
} else {
selectedPersonality
}
when {
personalityName.isNotBlank() && personalityName != "default" ->
personalityName.replaceFirstChar { it.uppercase() }
!activeConnection?.label.isNullOrBlank() -> activeConnection!!.label
else -> ""
}
}
AgentDisplay.agentName(
profile = profile,
selectedPersonality = selectedPersonality,
defaultPersonality = defaultPersonality,
connectionLabel = activeConnection?.label,
)
}
}
ModalNavigationDrawer(
drawerState = drawerState,
drawerContent = {
val drawerTitle = if (selectedProfile != null) {
"$agentDisplayName sessions"
} else {
"Server default sessions"
}
val drawerSubtitle = when {
selectedProfile?.hasIsolatedApi == true ->
"Profile API: ${selectedProfile?.apiServerUrl}"
selectedProfile != null ->
"Compatibility overlay on ${activeConnection?.label ?: "active connection"}"
activeConnection?.label?.isNotBlank() == true ->
"Connection: ${activeConnection?.label}"
else -> "Active connection"
}
SessionDrawerContent(
sessions = sessions,
currentSessionId = currentSessionId,
scopeTitle = drawerTitle,
scopeSubtitle = drawerSubtitle,
onNewChat = {
chatViewModel.createNewChat()
scope.launch { drawerState.close() }
@@ -721,13 +843,10 @@ fun ChatScreen(
// Model priority: profile.model (explicit or default
// profile pick) trumps /api/config's `serverModelName`.
// The profile picker is the more specific intent.
val personalityLabel = when {
selectedPersonality != "default" ->
selectedPersonality.replaceFirstChar { it.uppercase() }
defaultPersonality.isNotBlank() ->
defaultPersonality.replaceFirstChar { it.uppercase() }
else -> "Default"
}
val personalityLabel = AgentDisplay.personalityLabel(
selectedPersonality = selectedPersonality,
defaultPersonality = defaultPersonality,
)
val modelName = effectiveProfile?.model
?.takeIf { it.isNotBlank() }
?: serverModelName
@@ -1531,6 +1650,7 @@ fun ChatScreen(
onMicTap = { voiceViewModel.startListening() },
onMicRelease = { voiceViewModel.stopListening() },
onInterrupt = { voiceViewModel.interruptSpeaking() },
onPauseAutoMode = { voiceViewModel.pauseContinuousMode() },
onDismiss = { voiceViewModel.exitVoiceMode() },
onModeChange = { voiceViewModel.setInteractionMode(it) },
onClearError = { voiceViewModel.clearError() },
@@ -1540,9 +1660,22 @@ fun ChatScreen(
// Voice-first transcript: pass the last N chat messages so
// voice mode can show a compact rolling history including
// local-only voice-intent traces (agentName="Voice action").
// Bounded to 6 to keep voice mode visually focused on sphere
// + waveform + mic — the full scroll lives in the chat tab.
transcriptMessages = messages.takeLast(6),
// Bounded to 12 to keep voice mode focused while still
// preserving enough recent tool/context rows for voice turns.
transcriptMessages = messages.takeLast(12),
showThinking = showThinking,
voiceOutputProvider = voiceOutputConfig?.default_provider,
voiceOutputModel = voiceOutputConfig?.default_model,
voiceOutputVoice = voiceOutputConfig?.default_voice,
voiceProfileName = selectedProfile?.description?.takeIf { it.isNotBlank() }
?: selectedProfile?.name,
voiceConfigScope = voiceOutputConfig?.configScope,
voiceOutputEnabled = voiceOutputConfig?.enabled,
voiceOutputFallbackEnabled = voiceOutputConfig?.fallback_enabled,
onOverlayRequest = showVoiceSystemOverlay,
onCompactModeChange = { compact ->
voiceCompactMode = compact
},
// === v0.4.1 JIT permission-denied chip ===
// Tap deep-links to Settings → Apps → Hermes Relay →
// Permissions for the running package. Use BuildConfig
@@ -372,12 +372,20 @@ fun ChatSettingsScreen(
HorizontalDivider()
// Parse tool annotations toggle (Sessions mode only)
val isSessionsMode = streamingEndpoint == "sessions"
val serverCaps by connectionViewModel.serverCapabilities.collectAsState()
val resolvedStreamingEndpoint = if (streamingEndpoint == "auto") {
serverCaps.preferredChatEndpoint()
} else {
streamingEndpoint
}
// Parse tool annotations toggle (text-stream endpoints only)
val isTextAnnotationMode = resolvedStreamingEndpoint == "sessions" ||
resolvedStreamingEndpoint == "completions"
Row(
modifier = Modifier
.fillMaxWidth()
.then(if (!isSessionsMode) Modifier.alpha(0.5f) else Modifier),
.then(if (!isTextAnnotationMode) Modifier.alpha(0.5f) else Modifier),
horizontalArrangement = Arrangement.SpaceBetween,
verticalAlignment = Alignment.CenterVertically
) {
@@ -403,19 +411,19 @@ fun ChatSettingsScreen(
)
}
Text(
text = if (isSessionsMode) {
"Detect tool usage from text markers. May delay message display until stream completes."
text = if (isTextAnnotationMode) {
"Detect tool usage from text markers on endpoints without structured tool events."
} else {
"Only available in Sessions mode — Runs already provides structured tool events"
"Only available for text-stream endpoints — Runs should provide structured tool events."
},
style = MaterialTheme.typography.bodySmall,
color = MaterialTheme.colorScheme.onSurfaceVariant
)
}
Switch(
checked = parseToolAnnotations && isSessionsMode,
checked = parseToolAnnotations && isTextAnnotationMode,
onCheckedChange = { connectionViewModel.setParseToolAnnotations(it) },
enabled = isSessionsMode
enabled = isTextAnnotationMode
)
}
@@ -427,18 +435,21 @@ fun ChatSettingsScreen(
text = "Streaming endpoint",
style = MaterialTheme.typography.bodyMedium
)
val serverCaps by connectionViewModel.serverCapabilities.collectAsState()
val resolvedHelp = when (streamingEndpoint) {
"auto" -> {
val resolved = serverCaps.preferredChatEndpoint()
"Auto: picks the best path based on what your server exposes. " +
"Currently using: $resolved" +
if (!serverCaps.sessionsChatStream && serverCaps.sessionsApi)
" (sessions browse via /api/sessions, chat via /v1/runs)"
else ""
"Currently using: $resolvedStreamingEndpoint" +
when {
!serverCaps.sessionsChatStream && serverCaps.portable ->
" (chat via /v1/chat/completions)"
!serverCaps.sessionsChatStream && serverCaps.runs ->
" (chat via explicitly streamed /v1/runs)"
else -> ""
}
}
"sessions" -> "Sessions: tool calls shown as inline text annotations."
"runs" -> "Runs: structured tool events with real-time progress cards."
"sessions" -> "Sessions: Hermes-native /api/sessions/{id}/chat/stream."
"completions" -> "Chat: OpenAI-compatible SSE via /v1/chat/completions."
"runs" -> "Runs: use only when your server streams /v1/runs directly."
else -> ""
}
Text(
@@ -447,8 +458,8 @@ fun ChatSettingsScreen(
color = MaterialTheme.colorScheme.onSurfaceVariant
)
val endpointOptions = listOf("auto", "sessions", "runs")
val endpointLabels = listOf("Auto", "Sessions", "Runs")
val endpointOptions = listOf("auto", "sessions", "completions", "runs")
val endpointLabels = listOf("Auto", "Sessions", "Chat", "Runs")
val selectedEndpointIndex = endpointOptions.indexOf(streamingEndpoint).coerceAtLeast(0)
SingleChoiceSegmentedButtonRow(modifier = Modifier.fillMaxWidth()) {
@@ -62,6 +62,7 @@ import kotlinx.coroutines.launch
fun DeveloperSettingsScreen(
connectionViewModel: ConnectionViewModel,
onBack: () -> Unit,
onNavigateToRealtimeVoice: () -> Unit = {},
) {
val context = LocalContext.current
val scope = rememberCoroutineScope()
@@ -329,6 +330,43 @@ fun DeveloperSettingsScreen(
HorizontalDivider()
Row(
modifier = Modifier.fillMaxWidth(),
horizontalArrangement = Arrangement.SpaceBetween,
verticalAlignment = Alignment.CenterVertically
) {
Column(modifier = Modifier.weight(1f)) {
Row(
verticalAlignment = Alignment.CenterVertically,
horizontalArrangement = Arrangement.spacedBy(6.dp)
) {
Icon(
imageVector = Icons.Filled.Science,
contentDescription = null,
modifier = Modifier.size(16.dp),
tint = MaterialTheme.colorScheme.tertiary
)
Text(
text = "Realtime voice lab",
style = MaterialTheme.typography.bodyMedium
)
}
Text(
text = "Open the provider websocket testbench for dev builds",
style = MaterialTheme.typography.bodySmall,
color = MaterialTheme.colorScheme.onSurfaceVariant
)
}
IconButton(onClick = onNavigateToRealtimeVoice) {
Icon(
imageVector = Icons.Filled.Science,
contentDescription = "Open realtime voice lab"
)
}
}
HorizontalDivider()
// Lock developer options
Row(
modifier = Modifier.fillMaxWidth(),
@@ -0,0 +1,444 @@
package com.hermesandroid.relay.ui.screens
import android.Manifest
import android.content.pm.PackageManager
import android.os.Handler
import android.os.Looper
import androidx.activity.compose.rememberLauncherForActivityResult
import androidx.activity.result.contract.ActivityResultContracts
import androidx.compose.foundation.Canvas
import androidx.compose.foundation.layout.Arrangement
import androidx.compose.foundation.layout.Column
import androidx.compose.foundation.layout.Row
import androidx.compose.foundation.layout.Spacer
import androidx.compose.foundation.layout.fillMaxSize
import androidx.compose.foundation.layout.fillMaxWidth
import androidx.compose.foundation.layout.height
import androidx.compose.foundation.layout.padding
import androidx.compose.foundation.layout.size
import androidx.compose.foundation.rememberScrollState
import androidx.compose.foundation.shape.RoundedCornerShape
import androidx.compose.foundation.verticalScroll
import androidx.compose.material.icons.Icons
import androidx.compose.material.icons.automirrored.filled.ArrowBack
import androidx.compose.material.icons.filled.GraphicEq
import androidx.compose.material.icons.filled.Mic
import androidx.compose.material.icons.filled.PlayArrow
import androidx.compose.material3.Card
import androidx.compose.material3.CardDefaults
import androidx.compose.material3.ExperimentalMaterial3Api
import androidx.compose.material3.FilledTonalButton
import androidx.compose.material3.HorizontalDivider
import androidx.compose.material3.Icon
import androidx.compose.material3.IconButton
import androidx.compose.material3.MaterialTheme
import androidx.compose.material3.OutlinedTextField
import androidx.compose.material3.Scaffold
import androidx.compose.material3.Text
import androidx.compose.material3.TopAppBar
import androidx.compose.material3.TopAppBarDefaults
import androidx.compose.runtime.Composable
import androidx.compose.runtime.DisposableEffect
import androidx.compose.runtime.LaunchedEffect
import androidx.compose.runtime.getValue
import androidx.compose.runtime.mutableStateListOf
import androidx.compose.runtime.mutableStateOf
import androidx.compose.runtime.remember
import androidx.compose.runtime.rememberCoroutineScope
import androidx.compose.runtime.setValue
import androidx.compose.ui.Alignment
import androidx.compose.ui.Modifier
import androidx.compose.ui.geometry.Offset
import androidx.compose.ui.graphics.Color
import androidx.compose.ui.graphics.StrokeCap
import androidx.compose.ui.platform.LocalContext
import androidx.compose.ui.unit.dp
import androidx.core.content.ContextCompat
import com.hermesandroid.relay.audio.RealtimePcmPlayer
import com.hermesandroid.relay.audio.RealtimePcmRecorder
import com.hermesandroid.relay.network.RealtimeVoiceConfig
import com.hermesandroid.relay.network.RealtimeVoiceEvent
import com.hermesandroid.relay.network.RealtimeVoiceSummary
import com.hermesandroid.relay.network.RelayVoiceClient
import kotlinx.coroutines.launch
import java.util.Base64
/**
* Dev-only realtime provider testbench for Android Studio builds.
*/
@OptIn(ExperimentalMaterial3Api::class)
@Composable
fun RealtimeVoiceTestScreen(
voiceClient: RelayVoiceClient?,
onBack: () -> Unit,
) {
val context = LocalContext.current
val scope = rememberCoroutineScope()
val mainHandler = remember { Handler(Looper.getMainLooper()) }
val player = remember { RealtimePcmPlayer() }
val recorder = remember { RealtimePcmRecorder() }
val waveform = remember { mutableStateListOf<Float>() }
val events = remember { mutableStateListOf<String>() }
var prompt by remember {
mutableStateOf("Let me check that. Confirm the realtime voice provider path is working.")
}
var config by remember { mutableStateOf<RealtimeVoiceConfig?>(null) }
var status by remember { mutableStateOf("Idle") }
var summary by remember { mutableStateOf<RealtimeVoiceSummary?>(null) }
var running by remember { mutableStateOf(false) }
val permissionLauncher = rememberLauncherForActivityResult(
ActivityResultContracts.RequestPermission()
) { granted ->
if (granted) {
scope.launch {
runRealtimeDemo(
voiceClient = voiceClient,
recorder = recorder,
player = player,
prompt = prompt,
mainHandler = mainHandler,
waveform = waveform,
events = events,
onStatus = { status = it },
onSummary = { summary = it },
onRunning = { running = it },
)
}
} else {
status = "Microphone permission denied"
}
}
LaunchedEffect(voiceClient) {
val client = voiceClient ?: return@LaunchedEffect
val result = client.getRealtimeVoiceConfig()
if (result.isSuccess) {
config = result.getOrNull()
status = if (config?.enabled == true) "Ready" else "Realtime route disabled"
} else {
status = result.exceptionOrNull()?.message ?: "Realtime config failed"
}
}
DisposableEffect(Unit) {
onDispose { player.stop() }
}
Scaffold(
topBar = {
TopAppBar(
title = { Text("Realtime Voice Lab") },
navigationIcon = {
IconButton(onClick = onBack) {
Icon(
imageVector = Icons.AutoMirrored.Filled.ArrowBack,
contentDescription = "Back",
)
}
},
colors = TopAppBarDefaults.topAppBarColors(
containerColor = MaterialTheme.colorScheme.surface,
),
)
}
) { innerPadding ->
Column(
modifier = Modifier
.fillMaxSize()
.padding(innerPadding)
.verticalScroll(rememberScrollState())
.padding(16.dp),
verticalArrangement = Arrangement.spacedBy(16.dp),
) {
SectionCard(title = "Provider") {
ProviderRow("Status", status)
ProviderRow("Provider", config?.default_provider ?: "loading")
ProviderRow("Model", config?.default_model ?: "loading")
ProviderRow("Voice", config?.default_voice ?: "loading")
ProviderRow(
"Advertised",
config?.providers
?.map { it.id }
?.filter { it.isNotBlank() }
?.joinToString(", ")
?.ifBlank { "none" }
?: "loading",
)
ProviderRow(
"Auth",
when {
config?.auth?.xai_oauth == true -> "xAI OAuth"
config?.auth?.xai_env == true -> "xAI env"
else -> "not detected"
},
)
}
SectionCard(title = "Prompt") {
OutlinedTextField(
value = prompt,
onValueChange = { prompt = it },
modifier = Modifier.fillMaxWidth(),
minLines = 3,
label = { Text("Test prompt") },
)
Spacer(Modifier.height(8.dp))
Row(
modifier = Modifier.fillMaxWidth(),
horizontalArrangement = Arrangement.spacedBy(12.dp),
) {
FilledTonalButton(
onClick = {
val granted = ContextCompat.checkSelfPermission(
context,
Manifest.permission.RECORD_AUDIO,
) == PackageManager.PERMISSION_GRANTED
if (granted) {
scope.launch {
runRealtimeDemo(
voiceClient = voiceClient,
recorder = recorder,
player = player,
prompt = prompt,
mainHandler = mainHandler,
waveform = waveform,
events = events,
onStatus = { status = it },
onSummary = { summary = it },
onRunning = { running = it },
)
}
} else {
permissionLauncher.launch(Manifest.permission.RECORD_AUDIO)
}
},
enabled = !running && voiceClient != null,
modifier = Modifier.weight(1f),
) {
Icon(Icons.Filled.Mic, contentDescription = null)
Spacer(Modifier.size(8.dp))
Text("Mic demo")
}
FilledTonalButton(
onClick = {
scope.launch {
runRealtimeDemo(
voiceClient = voiceClient,
recorder = null,
player = player,
prompt = prompt,
mainHandler = mainHandler,
waveform = waveform,
events = events,
onStatus = { status = it },
onSummary = { summary = it },
onRunning = { running = it },
)
}
},
enabled = !running && voiceClient != null,
modifier = Modifier.weight(1f),
) {
Icon(Icons.Filled.PlayArrow, contentDescription = null)
Spacer(Modifier.size(8.dp))
Text("Text demo")
}
}
}
SectionCard(title = "Waveform") {
RealtimeWaveform(values = waveform.toList())
summary?.let { item ->
HorizontalDivider(modifier = Modifier.padding(vertical = 8.dp))
ProviderRow("First audio", item.firstAudioMs?.let { "${it.toInt()} ms" } ?: "---")
ProviderRow("Done", item.responseDoneMs?.let { "${it.toInt()} ms" } ?: "---")
ProviderRow("Chunks", item.audioChunks.toString())
ProviderRow("Bytes", item.audioBytes.toString())
item.eventLogPath?.let { ProviderRow("Log", it) }
}
}
SectionCard(title = "Events") {
if (events.isEmpty()) {
Text(
text = "No events yet",
style = MaterialTheme.typography.bodySmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
)
} else {
events.takeLast(16).forEach { line ->
Text(
text = line,
style = MaterialTheme.typography.bodySmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
)
}
}
}
}
}
}
private suspend fun runRealtimeDemo(
voiceClient: RelayVoiceClient?,
recorder: RealtimePcmRecorder?,
player: RealtimePcmPlayer,
prompt: String,
mainHandler: Handler,
waveform: MutableList<Float>,
events: MutableList<String>,
onStatus: (String) -> Unit,
onSummary: (RealtimeVoiceSummary?) -> Unit,
onRunning: (Boolean) -> Unit,
) {
val client = voiceClient ?: return
onRunning(true)
onSummary(null)
waveform.clear()
events.clear()
onStatus(if (recorder == null) "Opening websocket" else "Capturing mic")
val pcm = try {
recorder?.capture() ?: ByteArray(320)
} catch (e: Exception) {
onStatus("Mic capture failed: ${e.message}")
onRunning(false)
return
}
onStatus("Streaming")
val result = client.runRealtimeDemo(prompt, pcm) { event ->
handleRealtimeEvent(event, player, mainHandler, waveform, events)
}
if (result.isSuccess) {
onSummary(result.getOrNull())
onStatus("Complete")
} else {
onStatus(result.exceptionOrNull()?.message ?: "Realtime demo failed")
}
onRunning(false)
}
private fun handleRealtimeEvent(
event: RealtimeVoiceEvent,
player: RealtimePcmPlayer,
mainHandler: Handler,
waveform: MutableList<Float>,
events: MutableList<String>,
) {
if (event.type == "voice.audio.delta") {
val audio = event.audioBase64?.let {
runCatching { Base64.getDecoder().decode(it) }.getOrNull()
}
if (audio != null) {
player.write(audio, event.sampleRate ?: 24_000)
}
}
mainHandler.post {
if (event.rmsLevel != null) {
waveform.add(event.rmsLevel)
while (waveform.size > 64) waveform.removeAt(0)
}
val suffix = when {
event.byteCount != null -> " ${event.byteCount}b"
event.message != null -> " ${event.message}"
else -> ""
}
events.add("${event.type}$suffix")
while (events.size > 64) events.removeAt(0)
}
}
@Composable
private fun RealtimeWaveform(values: List<Float>) {
val color = MaterialTheme.colorScheme.primary
Card(
modifier = Modifier
.fillMaxWidth()
.height(96.dp),
shape = RoundedCornerShape(12.dp),
colors = CardDefaults.cardColors(
containerColor = MaterialTheme.colorScheme.surface.copy(alpha = 0.5f),
),
) {
Canvas(
modifier = Modifier
.fillMaxSize()
.padding(12.dp),
) {
val samples = if (values.isEmpty()) List(32) { 0f } else values
val step = size.width / samples.size.coerceAtLeast(1)
samples.forEachIndexed { index, value ->
val clamped = value.coerceIn(0f, 1f)
val x = index * step + step / 2f
val half = (size.height * (0.1f + clamped * 0.85f)) / 2f
drawLine(
color = if (values.isEmpty()) Color.Gray.copy(alpha = 0.25f) else color,
start = Offset(x, size.height / 2f - half),
end = Offset(x, size.height / 2f + half),
strokeWidth = step.coerceAtMost(8f).coerceAtLeast(2f),
cap = StrokeCap.Round,
)
}
}
}
}
@Composable
private fun SectionCard(
title: String,
content: @Composable () -> Unit,
) {
Card(
modifier = Modifier.fillMaxWidth(),
shape = RoundedCornerShape(16.dp),
colors = CardDefaults.cardColors(
containerColor = MaterialTheme.colorScheme.surfaceVariant.copy(alpha = 0.3f),
),
) {
Column(
modifier = Modifier
.fillMaxWidth()
.padding(16.dp),
verticalArrangement = Arrangement.spacedBy(6.dp),
) {
Row(
verticalAlignment = Alignment.CenterVertically,
horizontalArrangement = Arrangement.spacedBy(8.dp),
) {
Icon(
imageVector = Icons.Filled.GraphicEq,
contentDescription = null,
tint = MaterialTheme.colorScheme.primary,
)
Text(
text = title,
style = MaterialTheme.typography.titleMedium,
color = MaterialTheme.colorScheme.primary,
)
}
Spacer(Modifier.height(4.dp))
content()
}
}
}
@Composable
private fun ProviderRow(label: String, value: String) {
Row(
modifier = Modifier.fillMaxWidth(),
horizontalArrangement = Arrangement.SpaceBetween,
) {
Text(
text = label,
style = MaterialTheme.typography.bodyMedium,
color = MaterialTheme.colorScheme.onSurfaceVariant,
modifier = Modifier.weight(0.42f),
)
Text(
text = value,
style = MaterialTheme.typography.bodyMedium,
modifier = Modifier.weight(0.58f),
)
}
}
@@ -57,6 +57,7 @@ import androidx.compose.ui.graphics.vector.ImageVector
import androidx.compose.ui.platform.LocalContext
import androidx.compose.ui.text.style.TextOverflow
import androidx.compose.ui.unit.dp
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.FeatureFlags
import com.hermesandroid.relay.ui.components.AgentInfoSheet
import com.hermesandroid.relay.ui.components.ProfileInspectorCard
@@ -180,14 +181,20 @@ fun SettingsScreen(
// users can change Connection / Profile / Personality without
// having to navigate to Chat first and then hunt for the
// agent-name header.
val effectiveProfile = AgentDisplay.effectiveProfile(
selectedProfile = selectedProfile,
profiles = agentProfiles,
)
ActiveAgentCard(
agentName = agentDisplayName(
agentName = AgentDisplay.agentName(
profile = effectiveProfile,
selectedPersonality = selectedPersonality,
defaultPersonality = defaultPersonality,
connectionLabel = activeConnection?.label,
),
connectionLabel = activeConnection?.label ?: "No connection",
model = selectedProfile?.model ?: "default",
personalityLabel = personalityDisplayLabel(
model = effectiveProfile?.model ?: "default",
personalityLabel = AgentDisplay.personalityLabel(
selectedPersonality = selectedPersonality,
defaultPersonality = defaultPersonality,
),
@@ -449,40 +456,6 @@ private fun ActiveAgentCard(
}
}
/**
* Resolve the agent display name the same way ChatScreen's top bar does:
* when the user has not overridden the personality ("default") and the
* server has advertised a default personality name, use that; otherwise use
* whatever is currently selected. The result is title-cased.
*/
private fun agentDisplayName(
selectedPersonality: String,
defaultPersonality: String,
): String {
val name = if (selectedPersonality == "default" && defaultPersonality.isNotBlank()) {
defaultPersonality
} else {
selectedPersonality
}
return name.replaceFirstChar { it.uppercase() }
}
/**
* Title-cased personality label for the subtitle token. Falls back to the
* server default (when the user hasn't picked one) and finally to a literal
* "Default" so the subtitle never renders with a blank middle token.
*/
private fun personalityDisplayLabel(
selectedPersonality: String,
defaultPersonality: String,
): String = when {
selectedPersonality != "default" ->
selectedPersonality.replaceFirstChar { it.uppercase() }
defaultPersonality.isNotBlank() ->
defaultPersonality.replaceFirstChar { it.uppercase() }
else -> "Default"
}
/**
* One row in the root Settings category list. Matches the visual style of
* the existing Voice navigation row that was previously inline in the
File diff suppressed because it is too large Load Diff
@@ -5,6 +5,7 @@ import android.net.ConnectivityManager
import android.net.NetworkCapabilities
import androidx.lifecycle.ViewModel
import androidx.lifecycle.viewModelScope
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.AppAnalytics
import com.hermesandroid.relay.data.Attachment
import com.hermesandroid.relay.data.AttachmentState
@@ -17,6 +18,7 @@ import com.hermesandroid.relay.data.Profile
import com.hermesandroid.relay.data.ToolCallEvent
import com.hermesandroid.relay.network.HermesApiClient
import com.hermesandroid.relay.network.RelayHttpClient
import com.hermesandroid.relay.network.RealtimeVoiceEvent
import com.hermesandroid.relay.network.handlers.ChatHandler
import com.hermesandroid.relay.network.handlers.LocalDispatchResult
import com.hermesandroid.relay.network.handlers.formatPhoneActionResult
@@ -30,6 +32,7 @@ import com.hermesandroid.relay.util.MediaCacheWriter
import com.hermesandroid.relay.util.PhoneSnapshot
import com.hermesandroid.relay.util.buildPromptBlock
import com.hermesandroid.relay.util.classifyError
import kotlinx.coroutines.Job
import kotlinx.coroutines.channels.BufferOverflow
import kotlinx.coroutines.flow.MutableSharedFlow
import kotlinx.coroutines.flow.MutableStateFlow
@@ -42,6 +45,7 @@ import kotlinx.coroutines.flow.update
import kotlinx.coroutines.launch
import okhttp3.sse.EventSource
import java.util.UUID
import java.util.concurrent.atomic.AtomicInteger
class ChatViewModel : ViewModel() {
@@ -50,6 +54,9 @@ class ChatViewModel : ViewModel() {
private var activeStream: EventSource? = null
private var intentionallyCancelled = false
private var firstTokenNotified = false
private var toolHistoryJob: Job? = null
private var connectionSwitchJob: Job? = null
private val historyLoadGeneration = AtomicInteger(0)
// --- Media dependencies (wired via initializeMedia from RelayApp) ---
private var relayHttpClient: RelayHttpClient? = null
@@ -171,16 +178,15 @@ class ChatViewModel : ViewModel() {
/**
* Streaming endpoint to use for the next chat turn. Always one of
* "sessions" or "runs" — never "auto", since the auto-resolver in
* "sessions", "completions", or "runs" — never "auto", since the auto-resolver in
* ConnectionViewModel.resolveStreamingEndpoint() collapses "auto" to
* a concrete value before this field is written from RelayApp.
*
* Defaults to "runs" so that a fresh ChatViewModel (before RelayApp
* pushes the resolved value) prefers the standard upstream chat path.
* That's the safer fallback than the previous "sessions" default,
* which would 404 on vanilla upstream installs.
* Defaults to "completions" so that a fresh ChatViewModel (before
* RelayApp pushes the resolved value) prefers an EventSource-compatible
* OpenAI chat path instead of assuming `/v1/runs` is an SSE stream.
*/
var streamingEndpoint: String = "runs"
var streamingEndpoint: String = "completions"
/**
* Provider for the active agent-profile pick — wired from [RelayApp] at
@@ -195,6 +201,8 @@ class ChatViewModel : ViewModel() {
* wired the flow) behaves identically to pre-profile-picker installs.
*/
private var selectedProfileProvider: () -> Profile? = { null }
private var effectiveProfileProvider: () -> Profile? = { selectedProfileProvider() }
private var activeProfileContextKey: String? = null
/**
* Wire the agent-profile provider. The provider is typically a lambda
@@ -206,10 +214,17 @@ class ChatViewModel : ViewModel() {
*/
fun setSelectedProfileProvider(provider: () -> Profile?) {
selectedProfileProvider = provider
refreshActiveAgentName()
}
fun setEffectiveProfileProvider(provider: () -> Profile?) {
effectiveProfileProvider = provider
refreshActiveAgentName()
}
fun selectPersonality(name: String) {
_selectedPersonality.value = name
refreshActiveAgentName()
}
/** The display name of the currently active personality (for chat bubbles). */
@@ -274,7 +289,8 @@ class ChatViewModel : ViewModel() {
// messages. Subscribed on every initialize() call so a replaced
// handler (connection switch) picks up fresh events without leaking
// the previous connection's tail.
viewModelScope.launch {
toolHistoryJob?.cancel()
toolHistoryJob = viewModelScope.launch {
chatHandler.messages.collect { msgs ->
val events = msgs
.asSequence()
@@ -360,15 +376,19 @@ class ChatViewModel : ViewModel() {
* runs on [viewModelScope] so it's torn down with the VM.
*/
fun observeConnectionSwitches(events: SharedFlow<String>) {
viewModelScope.launch {
connectionSwitchJob?.cancel()
connectionSwitchJob = viewModelScope.launch {
events.collect { newConnectionId ->
historyLoadGeneration.incrementAndGet()
intentionallyCancelled = true
activeStream?.cancel()
activeStream = null
activeProfileContextKey = null
_queuedMessages.value = emptyList()
_pendingAttachments.value = emptyList()
chatHandler?.let { handler ->
handler.clearMessages()
handler.clearSessions()
handler.setSessionId(null)
}
// Forward the null session id to the persisted
@@ -379,6 +399,60 @@ class ChatViewModel : ViewModel() {
}
}
fun switchProfileContext(contextKey: String, sessionId: String?) {
val client = apiClient
val handler = chatHandler ?: return
handler.activeAgentName = currentAgentDisplayName()
if (
activeProfileContextKey == contextKey &&
handler.currentSessionId.value == sessionId
) {
return
}
if (
activeProfileContextKey == null &&
sessionId != null &&
handler.currentSessionId.value == sessionId
) {
activeProfileContextKey = contextKey
return
}
activeStream?.let { stream ->
intentionallyCancelled = true
stream.cancel()
}
activeStream = null
val loadGeneration = historyLoadGeneration.incrementAndGet()
activeProfileContextKey = contextKey
_queuedMessages.value = emptyList()
_pendingAttachments.value = emptyList()
handler.clearSessions()
handler.setSessionId(sessionId)
handler.clearMessages()
onSessionChanged?.invoke(sessionId)
if (sessionId == null || client == null) {
_isLoadingHistory.value = false
return
}
_isLoadingHistory.value = true
viewModelScope.launch {
val messages = client.getMessages(sessionId)
if (
historyLoadGeneration.get() == loadGeneration &&
activeProfileContextKey == contextKey &&
handler.currentSessionId.value == sessionId
) {
handler.loadMessageHistory(messages)
}
if (historyLoadGeneration.get() == loadGeneration) {
_isLoadingHistory.value = false
}
}
}
private fun fetchPersonalities() {
val client = apiClient ?: return
viewModelScope.launch {
@@ -387,6 +461,7 @@ class ChatViewModel : ViewModel() {
_defaultPersonality.value = config.defaultName
personalityPrompts = config.prompts
_serverModelName.value = config.modelName
refreshActiveAgentName(relabelGenericMessages = true)
}
}
@@ -410,22 +485,38 @@ class ChatViewModel : ViewModel() {
// Cancel any in-flight stream
activeStream?.cancel()
activeStream = null
val loadGeneration = historyLoadGeneration.incrementAndGet()
viewModelScope.launch {
client.createSessionResult().fold(
onSuccess = { session ->
val chatSession = ChatSession(
sessionId = session.id,
title = session.title ?: "New Chat",
model = session.model
)
handler.addSession(chatSession)
handler.setSessionId(session.id)
handler.clearMessages()
onSessionChanged?.invoke(session.id)
AppAnalytics.onSessionCreated()
val selectedProfile = selectedProfileProvider()
val useIsolatedProfileApi = selectedProfile?.hasIsolatedApi == true
client.createSessionResult(
profileName = if (useIsolatedProfileApi) null else selectedProfile?.name,
model = if (useIsolatedProfileApi) {
null
} else {
selectedProfile?.model?.takeIf { it.isNotBlank() }
},
onFailure = { error -> emitError(error, context = "create_session") }
).fold(
onSuccess = { session ->
if (historyLoadGeneration.get() == loadGeneration) {
val chatSession = ChatSession(
sessionId = session.id,
title = session.title ?: "New Chat",
model = session.model
)
handler.addSession(chatSession)
handler.setSessionId(session.id)
handler.clearMessages()
onSessionChanged?.invoke(session.id)
AppAnalytics.onSessionCreated()
}
},
onFailure = { error ->
if (historyLoadGeneration.get() == loadGeneration) {
emitError(error, context = "create_session")
}
}
)
}
}
@@ -438,6 +529,7 @@ class ChatViewModel : ViewModel() {
intentionallyCancelled = true
activeStream?.cancel()
activeStream = null
val loadGeneration = historyLoadGeneration.incrementAndGet()
handler.setSessionId(sessionId)
handler.clearMessages()
@@ -448,8 +540,15 @@ class ChatViewModel : ViewModel() {
_isLoadingHistory.value = true
viewModelScope.launch {
val messages = client.getMessages(sessionId)
handler.loadMessageHistory(messages)
_isLoadingHistory.value = false
if (
historyLoadGeneration.get() == loadGeneration &&
handler.currentSessionId.value == sessionId
) {
handler.loadMessageHistory(messages)
}
if (historyLoadGeneration.get() == loadGeneration) {
_isLoadingHistory.value = false
}
}
}
@@ -641,13 +740,22 @@ class ChatViewModel : ViewModel() {
val assistantMessageId = UUID.randomUUID().toString()
val sessionId = handler.currentSessionId.value
if (streamingEndpoint == "runs") {
if (streamingEndpoint == "runs" || streamingEndpoint == "completions") {
startStream(client, handler, sessionId ?: "", text.trim(), assistantMessageId, attachments)
} else if (sessionId != null) {
startStream(client, handler, sessionId, text.trim(), assistantMessageId, attachments)
} else {
viewModelScope.launch {
client.createSessionResult().fold(
val selectedProfile = selectedProfileProvider()
val useIsolatedProfileApi = selectedProfile?.hasIsolatedApi == true
client.createSessionResult(
profileName = if (useIsolatedProfileApi) null else selectedProfile?.name,
model = if (useIsolatedProfileApi) {
null
} else {
selectedProfile?.model?.takeIf { it.isNotBlank() }
},
).fold(
onSuccess = { session ->
val chatSession = ChatSession(
sessionId = session.id,
@@ -677,8 +785,95 @@ class ChatViewModel : ViewModel() {
}
}
fun startRealtimeAgentTurn(userText: String, chatSessionId: String?): String {
val handler = chatHandler ?: return UUID.randomUUID().toString()
AppAnalytics.onMessageSent()
val trimmed = userText.trim()
val userMessageId = UUID.randomUUID().toString()
val assistantMessageId = "realtime-agent-${UUID.randomUUID()}"
handler.activeAgentName = currentAgentDisplayName()
handler.addUserMessage(
ChatMessage(
id = userMessageId,
role = MessageRole.USER,
content = trimmed,
timestamp = System.currentTimeMillis(),
)
)
handler.setLastSentMessage(trimmed)
chatSessionId?.takeIf { it.isNotBlank() }?.let {
handler.setSessionId(it)
onSessionChanged?.invoke(it)
}
handler.addPlaceholderMessage(
ChatMessage(
id = assistantMessageId,
role = MessageRole.ASSISTANT,
content = "",
timestamp = System.currentTimeMillis(),
isStreaming = true,
agentName = handler.activeAgentName,
)
)
return assistantMessageId
}
fun applyRealtimeAgentEvent(assistantMessageId: String, event: RealtimeVoiceEvent) {
val handler = chatHandler ?: return
val hermesSessionId = when {
!event.chatSessionId.isNullOrBlank() -> event.chatSessionId
event.type.startsWith("hermes.") && !event.sessionId.isNullOrBlank() -> event.sessionId
else -> null
}
hermesSessionId?.let {
handler.setSessionId(it)
onSessionChanged?.invoke(it)
}
when (event.type) {
"hermes.message.started" -> Unit
"voice.response.delta" -> {
val delta = event.delta ?: return
handler.onTextDelta(assistantMessageId, delta)
}
"hermes.tool.delta" -> {
val delta = event.delta ?: return
handler.onThinkingDelta(assistantMessageId, delta)
}
"hermes.tool.started" -> {
val name = event.toolName?.takeIf { it.isNotBlank() } ?: "hermes"
val callId = event.toolCallId?.takeIf { it.isNotBlank() } ?: name
handler.onToolCallStart(assistantMessageId, callId, name)
}
"hermes.tool.completed" -> {
val callId = event.toolCallId?.takeIf { it.isNotBlank() }
?: event.toolName?.takeIf { it.isNotBlank() }
?: "hermes"
handler.onToolCallComplete(assistantMessageId, callId, event.resultPreview)
}
"hermes.tool.failed" -> {
val callId = event.toolCallId?.takeIf { it.isNotBlank() }
?: event.toolName?.takeIf { it.isNotBlank() }
?: "hermes"
handler.onToolCallFailed(assistantMessageId, callId, event.message ?: event.resultPreview)
}
"hermes.confirmation.requested" -> {
val prompt = event.message ?: "Waiting for confirmation"
handler.onThinkingDelta(assistantMessageId, prompt)
}
"voice.response.done", "hermes.run.completed" -> {
handler.onStreamComplete(assistantMessageId)
activeStream = null
}
"voice.error" -> {
handler.onStreamError(event.message ?: "Realtime agent failed")
activeStream = null
}
}
}
/**
* Kick off an SSE chat turn against either the runs or sessions endpoint.
* Kick off an SSE chat turn against the selected chat endpoint.
*
* **System-message precedence (Pass 3, 2026-04-18):**
*
@@ -687,7 +882,9 @@ class ChatViewModel : ViewModel() {
* personality: it bundles model + persona (from the profile's
* `SOUL.md`) into a single named unit, so a user who picks a profile
* has explicitly asked for that profile's full identity. The
* personality prompt is skipped in this case.
* personality prompt is skipped in this case. If the selected profile
* advertises an isolated API route, the server's profile config owns
* SOUL/default prompt and the phone does not resend it.
* 2. **Selected non-default personality** — send the personality's
* stored system prompt. This is the pre-Pass-3 path.
* 3. **Neither selected** — no personality prompt, server uses its own
@@ -708,12 +905,16 @@ class ChatViewModel : ViewModel() {
// Resolve the active profile pick once — used below for both
// modelOverride and the system_message precedence rule.
val selectedProfile = selectedProfileProvider()
val effectiveProfile = effectiveProfileProvider() ?: selectedProfile
val useIsolatedProfileApi = selectedProfile?.hasIsolatedApi == true
// Build persona prompt following the precedence rule documented on
// this function's KDoc. A profile's systemMessage wins over a
// selected personality when both are set.
val selected = _selectedPersonality.value
val profileSystemMessage = selectedProfile?.systemMessage?.takeIf { it.isNotBlank() }
val profileSystemMessage = selectedProfile
?.systemMessage
?.takeIf { !useIsolatedProfileApi && it.isNotBlank() }
val personaPrompt: String? = when {
profileSystemMessage != null -> profileSystemMessage
selected != "default" && selected != _defaultPersonality.value ->
@@ -729,9 +930,10 @@ class ChatViewModel : ViewModel() {
.ifBlank { null }
// === END PHASE3-status ===
// Set agent name for display on chat bubbles
handler.activeAgentName = activePersonalityName.replaceFirstChar { it.uppercase() }
.ifBlank { null }
// Set agent name for display on chat bubbles. The selected/effective
// Hermes profile is the active agent identity; personality is only a
// fallback when no profile metadata is available.
handler.activeAgentName = currentAgentDisplayName(effectiveProfile)
firstTokenNotified = false
var lastInputTokens: Int? = null
@@ -889,52 +1091,81 @@ class ChatViewModel : ViewModel() {
// [selectedProfile] resolved at the top of this function so a
// rapid switch doesn't give us a systemMessage from profile A but
// a model from profile B.
val modelOverride: String? = selectedProfile
?.model
?.takeIf { it.isNotBlank() }
activeStream = if (streamingEndpoint == "runs") {
client.sendRunStream(
message = message,
systemMessage = systemMsg,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
onSessionId = { sid ->
handler.setSessionId(sid)
onSessionChanged?.invoke(sid)
},
onMessageStarted = onMessageStartedCb,
onTextDelta = onTextDeltaCb,
onThinkingDelta = onThinkingDeltaCb,
onToolCallStart = onToolCallStartCb,
onToolCallDone = onToolCallDoneCb,
onToolCallFailed = onToolCallFailedCb,
onTurnComplete = onTurnCompleteCb,
onComplete = onCompleteCb,
onUsage = onUsageCb,
onError = onErrorCb,
modelOverride = modelOverride,
)
val modelOverride: String? = if (useIsolatedProfileApi) {
null
} else {
client.sendChatStream(
sessionId = sessionId,
message = message,
systemMessage = systemMsg,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
onSessionId = { /* already set */ },
onMessageStarted = onMessageStartedCb,
onTextDelta = onTextDeltaCb,
onThinkingDelta = onThinkingDeltaCb,
onToolCallStart = onToolCallStartCb,
onToolCallDone = onToolCallDoneCb,
onToolCallFailed = onToolCallFailedCb,
onTurnComplete = onTurnCompleteCb,
onComplete = onCompleteCb,
onUsage = onUsageCb,
onError = onErrorCb,
modelOverride = modelOverride,
)
selectedProfile
?.model
?.takeIf { it.isNotBlank() }
}
val profileName: String? = if (useIsolatedProfileApi) {
null
} else {
AgentDisplay.profileRequestName(selectedProfile?.name)
}
activeStream = when (streamingEndpoint) {
"runs" -> client.sendRunStream(
message = message,
systemMessage = systemMsg,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
onSessionId = { sid ->
handler.setSessionId(sid)
onSessionChanged?.invoke(sid)
},
onMessageStarted = onMessageStartedCb,
onTextDelta = onTextDeltaCb,
onThinkingDelta = onThinkingDeltaCb,
onToolCallStart = onToolCallStartCb,
onToolCallDone = onToolCallDoneCb,
onToolCallFailed = onToolCallFailedCb,
onTurnComplete = onTurnCompleteCb,
onComplete = onCompleteCb,
onUsage = onUsageCb,
onError = onErrorCb,
modelOverride = modelOverride,
profileName = profileName,
)
"completions" -> client.sendChatCompletionsStream(
message = message,
systemMessage = systemMsg,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
onSessionId = { /* stateless OpenAI-compatible endpoint */ },
onMessageStarted = onMessageStartedCb,
onTextDelta = onTextDeltaCb,
onThinkingDelta = onThinkingDeltaCb,
onToolCallStart = onToolCallStartCb,
onToolCallDone = onToolCallDoneCb,
onToolCallFailed = onToolCallFailedCb,
onTurnComplete = onTurnCompleteCb,
onComplete = onCompleteCb,
onUsage = onUsageCb,
onError = onErrorCb,
modelOverride = modelOverride,
profileName = profileName,
)
else -> client.sendChatStream(
sessionId = sessionId,
message = message,
systemMessage = systemMsg,
attachments = attachments,
voiceIntentMessages = voiceIntentMessages,
onSessionId = { /* already set */ },
onMessageStarted = onMessageStartedCb,
onTextDelta = onTextDeltaCb,
onThinkingDelta = onThinkingDeltaCb,
onToolCallStart = onToolCallStartCb,
onToolCallDone = onToolCallDoneCb,
onToolCallFailed = onToolCallFailedCb,
onTurnComplete = onTurnCompleteCb,
onComplete = onCompleteCb,
onUsage = onUsageCb,
onError = onErrorCb,
modelOverride = modelOverride,
profileName = profileName,
)
}
// Flip syncedToServer=true on every voice-intent trace AND every
@@ -953,6 +1184,33 @@ class ChatViewModel : ViewModel() {
}
}
private fun currentAgentDisplayName(
effectiveProfileOverride: Profile? = null,
): String? {
val selectedProfile = selectedProfileProvider()
val effectiveProfile = effectiveProfileOverride
?: effectiveProfileProvider()
?: selectedProfile
return AgentDisplay.agentName(
profile = effectiveProfile,
selectedPersonality = _selectedPersonality.value,
defaultPersonality = _defaultPersonality.value,
connectionLabel = null,
).ifBlank { null }
}
private fun refreshActiveAgentName(
effectiveProfileOverride: Profile? = null,
relabelGenericMessages: Boolean = false,
) {
val handler = chatHandler ?: return
val displayName = currentAgentDisplayName(effectiveProfileOverride)
handler.activeAgentName = displayName
if (relabelGenericMessages) {
handler.relabelGenericAssistantMessages(displayName)
}
}
fun cancelStream() {
intentionallyCancelled = true
activeStream?.cancel()
@@ -19,7 +19,9 @@ import com.hermesandroid.relay.data.PairingPreferences
import com.hermesandroid.relay.data.Connection
import com.hermesandroid.relay.data.ConnectionStore
import com.hermesandroid.relay.data.ConnectionValidation
import com.hermesandroid.relay.data.AgentDisplay
import com.hermesandroid.relay.data.Profile
import com.hermesandroid.relay.data.ProfileSessionStore
import com.hermesandroid.relay.data.ProfileSelectionStore
import com.hermesandroid.relay.data.relayDataStore
import com.hermesandroid.relay.util.TailscaleDetector
@@ -30,6 +32,7 @@ import com.hermesandroid.relay.network.ConnectionManager
import com.hermesandroid.relay.network.ConnectionState
import com.hermesandroid.relay.network.EndpointResolver
import com.hermesandroid.relay.network.HermesApiClient
import com.hermesandroid.relay.network.ProfileApiUrlResolver
import com.hermesandroid.relay.network.ServerCapabilities
import com.hermesandroid.relay.network.RelayHttpClient
import com.hermesandroid.relay.network.RelayUrlDeriver
@@ -388,6 +391,12 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
private val _apiClient = MutableStateFlow<HermesApiClient?>(null)
val apiClient: StateFlow<HermesApiClient?> = _apiClient.asStateFlow()
private val _chatApiClient = MutableStateFlow<HermesApiClient?>(null)
val chatApiClient: StateFlow<HermesApiClient?> = _chatApiClient.asStateFlow()
private var profileChatApiClient: HermesApiClient? = null
private var profileChatApiClientUrl: String? = null
private var profileChatApiClientKey: String? = null
private val _chatMode = MutableStateFlow(ChatMode.DISCONNECTED)
val chatMode: StateFlow<ChatMode> = _chatMode.asStateFlow()
@@ -398,8 +407,8 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
private val _serverCapabilities = MutableStateFlow(ServerCapabilities.DISCONNECTED)
val serverCapabilities: StateFlow<ServerCapabilities> = _serverCapabilities.asStateFlow()
// Chat is ready when API client exists and server is reachable
val chatReady: StateFlow<Boolean> = combine(_apiClient, _apiServerReachable) { client, reachable ->
// Chat is ready when a chat-routed API client exists and the base server is reachable.
val chatReady: StateFlow<Boolean> = combine(_chatApiClient, _apiServerReachable) { client, reachable ->
client != null && reachable
}.stateIn(viewModelScope, SharingStarted.Eagerly, false)
// NOTE: [relayReady] / [voiceReady] are declared below the [_relayUrl]
@@ -573,6 +582,8 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
*/
private val profileSelectionStore: ProfileSelectionStore =
ProfileSelectionStore(application)
private val profileSessionStore: ProfileSessionStore =
ProfileSessionStore(application)
/**
* Set (or clear, with `null`) the active profile pick. Writes through
@@ -580,36 +591,79 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
* the selection survives process death and connection switches.
*/
fun selectProfile(profile: Profile?) {
_selectedProfile.value = profile
val normalizedProfile = AgentDisplay.normalizeSelection(profile)
_selectedProfile.value = normalizedProfile
_lastSessionId.value = null
val connectionId = activeConnectionId.value ?: return
_pendingSelectedProfileConnectionId.value = connectionId
_pendingSelectedProfileName.value = profile?.name
_pendingSelectedProfileName.value = normalizedProfile?.name
viewModelScope.launch {
profileSelectionStore.setSelectedProfile(connectionId, profile?.name)
profileSelectionStore.setSelectedProfile(connectionId, normalizedProfile?.name)
rebuildChatApiClient()
}
refreshLastSessionForProfile(connectionId, normalizedProfile?.name)
}
private fun resolvePendingProfileFrom(list: List<Profile>) {
val connectionId = activeConnectionId.value ?: return
private fun resolvePendingProfileFrom(list: List<Profile>): Boolean {
val connectionId = activeConnectionId.value ?: return false
if (_pendingSelectedProfileConnectionId.value != connectionId) {
return
return false
}
val current = _selectedProfile.value
if (current != null) {
if (AgentDisplay.isServerDefaultAlias(current.name)) {
_selectedProfile.value = null
_pendingSelectedProfileName.value = null
return true
}
val refreshed = list.firstOrNull { it.name == current.name }
if (refreshed != null) {
if (refreshed != current) {
_selectedProfile.value = refreshed
return true
}
return
return false
}
_selectedProfile.value = null
_pendingSelectedProfileName.value = current.name
return true
}
val pendingName = _pendingSelectedProfileName.value ?: return false
if (AgentDisplay.isServerDefaultAlias(pendingName)) {
_pendingSelectedProfileName.value = null
_selectedProfile.value = null
return true
}
val pendingName = _pendingSelectedProfileName.value ?: return
val resolved = list.firstOrNull { it.name == pendingName }
if (resolved != null) {
_selectedProfile.value = resolved
return true
}
return false
}
private fun refreshLastSessionForProfile(
connectionId: String?,
profileName: String?,
) {
_lastSessionId.value = null
if (connectionId == null) return
viewModelScope.launch {
val profileScoped = profileSessionStore
.sessionIdFlow(connectionId, profileName)
.first()
val legacyDefault = if (profileName == null) {
getApplication<Application>().relayDataStore.data
.first()[KEY_LAST_SESSION_ID]
} else {
null
}
if (
activeConnectionId.value == connectionId &&
_selectedProfile.value?.name == profileName
) {
_lastSessionId.value = profileScoped ?: legacyDefault
}
}
}
@@ -748,13 +802,15 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
}
// === END PHASE3-status ===
// Streaming endpoint preference. Three values:
// Streaming endpoint preference. Four values:
// "auto" — pick based on per-endpoint capability detection (default
// for new installs as of v0.3.0). Resolves to "sessions"
// when the server has /api/sessions/{id}/chat/stream
// (fork or upstream-merged), otherwise "runs".
// (fork or upstream-merged), then "completions" when
// /v1/chat/completions is available.
// "sessions" — force /api/sessions/{id}/chat/stream
// "runs" — force /v1/runs
// "completions" — force /v1/chat/completions with stream=true
// "runs" — force /v1/runs for servers known to stream that route
//
// Existing users keep whatever they previously chose. Only fresh installs
// (no value persisted yet) get the new "auto" default.
@@ -772,14 +828,14 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
/**
* Resolve the user's `streamingEndpoint` preference to a concrete value
* based on the latest capability probe. Returns "sessions" or "runs"
* based on the latest capability probe. Returns a concrete endpoint
* (never "auto"). Used by ChatViewModel right before kicking off a stream.
*
* - "sessions" / "runs" pass through unchanged (manual override wins).
* - "sessions" / "completions" / "runs" pass through unchanged (manual override wins).
* - "auto" → reads `serverCapabilities.value.preferredChatEndpoint()`.
*/
fun resolveStreamingEndpoint(preference: String): String = when (preference) {
"sessions", "runs" -> preference
"sessions", "completions", "runs" -> preference
else -> _serverCapabilities.value.preferredChatEndpoint()
}
@@ -1393,6 +1449,7 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
// ProfileSelectionStore is a separate DataStore file from
// ConnectionStore's EncryptedSharedPrefs.
profileSelectionStore.clear(connectionId)
profileSessionStore.clearConnection(connectionId)
}
private suspend fun readStoredDeviceIdForRemoval(
@@ -1700,6 +1757,7 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
)
connectionStore.removeConnection(duplicate.id)
profileSelectionStore.clear(duplicate.id)
profileSessionStore.clearConnection(duplicate.id)
}
// Auto-rename the placeholder label created by
@@ -1869,11 +1927,10 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
_relayUrl.value = savedRelayUrl
}
// Load last session ID
val savedSessionId = preferences[KEY_LAST_SESSION_ID]
if (savedSessionId != null) {
_lastSessionId.value = savedSessionId
}
// Last-session restore is profile-scoped now. Keep that
// state owned by refreshLastSessionForProfile(); otherwise
// any unrelated relayDataStore emission can overwrite an
// active profile's session with the legacy default id.
// Check if this is a new version → show What's New
val currentVersion = getAppVersionName()
@@ -1994,11 +2051,14 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
viewModelScope.launch {
activeConnectionId.collect { connectionId ->
_selectedProfile.value = null
_lastSessionId.value = null
_pendingSelectedProfileConnectionId.value = connectionId
_pendingSelectedProfileName.value = connectionId?.let { cid ->
profileSelectionStore.selectedProfileFlow(cid).first()
}
resolvePendingProfileFrom(agentProfiles.value)
refreshLastSessionForProfile(connectionId, _selectedProfile.value?.name)
rebuildChatApiClient()
}
}
@@ -2007,7 +2067,13 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
// remains pending so it can recover if the server advertises it later.
viewModelScope.launch {
agentProfiles.collect { list ->
resolvePendingProfileFrom(list)
if (resolvePendingProfileFrom(list)) {
refreshLastSessionForProfile(
activeConnectionId.value,
_selectedProfile.value?.name,
)
rebuildChatApiClient()
}
}
}
}
@@ -2487,6 +2553,44 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
_chatMode.value = ChatMode.DISCONNECTED
_serverCapabilities.value = ServerCapabilities.DISCONNECTED
}
rebuildChatApiClient()
}
private suspend fun rebuildChatApiClient() {
val baseApiUrl = ProfileApiUrlResolver.normalize(effectiveApiServerUrlSnapshot())
val profileApiUrl = ProfileApiUrlResolver.resolveForConnection(
profileApiUrl = _selectedProfile.value?.apiServerUrl,
baseApiUrl = baseApiUrl,
)
val baseClient = _apiClient.value
val key = authManager.getApiKey() ?: ""
if (profileApiUrl == null || profileApiUrl == baseApiUrl) {
val oldProfileClient = profileChatApiClient
profileChatApiClient = null
profileChatApiClientUrl = null
profileChatApiClientKey = null
_chatApiClient.value = baseClient
shutdownClientOffMain(oldProfileClient)
return
}
val existingProfileClient = profileChatApiClient
if (
existingProfileClient != null &&
profileChatApiClientUrl == profileApiUrl &&
profileChatApiClientKey == key
) {
_chatApiClient.value = existingProfileClient
return
}
val nextProfileClient = HermesApiClient(baseUrl = profileApiUrl, apiKey = key)
profileChatApiClient = nextProfileClient
profileChatApiClientUrl = profileApiUrl
profileChatApiClientKey = key
_chatApiClient.value = nextProfileClient
shutdownClientOffMain(existingProfileClient)
}
/**
@@ -2790,12 +2894,19 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
fun saveLastSessionId(sessionId: String?) {
_lastSessionId.value = sessionId
val connectionId = activeConnectionId.value
val profileName = _selectedProfile.value?.name
viewModelScope.launch {
getApplication<Application>().relayDataStore.edit { preferences ->
if (sessionId != null) {
preferences[KEY_LAST_SESSION_ID] = sessionId
} else {
preferences.remove(KEY_LAST_SESSION_ID)
if (connectionId != null) {
profileSessionStore.setSessionId(connectionId, profileName, sessionId)
}
if (profileName == null) {
getApplication<Application>().relayDataStore.edit { preferences ->
if (sessionId != null) {
preferences[KEY_LAST_SESSION_ID] = sessionId
} else {
preferences.remove(KEY_LAST_SESSION_ID)
}
}
}
}
@@ -2850,7 +2961,12 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
_apiServerUrl.value = DEFAULT_API_URL
_relayUrl.value = DEFAULT_RELAY_URL
shutdownClientOffMain(_apiClient.value)
shutdownClientOffMain(profileChatApiClient)
_apiClient.value = null
_chatApiClient.value = null
profileChatApiClient = null
profileChatApiClientUrl = null
profileChatApiClientKey = null
_apiServerReachable.value = false
}
}
@@ -3119,6 +3235,12 @@ class ConnectionViewModel(application: Application) : AndroidViewModel(applicati
_apiClient.value?.let { client ->
Thread({ runCatching { client.shutdown() } }, "HermesApiClient-shutdown").start()
}
profileChatApiClient?.let { client ->
Thread(
{ runCatching { client.shutdown() } },
"HermesProfileApiClient-shutdown",
).start()
}
tailscaleDetector.shutdown()
// Release the cached VirtualDisplay + ImageReader + HandlerThread
// built by ScreenCapture on the first /screenshot call. Without
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,734 @@
package com.hermesandroid.relay.voice
import android.annotation.SuppressLint
import android.content.Context
import android.content.Intent
import android.graphics.PixelFormat
import android.os.Build
import android.provider.Settings
import android.util.Log
import android.view.Gravity
import android.view.View
import android.view.WindowManager
import androidx.compose.animation.animateColorAsState
import androidx.compose.animation.core.tween
import androidx.compose.foundation.background
import androidx.compose.foundation.clickable
import androidx.compose.foundation.combinedClickable
import androidx.compose.foundation.ExperimentalFoundationApi
import androidx.compose.foundation.Canvas
import androidx.compose.foundation.gestures.detectDragGestures
import androidx.compose.foundation.layout.Arrangement
import androidx.compose.foundation.layout.Box
import androidx.compose.foundation.layout.Column
import androidx.compose.foundation.layout.Row
import androidx.compose.foundation.layout.Spacer
import androidx.compose.foundation.layout.fillMaxSize
import androidx.compose.foundation.layout.fillMaxWidth
import androidx.compose.foundation.layout.height
import androidx.compose.foundation.layout.padding
import androidx.compose.foundation.layout.size
import androidx.compose.foundation.layout.width
import androidx.compose.foundation.shape.CircleShape
import androidx.compose.foundation.shape.RoundedCornerShape
import androidx.compose.material.icons.Icons
import androidx.compose.material.icons.filled.GraphicEq
import androidx.compose.material.icons.filled.Mic
import androidx.compose.material.icons.filled.Stop
import androidx.compose.material3.Icon
import androidx.compose.material3.LinearProgressIndicator
import androidx.compose.material3.MaterialTheme
import androidx.compose.material3.Surface
import androidx.compose.material3.Text
import androidx.compose.material3.TextButton
import androidx.compose.runtime.Composable
import androidx.compose.runtime.LaunchedEffect
import androidx.compose.runtime.collectAsState
import androidx.compose.runtime.getValue
import androidx.compose.runtime.mutableStateOf
import androidx.compose.runtime.remember
import androidx.compose.runtime.rememberUpdatedState
import androidx.compose.runtime.setValue
import androidx.compose.runtime.withFrameNanos
import androidx.compose.ui.Alignment
import androidx.compose.ui.Modifier
import androidx.compose.ui.draw.clip
import androidx.compose.ui.graphics.Color
import androidx.compose.ui.graphics.StrokeCap
import androidx.compose.ui.graphics.lerp
import androidx.compose.ui.input.pointer.pointerInput
import androidx.compose.ui.platform.ComposeView
import androidx.compose.ui.platform.ViewCompositionStrategy
import androidx.compose.ui.text.font.FontWeight
import androidx.compose.ui.text.style.TextOverflow
import androidx.compose.ui.unit.dp
import androidx.lifecycle.Lifecycle
import androidx.lifecycle.LifecycleOwner
import androidx.lifecycle.LifecycleRegistry
import androidx.lifecycle.ViewModelStore
import androidx.lifecycle.ViewModelStoreOwner
import androidx.lifecycle.setViewTreeLifecycleOwner
import androidx.lifecycle.setViewTreeViewModelStoreOwner
import androidx.savedstate.SavedStateRegistry
import androidx.savedstate.SavedStateRegistryController
import androidx.savedstate.SavedStateRegistryOwner
import androidx.savedstate.setViewTreeSavedStateRegistryOwner
import com.hermesandroid.relay.ui.theme.HermesRelayTheme
import com.hermesandroid.relay.util.ComposeArrWorkaround
import com.hermesandroid.relay.viewmodel.InteractionMode
import com.hermesandroid.relay.viewmodel.VoiceState
import com.hermesandroid.relay.viewmodel.VoiceUiState
import kotlinx.coroutines.flow.MutableStateFlow
import kotlinx.coroutines.flow.StateFlow
import kotlin.math.cos
import kotlin.math.max
import kotlin.math.min
import kotlin.math.roundToInt
import kotlin.math.sin
private const val TWO_PI = 6.2831855f
private const val HALF_PI = 1.5707964f
// Matches the in-app VoiceWaveform palette so minimized overlay mode reads as
// the same voice surface, just wrapped around the mic control.
private val OverlayListeningPrimary = Color(0xFF597EF2)
private val OverlayListeningSecondary = Color(0xFFA573F2)
private val OverlaySpeakingPrimary = Color(0xFF40EB8C)
private val OverlaySpeakingSecondary = Color(0xFF4DD9E0)
/**
* WindowManager-backed host for the voice overlay.
*
* This deliberately mirrors BridgeStatusOverlay's permission and lifecycle
* pattern, but remains voice-owned so the Bridge safety chip and confirmation
* modal are not coupled to realtime voice mode.
*/
class VoiceOverlayHost(context: Context) {
companion object {
private const val TAG = "VoiceOverlayHost"
@Volatile
private var INSTANCE: VoiceOverlayHost? = null
fun install(context: Context): VoiceOverlayHost {
val existing = INSTANCE
if (existing != null) return existing
val created = VoiceOverlayHost(context.applicationContext)
INSTANCE = created
return created
}
fun peek(): VoiceOverlayHost? = INSTANCE
}
private val appContext: Context = context.applicationContext
private val wm: WindowManager =
appContext.getSystemService(Context.WINDOW_SERVICE) as WindowManager
private val sessionState = MutableStateFlow<VoiceOverlaySession?>(null)
private var overlayView: View? = null
private var overlayOwner: VoiceOverlayLifecycleOwner? = null
private var overlayParams: WindowManager.LayoutParams? = null
fun hasOverlayPermission(): Boolean = Settings.canDrawOverlays(appContext)
@SuppressLint("InflateParams")
fun show(session: VoiceOverlaySession): Boolean {
sessionState.value = session
if (!hasOverlayPermission()) {
Log.w(TAG, "show: SYSTEM_ALERT_WINDOW not granted")
return false
}
if (overlayView != null) return true
val compose = ComposeView(appContext).apply {
setViewCompositionStrategy(ViewCompositionStrategy.DisposeOnDetachedFromWindow)
setContent {
HermesRelayTheme {
val activeSession by sessionState.collectAsState()
activeSession?.let {
VoiceFloatingOverlayPill(
session = it,
onDragBy = { dx, dy -> moveBy(dx, dy) },
)
}
}
}
}
attachLifecycle(compose)
val params = WindowManager.LayoutParams(
WindowManager.LayoutParams.WRAP_CONTENT,
WindowManager.LayoutParams.WRAP_CONTENT,
overlayType(),
WindowManager.LayoutParams.FLAG_NOT_FOCUSABLE or
WindowManager.LayoutParams.FLAG_NOT_TOUCH_MODAL or
WindowManager.LayoutParams.FLAG_LAYOUT_IN_SCREEN or
WindowManager.LayoutParams.FLAG_LAYOUT_NO_LIMITS,
PixelFormat.TRANSLUCENT,
).apply {
gravity = Gravity.TOP or Gravity.START
x = 24
y = 96
}
val added = runCatching { wm.addView(compose, params) }
.onFailure { Log.w(TAG, "addView(voice overlay) failed", it) }
.isSuccess
if (!added) {
sessionState.value = null
overlayOwner?.stop()
overlayOwner = null
return false
}
compose.post { ComposeArrWorkaround.disableForViewTree(compose) }
overlayView = compose
overlayParams = params
return true
}
fun hide() {
val view = overlayView
overlayView = null
overlayParams = null
sessionState.value = null
if (view != null) {
runCatching { wm.removeView(view) }
.onFailure { Log.w(TAG, "removeView(voice overlay) failed", it) }
}
overlayOwner?.stop()
overlayOwner = null
}
private fun moveBy(dx: Float, dy: Float) {
val view = overlayView ?: return
val params = overlayParams ?: return
params.x = (params.x + dx.roundToInt()).coerceAtLeast(0)
params.y = (params.y + dy.roundToInt()).coerceAtLeast(0)
runCatching { wm.updateViewLayout(view, params) }
.onFailure { Log.w(TAG, "updateViewLayout(voice overlay) failed", it) }
}
private fun overlayType(): Int =
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.O) {
WindowManager.LayoutParams.TYPE_APPLICATION_OVERLAY
} else {
@Suppress("DEPRECATION")
WindowManager.LayoutParams.TYPE_PHONE
}
private fun attachLifecycle(view: View) {
val owner = VoiceOverlayLifecycleOwner().also { it.start() }
overlayOwner = owner
view.setViewTreeLifecycleOwner(owner)
view.setViewTreeViewModelStoreOwner(owner)
view.setViewTreeSavedStateRegistryOwner(owner)
}
}
data class VoiceOverlaySession(
val uiState: StateFlow<VoiceUiState>,
val provider: String?,
val model: String?,
val voice: String?,
val profileName: String?,
val configScope: String?,
val outputEnabled: Boolean?,
val fallbackEnabled: Boolean?,
val onStartListening: () -> Unit,
val onStopListening: () -> Unit,
val onInterrupt: () -> Unit,
val onPauseAutoMode: () -> Unit,
val onReturnToHermes: () -> Unit,
val onDismissOverlay: () -> Unit,
val onExit: () -> Unit,
)
@Composable
private fun VoiceFloatingOverlayPill(
session: VoiceOverlaySession,
onDragBy: (Float, Float) -> Unit,
) {
val uiState by session.uiState.collectAsState()
var minimized by remember { mutableStateOf(false) }
val providerText = voiceProviderLabel(
session.provider,
session.model,
session.voice,
session.outputEnabled,
)
val profileText = session.profileName?.takeIf { it.isNotBlank() } ?: "default profile"
val stateText = when (uiState.state) {
VoiceState.Idle -> "Ready"
VoiceState.Listening -> "Listening"
VoiceState.Transcribing -> "Transcribing"
VoiceState.Thinking -> "Thinking"
VoiceState.Speaking -> "Speaking"
VoiceState.Error -> "Error"
}
if (minimized) {
VoiceFloatingOverlayBubble(
uiState = uiState,
stateText = stateText,
onExpand = { minimized = false },
onStartListening = session.onStartListening,
onStopListening = session.onStopListening,
onInterrupt = session.onInterrupt,
onPauseAutoMode = session.onPauseAutoMode,
onDragBy = onDragBy,
)
return
}
Surface(
modifier = Modifier
.width(332.dp)
.pointerInput(Unit) {
detectDragGestures { change, dragAmount ->
change.consume()
onDragBy(dragAmount.x, dragAmount.y)
}
},
shape = RoundedCornerShape(24.dp),
color = MaterialTheme.colorScheme.surface.copy(alpha = 0.97f),
tonalElevation = 6.dp,
shadowElevation = 8.dp,
) {
Column(
modifier = Modifier.padding(horizontal = 12.dp, vertical = 10.dp),
verticalArrangement = Arrangement.spacedBy(8.dp),
) {
Row(
modifier = Modifier.fillMaxWidth(),
verticalAlignment = Alignment.CenterVertically,
horizontalArrangement = Arrangement.spacedBy(8.dp),
) {
Box(
modifier = Modifier
.size(34.dp)
.clip(CircleShape)
.background(MaterialTheme.colorScheme.primaryContainer),
contentAlignment = Alignment.Center,
) {
Icon(
imageVector = Icons.Filled.GraphicEq,
contentDescription = null,
tint = MaterialTheme.colorScheme.onPrimaryContainer,
modifier = Modifier.size(18.dp),
)
}
Column(modifier = Modifier.weight(1f)) {
Row(horizontalArrangement = Arrangement.spacedBy(6.dp)) {
Text(
text = "Voice Overlay",
style = MaterialTheme.typography.labelLarge,
color = MaterialTheme.colorScheme.onSurface,
fontWeight = FontWeight.SemiBold,
)
}
Text(
text = "$stateText · $profileText · $providerText",
style = MaterialTheme.typography.labelSmall,
color = MaterialTheme.colorScheme.onSurfaceVariant,
maxLines = 1,
overflow = TextOverflow.Ellipsis,
)
}
MicControlButton(
uiState = uiState,
onStartListening = session.onStartListening,
onStopListening = session.onStopListening,
onInterrupt = session.onInterrupt,
onPauseAutoMode = session.onPauseAutoMode,
)
}
if (uiState.state == VoiceState.Transcribing || uiState.state == VoiceState.Thinking) {
LinearProgressIndicator(modifier = Modifier.fillMaxWidth())
}
Row(
modifier = Modifier.fillMaxWidth(),
horizontalArrangement = Arrangement.spacedBy(6.dp),
) {
TextButton(
onClick = { minimized = true },
modifier = Modifier.weight(1f),
) {
Text("Minimize")
}
TextButton(
onClick = session.onReturnToHermes,
modifier = Modifier.weight(1f),
) {
Text("Hermes")
}
TextButton(
onClick = session.onDismissOverlay,
modifier = Modifier.weight(1f),
) {
Text("Hide")
}
TextButton(
onClick = session.onExit,
modifier = Modifier.weight(1f),
) {
Text("Exit")
}
}
}
}
}
@OptIn(ExperimentalFoundationApi::class)
@Composable
private fun VoiceFloatingOverlayBubble(
uiState: VoiceUiState,
stateText: String,
onExpand: () -> Unit,
onStartListening: () -> Unit,
onStopListening: () -> Unit,
onInterrupt: () -> Unit,
onPauseAutoMode: () -> Unit,
onDragBy: (Float, Float) -> Unit,
) {
val isHot = uiState.state == VoiceState.Listening || uiState.state == VoiceState.Speaking
val stateLabel = overlayBubbleStateLabel(uiState.state)
val tapAction = when (uiState.state) {
VoiceState.Idle, VoiceState.Error -> "start listening"
VoiceState.Listening -> "stop listening"
else -> "stop voice"
}
val containerColor = when (uiState.state) {
VoiceState.Listening, VoiceState.Speaking -> Color(0xFFE53935)
VoiceState.Transcribing, VoiceState.Thinking -> MaterialTheme.colorScheme.tertiary
VoiceState.Error -> MaterialTheme.colorScheme.error
VoiceState.Idle -> MaterialTheme.colorScheme.primary
}
val icon = when (uiState.state) {
VoiceState.Listening, VoiceState.Speaking -> Icons.Filled.Stop
VoiceState.Transcribing, VoiceState.Thinking -> Icons.Filled.GraphicEq
VoiceState.Idle, VoiceState.Error -> Icons.Filled.Mic
}
Box(
modifier = Modifier
.size(94.dp)
.pointerInput(Unit) {
detectDragGestures { change, dragAmount ->
change.consume()
onDragBy(dragAmount.x, dragAmount.y)
}
},
contentAlignment = Alignment.Center,
) {
OverlayCircularWaveformRing(
amplitude = uiState.amplitude,
state = uiState.state,
modifier = Modifier.fillMaxSize(),
)
Surface(
modifier = Modifier
.size(70.dp)
.clip(CircleShape)
.combinedClickable(
onClick = {
dispatchMicAction(
uiState = uiState,
onStartListening = onStartListening,
onStopListening = onStopListening,
onInterrupt = onInterrupt,
onPauseAutoMode = onPauseAutoMode,
)
},
onLongClick = onExpand,
onDoubleClick = onExpand,
),
shape = CircleShape,
color = containerColor.copy(alpha = if (isHot) 0.96f else 0.92f),
shadowElevation = 8.dp,
tonalElevation = 4.dp,
) {
Column(
modifier = Modifier.padding(horizontal = 8.dp, vertical = 8.dp),
horizontalAlignment = Alignment.CenterHorizontally,
verticalArrangement = Arrangement.Center,
) {
Icon(
imageVector = icon,
contentDescription = "Voice overlay $stateText. Tap to $tapAction, double tap to expand.",
tint = Color.White,
modifier = Modifier.size(23.dp),
)
Spacer(Modifier.height(2.dp))
Text(
text = stateLabel,
style = MaterialTheme.typography.labelSmall,
color = Color.White,
maxLines = 1,
overflow = TextOverflow.Ellipsis,
)
}
}
}
}
@Composable
private fun OverlayCircularWaveformRing(
amplitude: Float,
state: VoiceState,
modifier: Modifier = Modifier,
) {
val displayAmplitude = amplitude.coerceIn(0f, 1f)
val phase = rememberOverlayWaveformPhase(displayAmplitude)
val dim = MaterialTheme.colorScheme.onSurfaceVariant
val errorColor = MaterialTheme.colorScheme.error
val targetPrimary = when (state) {
VoiceState.Idle -> dim.copy(alpha = 0.38f)
VoiceState.Listening -> OverlayListeningPrimary
VoiceState.Transcribing -> OverlayListeningPrimary.copy(alpha = 0.72f)
VoiceState.Thinking -> dim.copy(alpha = 0.62f)
VoiceState.Speaking -> OverlaySpeakingPrimary
VoiceState.Error -> errorColor.copy(alpha = 0.9f)
}
val targetSecondary = when (state) {
VoiceState.Idle -> dim.copy(alpha = 0.28f)
VoiceState.Listening -> OverlayListeningSecondary
VoiceState.Transcribing -> dim.copy(alpha = 0.58f)
VoiceState.Thinking -> dim.copy(alpha = 0.48f)
VoiceState.Speaking -> OverlaySpeakingSecondary
VoiceState.Error -> errorColor.copy(alpha = 0.72f)
}
val primary by animateColorAsState(
targetValue = targetPrimary,
animationSpec = tween(durationMillis = 350),
label = "overlayRingPrimary",
)
val secondary by animateColorAsState(
targetValue = targetSecondary,
animationSpec = tween(durationMillis = 350),
label = "overlayRingSecondary",
)
Canvas(modifier = modifier) {
val minDimension = min(size.width, size.height)
if (minDimension <= 0f) return@Canvas
val center = androidx.compose.ui.geometry.Offset(size.width / 2f, size.height / 2f)
val innerRadius = minDimension * 0.385f
val baseLength = minDimension * 0.035f
val reactiveLength = minDimension * 0.105f
val strokePx = max(2f, minDimension * 0.024f)
val bars = 72
val baseline = when (state) {
VoiceState.Idle -> 0.04f
VoiceState.Thinking, VoiceState.Transcribing -> 0.11f
VoiceState.Error -> 0.08f
VoiceState.Listening, VoiceState.Speaking -> 0.08f
}
val envelope = max(displayAmplitude, baseline)
for (index in 0 until bars) {
val fraction = index / bars.toFloat()
val angle = (fraction * TWO_PI) - HALF_PI
val wobble = (
sin(angle * 3.0f + phase) * 0.48f +
sin(angle * 7.0f - phase * 0.7f) * 0.28f +
sin(angle * 13.0f + phase * 1.35f) * 0.16f
).coerceIn(-1f, 1f)
val normalized = (wobble + 1f) * 0.5f
val lineLength = baseLength + reactiveLength * envelope * normalized
val startRadius = innerRadius
val endRadius = innerRadius + lineLength
val start = androidx.compose.ui.geometry.Offset(
x = center.x + cos(angle) * startRadius,
y = center.y + sin(angle) * startRadius,
)
val end = androidx.compose.ui.geometry.Offset(
x = center.x + cos(angle) * endRadius,
y = center.y + sin(angle) * endRadius,
)
val colorMix = (sin(angle + phase * 0.35f) + 1f) * 0.5f
drawLine(
color = lerp(primary, secondary, colorMix),
start = start,
end = end,
strokeWidth = strokePx,
cap = StrokeCap.Round,
alpha = (0.56f + normalized * 0.36f).coerceIn(0f, 1f),
)
}
}
}
@Composable
private fun rememberOverlayWaveformPhase(amplitude: Float): Float {
val ampRef = rememberUpdatedState(amplitude.coerceIn(0f, 1f))
var phase by remember { mutableStateOf(0f) }
LaunchedEffect(Unit) {
var lastNanos = 0L
while (true) {
withFrameNanos { now ->
if (lastNanos != 0L) {
val deltaSeconds = (now - lastNanos) / 1_000_000_000f
val cyclesPerSecond = 0.28f + ampRef.value * 1.55f
phase = (phase + deltaSeconds * cyclesPerSecond * TWO_PI) % TWO_PI
}
lastNanos = now
}
}
}
return phase
}
private fun overlayBubbleStateLabel(state: VoiceState): String = when (state) {
VoiceState.Idle -> "Ready"
VoiceState.Listening -> "Listen"
VoiceState.Transcribing -> "STT"
VoiceState.Thinking -> "Think"
VoiceState.Speaking -> "Speak"
VoiceState.Error -> "Error"
}
@Composable
private fun MicControlButton(
uiState: VoiceUiState,
onStartListening: () -> Unit,
onStopListening: () -> Unit,
onInterrupt: () -> Unit,
onPauseAutoMode: () -> Unit,
) {
val isHot = uiState.state != VoiceState.Idle && uiState.state != VoiceState.Error
Surface(
modifier = Modifier
.size(44.dp)
.clip(CircleShape)
.clickable {
dispatchMicAction(
uiState = uiState,
onStartListening = onStartListening,
onStopListening = onStopListening,
onInterrupt = onInterrupt,
onPauseAutoMode = onPauseAutoMode,
)
},
shape = CircleShape,
color = if (isHot) Color(0xFFE53935) else MaterialTheme.colorScheme.primary,
shadowElevation = 4.dp,
) {
Box(contentAlignment = Alignment.Center) {
Icon(
imageVector = if (isHot) Icons.Filled.Stop else Icons.Filled.Mic,
contentDescription = "Voice overlay mic",
tint = Color.White,
modifier = Modifier.size(22.dp),
)
}
}
}
private fun dispatchMicAction(
uiState: VoiceUiState,
onStartListening: () -> Unit,
onStopListening: () -> Unit,
onInterrupt: () -> Unit,
onPauseAutoMode: () -> Unit,
) {
when (uiState.state) {
VoiceState.Listening -> {
if (uiState.interactionMode == InteractionMode.Continuous) {
onPauseAutoMode()
} else {
onStopListening()
}
}
VoiceState.Speaking, VoiceState.Transcribing, VoiceState.Thinking -> {
if (uiState.interactionMode == InteractionMode.Continuous) {
onPauseAutoMode()
} else {
onInterrupt()
}
}
VoiceState.Idle, VoiceState.Error -> onStartListening()
}
}
@Composable
private fun StatusChip(
text: String,
modifier: Modifier = Modifier,
) {
Surface(
modifier = modifier.height(24.dp),
shape = RoundedCornerShape(999.dp),
color = MaterialTheme.colorScheme.surfaceVariant.copy(alpha = 0.72f),
contentColor = MaterialTheme.colorScheme.onSurfaceVariant,
) {
Box(
modifier = Modifier.padding(horizontal = 8.dp),
contentAlignment = Alignment.Center,
) {
Text(
text = text,
style = MaterialTheme.typography.labelSmall,
maxLines = 1,
overflow = TextOverflow.Ellipsis,
)
}
}
}
private fun voiceProviderLabel(
provider: String?,
model: String?,
voice: String?,
outputEnabled: Boolean?,
): String {
if (outputEnabled == false) return "output off"
val providerPart = provider?.takeIf { it.isNotBlank() } ?: "provider ..."
val modelPart = model?.takeIf { it.isNotBlank() }
val voicePart = voice?.takeIf { it.isNotBlank() }
return listOfNotNull(providerPart, modelPart, voicePart).joinToString(" / ")
}
fun openHermesFromOverlay(context: Context) {
val appContext = context.applicationContext
val launchIntent = appContext.packageManager.getLaunchIntentForPackage(appContext.packageName)
?: Intent().setPackage(appContext.packageName)
launchIntent
.addFlags(Intent.FLAG_ACTIVITY_NEW_TASK)
.addFlags(Intent.FLAG_ACTIVITY_REORDER_TO_FRONT)
.addFlags(Intent.FLAG_ACTIVITY_SINGLE_TOP)
runCatching { appContext.startActivity(launchIntent) }
.onFailure { Log.w("VoiceOverlayHost", "return to Hermes failed", it) }
}
private class VoiceOverlayLifecycleOwner :
LifecycleOwner,
ViewModelStoreOwner,
SavedStateRegistryOwner {
private val registry = LifecycleRegistry(this)
override val lifecycle: Lifecycle get() = registry
private val store = ViewModelStore()
override val viewModelStore: ViewModelStore get() = store
private val savedStateController = SavedStateRegistryController.create(this)
override val savedStateRegistry: SavedStateRegistry
get() = savedStateController.savedStateRegistry
fun start() {
savedStateController.performRestore(null)
registry.currentState = Lifecycle.State.CREATED
registry.currentState = Lifecycle.State.RESUMED
}
fun stop() {
registry.currentState = Lifecycle.State.DESTROYED
store.clear()
}
}
@@ -0,0 +1,86 @@
package com.hermesandroid.relay.audio
import android.media.audiofx.AcousticEchoCanceler
import android.media.audiofx.NoiseSuppressor
import android.util.Log
import io.mockk.every
import io.mockk.mockk
import io.mockk.mockkStatic
import io.mockk.unmockkStatic
import kotlinx.coroutines.ExperimentalCoroutinesApi
import kotlinx.coroutines.test.StandardTestDispatcher
import kotlinx.coroutines.test.advanceUntilIdle
import kotlinx.coroutines.test.runTest
import org.junit.After
import org.junit.Assert.assertEquals
import org.junit.Assert.assertTrue
import org.junit.Before
import org.junit.Test
@OptIn(ExperimentalCoroutinesApi::class)
class BargeInListenerShutdownRaceTest {
@Before
fun setUp() {
mockkStatic(Log::class)
every { Log.w(any(), any<String>()) } returns 0
every { Log.i(any(), any<String>()) } returns 0
mockkStatic(AcousticEchoCanceler::class)
every { AcousticEchoCanceler.isAvailable() } returns false
mockkStatic(NoiseSuppressor::class)
every { NoiseSuppressor.isAvailable() } returns false
}
@After
fun tearDown() {
unmockkStatic(Log::class)
unmockkStatic(AcousticEchoCanceler::class)
unmockkStatic(NoiseSuppressor::class)
}
@Test
fun `closed vad during shutdown exits reader without crashing`() = runTest {
val vadEngine = mockk<VadEngine>()
every { vadEngine.analyze(any()) } throws
IllegalArgumentException("You can't use Vad after closing session!")
val source = OneFrameAudioSource()
val listener = BargeInListener(
audioSource = source,
vadEngine = vadEngine,
audioSessionIdProvider = { 42 },
readerDispatcher = StandardTestDispatcher(testScheduler),
)
listener.start(this)
advanceUntilIdle()
assertEquals(1, source.readCount)
assertTrue("source should be released after VAD shutdown race", source.released)
}
private class OneFrameAudioSource : BargeInListener.AudioFrameSource {
var readCount: Int = 0
private set
var released: Boolean = false
private set
override fun initialize(): Boolean = true
override fun start() { /* no-op */ }
override fun read(buffer: ShortArray, sizeInShorts: Int): Int {
readCount++
return if (readCount == 1) {
sizeInShorts
} else {
0
}
}
override fun stop() { /* no-op */ }
override fun release() {
released = true
}
}
}
@@ -291,4 +291,49 @@ class AuthManagerProfilesParseTest {
assertFalse(parsed[0].hasSoul)
assertEquals(0, parsed[0].skillCount)
}
@Test
fun parsesProfileApiMetadataWhenPresent() {
val profilesArray = buildJsonArray {
add(buildJsonObject {
put("name", "mizu")
put("model", "xai/grok")
put("api_server_enabled", true)
put("api_server_url", "https://hermes.example.test:8643")
put("api_server_host", "0.0.0.0")
put("api_server_port", 8643)
put("api_server_key_present", true)
})
}
val parsed = AuthManager.parseAgentProfiles(profilesArray)
assertEquals(1, parsed.size)
assertTrue(parsed[0].apiServerEnabled)
assertEquals("https://hermes.example.test:8643", parsed[0].apiServerUrl)
assertEquals("0.0.0.0", parsed[0].apiServerHost)
assertEquals(8643, parsed[0].apiServerPort)
assertTrue(parsed[0].apiServerKeyPresent)
assertTrue(parsed[0].hasIsolatedApi)
}
@Test
fun defaultsProfileApiMetadataWhenKeysMissing() {
val profilesArray = buildJsonArray {
add(buildJsonObject {
put("name", "legacy")
put("model", "m1")
})
}
val parsed = AuthManager.parseAgentProfiles(profilesArray)
assertEquals(1, parsed.size)
assertFalse(parsed[0].apiServerEnabled)
assertNull(parsed[0].apiServerUrl)
assertNull(parsed[0].apiServerHost)
assertNull(parsed[0].apiServerPort)
assertFalse(parsed[0].apiServerKeyPresent)
assertFalse(parsed[0].hasIsolatedApi)
}
}
@@ -0,0 +1,126 @@
package com.hermesandroid.relay.data
import org.junit.Assert.assertEquals
import org.junit.Assert.assertNull
import org.junit.Test
class AgentDisplayTest {
private val defaultProfile = Profile(
name = "default",
model = "grok-default",
description = "House Agent",
)
private val mizu = Profile(
name = "mizu",
model = "grok-mizu",
description = "Mizu",
)
@Test
fun effectiveProfile_prefersSelectedProfileOverDefaultProfile() {
val effective = AgentDisplay.effectiveProfile(
selectedProfile = mizu,
profiles = listOf(defaultProfile, mizu),
)
assertEquals(mizu, effective)
}
@Test
fun effectiveProfile_fallsBackToAdvertisedDefaultProfile() {
val effective = AgentDisplay.effectiveProfile(
selectedProfile = null,
profiles = listOf(mizu, defaultProfile),
)
assertEquals(defaultProfile, effective)
}
@Test
fun agentName_prefersProfileDescriptionThenProfileName() {
assertEquals(
"Mizu",
AgentDisplay.agentName(
profile = mizu,
selectedPersonality = "friendly",
defaultPersonality = "default-persona",
connectionLabel = "Lab",
),
)
assertEquals(
"Coder",
AgentDisplay.agentName(
profile = mizu.copy(name = "coder", description = ""),
selectedPersonality = "friendly",
defaultPersonality = "default-persona",
connectionLabel = "Lab",
),
)
}
@Test
fun agentName_fallsBackToPersonalityConnectionThenHermes() {
assertEquals(
"Research",
AgentDisplay.agentName(
profile = null,
selectedPersonality = "research",
defaultPersonality = "default",
connectionLabel = "Lab",
),
)
assertEquals(
"Default persona",
AgentDisplay.agentName(
profile = null,
selectedPersonality = "default",
defaultPersonality = "default persona",
connectionLabel = "Lab",
),
)
assertEquals(
"Lab",
AgentDisplay.agentName(
profile = null,
selectedPersonality = "default",
defaultPersonality = "",
connectionLabel = "Lab",
),
)
assertEquals(
"Hermes",
AgentDisplay.agentName(
profile = null,
selectedPersonality = "default",
defaultPersonality = "",
connectionLabel = "",
),
)
}
@Test
fun profileRequestName_normalizesDefaultAliasAndDropsBlankOrNull() {
assertNull(AgentDisplay.profileRequestName(null))
assertNull(AgentDisplay.profileRequestName(" "))
assertNull(AgentDisplay.profileRequestName(" default "))
assertEquals("mizu", AgentDisplay.profileRequestName("mizu"))
}
@Test
fun normalizeSelection_collapsesSyntheticDefaultProfile() {
assertNull(AgentDisplay.normalizeSelection(defaultProfile))
assertEquals(mizu, AgentDisplay.normalizeSelection(mizu))
}
@Test
fun profileContextKey_treatsDefaultAliasAsServerDefault() {
assertEquals(
"conn::__server_default__",
AgentDisplay.profileContextKey("conn", "default"),
)
assertEquals("conn::mizu", AgentDisplay.profileContextKey("conn", "mizu"))
}
}
@@ -4,12 +4,11 @@ import androidx.datastore.core.DataStore
import androidx.datastore.preferences.core.Preferences
import androidx.datastore.preferences.core.PreferenceDataStoreFactory
import kotlinx.coroutines.CoroutineScope
import kotlinx.coroutines.Job
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.SupervisorJob
import kotlinx.coroutines.cancel
import kotlinx.coroutines.flow.first
import kotlinx.coroutines.test.StandardTestDispatcher
import kotlinx.coroutines.test.TestScope
import kotlinx.coroutines.test.runTest
import kotlinx.coroutines.runBlocking
import org.junit.After
import org.junit.Assert.assertEquals
import org.junit.Assert.assertNull
@@ -39,7 +38,7 @@ class ProfileSelectionStoreTest {
@Before
fun setUp() {
scope = TestScope(StandardTestDispatcher() + Job())
scope = CoroutineScope(Dispatchers.IO + SupervisorJob())
val file: File = tempFolder.newFile("profile_selections_test.preferences_pb")
// PreferenceDataStoreFactory requires the backing file NOT exist yet;
// TemporaryFolder.newFile() creates it, so we delete first.
@@ -57,20 +56,20 @@ class ProfileSelectionStoreTest {
}
@Test
fun unset_connection_emitsNull() = runTest {
fun unset_connection_emitsNull() = runBlocking {
// Fresh store — every connection id reads as null until set.
assertNull(store.selectedProfileFlow("conn-1").first())
assertNull(store.selectedProfileFlow("conn-unknown").first())
}
@Test
fun set_then_get_roundTrips() = runTest {
fun set_then_get_roundTrips() = runBlocking {
store.setSelectedProfile("conn-1", "mizu")
assertEquals("mizu", store.selectedProfileFlow("conn-1").first())
}
@Test
fun set_null_clearsTheKey() = runTest {
fun set_null_clearsTheKey() = runBlocking {
store.setSelectedProfile("conn-1", "mizu")
assertEquals("mizu", store.selectedProfileFlow("conn-1").first())
@@ -82,7 +81,7 @@ class ProfileSelectionStoreTest {
}
@Test
fun clear_removesOnlyTheGivenConnection() = runTest {
fun clear_removesOnlyTheGivenConnection() = runBlocking {
store.setSelectedProfile("conn-1", "mizu")
store.setSelectedProfile("conn-2", "coder")
@@ -93,7 +92,7 @@ class ProfileSelectionStoreTest {
}
@Test
fun perConnectionKeys_areIndependent() = runTest {
fun perConnectionKeys_areIndependent() = runBlocking {
// Writing to one connection must not touch another — the key
// factory is the contract and regressions here would cascade.
store.setSelectedProfile("conn-A", "alpha")
@@ -106,7 +105,7 @@ class ProfileSelectionStoreTest {
}
@Test
fun overwrite_replacesPriorValue() = runTest {
fun overwrite_replacesPriorValue() = runBlocking {
store.setSelectedProfile("conn-1", "mizu")
store.setSelectedProfile("conn-1", "coder")
assertEquals("coder", store.selectedProfileFlow("conn-1").first())
@@ -0,0 +1,93 @@
package com.hermesandroid.relay.data
import androidx.datastore.core.DataStore
import androidx.datastore.preferences.core.PreferenceDataStoreFactory
import androidx.datastore.preferences.core.Preferences
import kotlinx.coroutines.CoroutineScope
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.SupervisorJob
import kotlinx.coroutines.cancel
import kotlinx.coroutines.flow.first
import kotlinx.coroutines.runBlocking
import org.junit.After
import org.junit.Assert.assertEquals
import org.junit.Assert.assertNull
import org.junit.Before
import org.junit.Rule
import org.junit.Test
import org.junit.rules.TemporaryFolder
import java.io.File
class ProfileSessionStoreTest {
@get:Rule
val tempFolder = TemporaryFolder()
private lateinit var scope: CoroutineScope
private lateinit var dataStore: DataStore<Preferences>
private lateinit var store: ProfileSessionStore
@Before
fun setUp() {
scope = CoroutineScope(Dispatchers.IO + SupervisorJob())
val file: File = tempFolder.newFile("profile_sessions_test.preferences_pb")
if (file.exists()) file.delete()
dataStore = PreferenceDataStoreFactory.create(
scope = scope,
produceFile = { file },
)
store = ProfileSessionStore(dataStore)
}
@After
fun tearDown() {
scope.cancel()
}
@Test
fun setAndGet_defaultProfileSession() = runBlocking {
store.setSessionId("conn-1", null, "session-default")
assertEquals(
"session-default",
store.sessionIdFlow("conn-1", null).first(),
)
}
@Test
fun profileSessionsAreIndependentFromDefaultAndEachOther() = runBlocking {
store.setSessionId("conn-1", null, "session-default")
store.setSessionId("conn-1", "mizu", "session-mizu")
store.setSessionId("conn-1", "coder", "session-coder")
store.setSessionId("conn-2", "mizu", "session-other")
assertEquals("session-default", store.sessionIdFlow("conn-1", null).first())
assertEquals("session-mizu", store.sessionIdFlow("conn-1", "mizu").first())
assertEquals("session-coder", store.sessionIdFlow("conn-1", "coder").first())
assertEquals("session-other", store.sessionIdFlow("conn-2", "mizu").first())
}
@Test
fun nullSessionClearsOnlyThatProfileSlot() = runBlocking {
store.setSessionId("conn-1", "mizu", "session-mizu")
store.setSessionId("conn-1", "coder", "session-coder")
store.setSessionId("conn-1", "mizu", null)
assertNull(store.sessionIdFlow("conn-1", "mizu").first())
assertEquals("session-coder", store.sessionIdFlow("conn-1", "coder").first())
}
@Test
fun clearConnectionRemovesAllProfilesForThatConnectionOnly() = runBlocking {
store.setSessionId("conn-1", null, "session-default")
store.setSessionId("conn-1", "mizu", "session-mizu")
store.setSessionId("conn-2", "mizu", "session-other")
store.clearConnection("conn-1")
assertNull(store.sessionIdFlow("conn-1", null).first())
assertNull(store.sessionIdFlow("conn-1", "mizu").first())
assertEquals("session-other", store.sessionIdFlow("conn-2", "mizu").first())
}
}
@@ -199,6 +199,51 @@ class ProfileTest {
assertEquals(original, decoded)
}
@Test
fun deserializesProfileApiMetadata() {
val payload = """
{
"name": "mizu",
"model": "xai/grok",
"description": "",
"api_server_enabled": true,
"api_server_url": "https://hermes.example.test:8643",
"api_server_host": "0.0.0.0",
"api_server_port": 8643,
"api_server_key_present": true
}
""".trimIndent()
val profile = json.decodeFromString(Profile.serializer(), payload)
assertTrue(profile.apiServerEnabled)
assertEquals("https://hermes.example.test:8643", profile.apiServerUrl)
assertEquals("0.0.0.0", profile.apiServerHost)
assertEquals(8643, profile.apiServerPort)
assertTrue(profile.apiServerKeyPresent)
assertTrue(profile.hasIsolatedApi)
}
@Test
fun profileApiMetadataDefaultsWhenKeysMissing() {
val payload = """
{
"name": "legacy",
"model": "m1",
"description": ""
}
""".trimIndent()
val profile = json.decodeFromString(Profile.serializer(), payload)
assertFalse(profile.apiServerEnabled)
assertNull(profile.apiServerUrl)
assertNull(profile.apiServerHost)
assertNull(profile.apiServerPort)
assertFalse(profile.apiServerKeyPresent)
assertFalse(profile.hasIsolatedApi)
}
@Test
fun ignoresUnknownServerSideFields() {
// Upstream might add fields (temperature, tools, …) without a
@@ -0,0 +1,27 @@
package com.hermesandroid.relay.data
import org.junit.Assert.assertEquals
import org.junit.Test
class VoiceEngineModeTest {
@Test
fun fromStorage_defaultsToStableEngine() {
assertEquals(
VoiceEngineMode.HermesVoiceOutput,
VoiceEngineMode.fromStorage(null),
)
assertEquals(
VoiceEngineMode.HermesVoiceOutput,
VoiceEngineMode.fromStorage("missing"),
)
}
@Test
fun fromStorage_readsRealtimeAgent() {
assertEquals(
VoiceEngineMode.RealtimeAgent,
VoiceEngineMode.fromStorage("realtime_agent"),
)
}
}
@@ -116,6 +116,61 @@ class HermesApiClientTest {
assertEquals(ChatMode.DISCONNECTED, ChatMode.valueOf("DISCONNECTED"))
}
// --- ServerCapabilities endpoint resolution ---
@Test
fun serverCapabilities_preferredEndpoint_prefersSessionsStream() {
val capabilities = ServerCapabilities(
sessionsApi = true,
sessionsChatStream = true,
runs = true,
portable = true,
healthy = true,
)
assertEquals("sessions", capabilities.preferredChatEndpoint())
}
@Test
fun serverCapabilities_preferredEndpoint_prefersCompletionsOverRuns() {
val capabilities = ServerCapabilities(
sessionsApi = true,
sessionsChatStream = false,
runs = true,
portable = true,
healthy = true,
)
assertEquals("completions", capabilities.preferredChatEndpoint())
}
@Test
fun serverCapabilities_preferredEndpoint_usesRunsOnlyWhenExplicitlyStreaming() {
val capabilities = ServerCapabilities(
sessionsApi = true,
sessionsChatStream = false,
runs = true,
portable = false,
healthy = true,
)
assertEquals("runs", capabilities.preferredChatEndpoint())
}
@Test
fun serverCapabilities_preferredEndpoint_usesCompletionsForIssue52Shape() {
val capabilities = ServerCapabilities(
sessionsApi = true,
sessionsChatStream = false,
runs = false,
portable = true,
healthy = true,
)
assertEquals("completions", capabilities.preferredChatEndpoint())
assertEquals(ChatMode.ENHANCED_HERMES, capabilities.toChatMode())
}
// --- URL construction patterns ---
// These verify the string patterns used by authRequest() inside the client.
@@ -0,0 +1,100 @@
package com.hermesandroid.relay.network
import com.hermesandroid.relay.network.models.CreateSessionRequest
import kotlinx.serialization.encodeToString
import kotlinx.serialization.json.Json
import kotlinx.serialization.json.JsonObject
import kotlinx.serialization.json.boolean
import kotlinx.serialization.json.contentOrNull
import kotlinx.serialization.json.jsonArray
import kotlinx.serialization.json.jsonObject
import kotlinx.serialization.json.jsonPrimitive
import org.junit.Assert.assertEquals
import org.junit.Assert.assertFalse
import org.junit.Assert.assertTrue
import org.junit.Test
class HermesChatPayloadsTest {
private val json = Json { ignoreUnknownKeys = true }
@Test
fun createSessionRequest_serializesProfileWhenExplicitlySelected() {
val body = json.encodeToString(
CreateSessionRequest(
title = "Mizu chat",
model = "grok-mizu",
profile = "mizu",
),
)
val parsed = json.decodeFromString<JsonObject>(body)
assertEquals("Mizu chat", parsed["title"]?.jsonPrimitive?.contentOrNull)
assertEquals("grok-mizu", parsed["model"]?.jsonPrimitive?.contentOrNull)
assertEquals("mizu", parsed["profile"]?.jsonPrimitive?.contentOrNull)
}
@Test
fun sessionChatPayload_includesProfileModelAndSystemMessage() {
val payload = buildSessionChatStreamPayload(
message = "hello",
systemMessage = "You are Mizu.",
modelOverride = "grok-mizu",
profileName = "mizu",
)
assertEquals("hello", payload["message"]?.jsonPrimitive?.contentOrNull)
assertEquals("You are Mizu.", payload["system_message"]?.jsonPrimitive?.contentOrNull)
assertEquals("grok-mizu", payload["model"]?.jsonPrimitive?.contentOrNull)
assertEquals("mizu", payload["profile"]?.jsonPrimitive?.contentOrNull)
}
@Test
fun sessionChatPayload_omitsProfileWhenServerDefaultSelected() {
val payload = buildSessionChatStreamPayload(
message = "hello",
profileName = null,
)
assertFalse(payload.containsKey("profile"))
}
@Test
fun runPayload_includesProfileAlongsideFallbackModelAndSystemMessage() {
val payload = buildRunStreamPayload(
message = "hello",
model = "default-model",
systemMessage = "You are Coder.",
modelOverride = "grok-coder",
profileName = "coder",
)
assertEquals("hello", payload["input"]?.jsonPrimitive?.contentOrNull)
assertEquals("grok-coder", payload["model"]?.jsonPrimitive?.contentOrNull)
assertEquals("You are Coder.", payload["system_message"]?.jsonPrimitive?.contentOrNull)
assertEquals("coder", payload["profile"]?.jsonPrimitive?.contentOrNull)
assertTrue(payload["stream"]?.jsonPrimitive?.boolean == true)
}
@Test
fun chatCompletionsPayload_usesOpenAiMessagesAndSseStream() {
val payload = buildChatCompletionsStreamPayload(
message = "hello",
model = "default-model",
systemMessage = "You are Coder.",
modelOverride = "grok-coder",
profileName = "coder",
)
assertEquals("grok-coder", payload["model"]?.jsonPrimitive?.contentOrNull)
assertEquals("coder", payload["profile"]?.jsonPrimitive?.contentOrNull)
assertTrue(payload["stream"]?.jsonPrimitive?.boolean == true)
val messages = payload["messages"]!!.jsonArray
assertEquals(2, messages.size)
assertEquals("system", messages[0].jsonObject["role"]?.jsonPrimitive?.contentOrNull)
assertEquals("You are Coder.", messages[0].jsonObject["content"]?.jsonPrimitive?.contentOrNull)
assertEquals("user", messages[1].jsonObject["role"]?.jsonPrimitive?.contentOrNull)
assertEquals("hello", messages[1].jsonObject["content"]?.jsonPrimitive?.contentOrNull)
}
}
@@ -0,0 +1,62 @@
package com.hermesandroid.relay.network
import org.junit.Assert.assertEquals
import org.junit.Assert.assertNull
import org.junit.Test
class ProfileApiUrlResolverTest {
@Test
fun resolveForConnection_rewritesLoopbackProfileHostToBaseHost() {
assertEquals(
"http://172.16.24.250:8647",
ProfileApiUrlResolver.resolveForConnection(
profileApiUrl = "http://127.0.0.1:8647",
baseApiUrl = "http://172.16.24.250:8642",
),
)
}
@Test
fun resolveForConnection_rewritesZeroBindHostToBaseHost() {
assertEquals(
"https://docker-server.tailnet.ts.net:8646",
ProfileApiUrlResolver.resolveForConnection(
profileApiUrl = "http://0.0.0.0:8646/",
baseApiUrl = "https://docker-server.tailnet.ts.net:8642/",
),
)
}
@Test
fun resolveForConnection_keepsRemoteProfileHost() {
assertEquals(
"http://192.168.1.50:8647",
ProfileApiUrlResolver.resolveForConnection(
profileApiUrl = "http://192.168.1.50:8647",
baseApiUrl = "http://172.16.24.250:8642",
),
)
}
@Test
fun resolveForConnection_keepsLoopbackWhenBaseIsAlsoLoopback() {
assertEquals(
"http://127.0.0.1:8647",
ProfileApiUrlResolver.resolveForConnection(
profileApiUrl = "http://127.0.0.1:8647",
baseApiUrl = "http://localhost:8642",
),
)
}
@Test
fun resolveForConnection_handlesBlankProfileUrl() {
assertNull(
ProfileApiUrlResolver.resolveForConnection(
profileApiUrl = " ",
baseApiUrl = "http://172.16.24.250:8642",
),
)
}
}
@@ -491,6 +491,40 @@ class ChatHandlerTest {
assertEquals(MessageRole.ASSISTANT, handler.messages.value[0].role)
}
@Test
fun loadMessageHistory_preservesActiveAgentNameOnAssistantMessages() {
handler.activeAgentName = "Mizuki"
val items = listOf(
MessageItem(id = "1", role = "user", content = JsonPrimitive("Hello")),
MessageItem(id = "2", role = "assistant", content = JsonPrimitive("Hi there")),
)
handler.loadMessageHistory(items)
assertNull(handler.messages.value[0].agentName)
assertEquals("Mizuki", handler.messages.value[1].agentName)
}
@Test
fun relabelGenericAssistantMessages_replacesHermesButPreservesActionLabels() {
handler.loadMessageHistory(
listOf(
MessageItem(id = "1", role = "assistant", content = JsonPrimitive("Hi")),
MessageItem(id = "2", role = "assistant", content = JsonPrimitive("Custom")),
),
)
handler.activeAgentName = "Hermes"
handler.relabelGenericAssistantMessages("Hermes")
handler.appendLocalVoiceIntentResult("Opened Chrome")
handler.relabelGenericAssistantMessages("Victor")
val messages = handler.messages.value
assertEquals("Victor", messages[0].agentName)
assertEquals("Victor", messages[1].agentName)
assertEquals("Voice action", messages.last().agentName)
}
@Test
fun loadMessageHistory_convertsSystemMessages() {
val items = listOf(
@@ -0,0 +1,111 @@
package com.hermesandroid.relay.ui.components
import com.hermesandroid.relay.data.ChatMessage
import com.hermesandroid.relay.data.MessageRole
import com.hermesandroid.relay.viewmodel.VoiceState
import com.hermesandroid.relay.viewmodel.VoiceUiState
import org.junit.Assert.assertEquals
import org.junit.Assert.assertNull
import org.junit.Test
class VoiceModeOverlayStateTest {
@Test
fun pendingTranscript_showsWhileThinkingBeforeChatHistoryCatchesUp() {
val text = pendingVoiceTranscriptText(
uiState = VoiceUiState(
state = VoiceState.Thinking,
transcribedText = "What did you hear?",
),
visibleTranscriptMessages = emptyList(),
)
assertEquals("What did you hear?", text)
}
@Test
fun pendingTranscript_hidesWhenSameUserMessageIsAlreadyInChatHistory() {
val text = pendingVoiceTranscriptText(
uiState = VoiceUiState(
state = VoiceState.Thinking,
transcribedText = "Open settings",
),
visibleTranscriptMessages = listOf(
ChatMessage(
id = "user-1",
role = MessageRole.USER,
content = "Open settings",
timestamp = 1L,
),
),
)
assertNull(text)
}
@Test
fun pendingTranscript_hidesWhenAssistantPlaceholderFollowsSameUserMessage() {
val text = pendingVoiceTranscriptText(
uiState = VoiceUiState(
state = VoiceState.Thinking,
transcribedText = "Open settings",
),
visibleTranscriptMessages = listOf(
ChatMessage(
id = "user-1",
role = MessageRole.USER,
content = "Open settings",
timestamp = 1L,
),
ChatMessage(
id = "assistant-1",
role = MessageRole.ASSISTANT,
content = "",
timestamp = 2L,
isStreaming = true,
),
),
)
assertNull(text)
}
@Test
fun pendingTranscript_doesNotDeduplicateAgainstOlderMatchingTurn() {
val text = pendingVoiceTranscriptText(
uiState = VoiceUiState(
state = VoiceState.Thinking,
transcribedText = "Open settings",
),
visibleTranscriptMessages = listOf(
ChatMessage(
id = "user-1",
role = MessageRole.USER,
content = "Open settings",
timestamp = 1L,
),
ChatMessage(
id = "assistant-1",
role = MessageRole.ASSISTANT,
content = "Settings are open.",
timestamp = 2L,
),
),
)
assertEquals("Open settings", text)
}
@Test
fun pendingTranscript_hidesOutsideWaitingState() {
val text = pendingVoiceTranscriptText(
uiState = VoiceUiState(
state = VoiceState.Speaking,
transcribedText = "Tell me a joke",
),
visibleTranscriptMessages = emptyList(),
)
assertNull(text)
}
}
@@ -0,0 +1,100 @@
package com.hermesandroid.relay.viewmodel
import org.junit.Assert.assertEquals
import org.junit.Assert.assertTrue
import org.junit.Test
class BalancedRealtimeTtsCoalescerTest {
@Test
fun `normal speech waits until balanced first chunk target`() {
val coalescer = BalancedRealtimeTtsCoalescer(
firstChunkTargetChars = 90,
followupChunkTargetChars = 140,
maxChunkChars = 220,
)
assertTrue(coalescer.append("Test two, coming through.").isEmpty())
val out = coalescer.append(
"This is a longer sentence to check pacing, pauses, and voice stability.",
)
assertEquals(
listOf(
"Test two, coming through. " +
"This is a longer sentence to check pacing, pauses, and voice stability.",
),
out,
)
}
@Test
fun `stream completion flushes trailing short speech`() {
val coalescer = BalancedRealtimeTtsCoalescer(
firstChunkTargetChars = 90,
followupChunkTargetChars = 140,
maxChunkChars = 220,
)
assertTrue(coalescer.append("Okay.").isEmpty())
assertEquals(listOf("Okay."), coalescer.flush())
}
@Test
fun `immediate status flushes pending prose before status`() {
val coalescer = BalancedRealtimeTtsCoalescer(
firstChunkTargetChars = 120,
followupChunkTargetChars = 160,
maxChunkChars = 220,
)
assertTrue(coalescer.append("I can check that from the relay logs.").isEmpty())
assertEquals(
listOf(
"I can check that from the relay logs.",
"I'm checking the relay logs.",
),
coalescer.enqueueImmediate("I'm checking the relay logs."),
)
}
@Test
fun `long prose splits at a natural boundary before max chars`() {
val coalescer = BalancedRealtimeTtsCoalescer(
firstChunkTargetChars = 80,
followupChunkTargetChars = 120,
maxChunkChars = 130,
)
val text = "This first sentence is intentionally long enough to trigger a render. " +
"This second sentence should remain queued for a later flush instead of forcing " +
"one oversized provider render."
val first = coalescer.append(text)
assertEquals(1, first.size)
assertTrue(first.single().length <= 130)
assertTrue(first.single().endsWith("."))
val rest = coalescer.flush()
assertEquals(1, rest.size)
assertTrue(rest.single().startsWith("This second sentence"))
}
@Test
fun `clear drops pending speech and resets first chunk behavior`() {
val coalescer = BalancedRealtimeTtsCoalescer(
firstChunkTargetChars = 50,
followupChunkTargetChars = 100,
maxChunkChars = 180,
)
assertTrue(coalescer.append("This pending chunk should be discarded.").isEmpty())
coalescer.clear()
assertTrue(coalescer.append("Short reset.").isEmpty())
assertEquals(listOf("Short reset."), coalescer.flush())
}
}
@@ -0,0 +1,23 @@
package com.hermesandroid.relay.voice
import com.hermesandroid.relay.viewmodel.brokeredToolStartStatusForTts
import org.junit.Assert.assertEquals
import org.junit.Test
class VoiceToolStatusSpeechTest {
@Test
fun `maps common Hermes tool starts to short spoken status`() {
assertEquals("I'll search that now.", brokeredToolStartStatusForTts("web_search", 0))
assertEquals("I'll run a quick check.", brokeredToolStartStatusForTts("terminal", 0))
assertEquals("I'll check the phone.", brokeredToolStartStatusForTts("android_tap", 0))
assertEquals("I'll check the desktop.", brokeredToolStartStatusForTts("desktop_computer_action", 0))
assertEquals("I'll check the relevant files.", brokeredToolStartStatusForTts("read_file", 0))
}
@Test
fun `keeps repeated tool status brief`() {
assertEquals("I'm checking one more thing.", brokeredToolStartStatusForTts("web_search", 1))
assertEquals("Let me check that.", brokeredToolStartStatusForTts("unknown_tool", 0))
}
}
+2
View File
@@ -1,5 +1,7 @@
node_modules/
dist/
tray/src-tauri/bin/
tray/src-tauri/target/
*.log
.DS_Store
*.tsbuildinfo
+156 -42
View File
@@ -1,30 +1,67 @@
# hermes-relay-cli
Thin-client CLI for [Hermes-Relay](https://github.com/Codename-11/hermes-relay) — talk to a remote [Hermes agent](https://github.com/NousResearch/hermes-agent) over WSS from any terminal.
Desktop tray app and thin-client CLI for [Hermes-Relay](https://github.com/Codename-11/hermes-relay) — talk to a remote [Hermes agent](https://github.com/NousResearch/hermes-agent) over WSS from a native tray surface or any terminal.
The agent brain (LLM + tools + sessions + memory) runs on your Hermes host. This CLI is the local line-mode thin-client: it handles pairing, persists the session token, and renders the agent's stream to plain stdout so `>`, `|`, and `jq` all work.
The agent brain (LLM + tools + sessions + memory) runs on your Hermes host. The Windows tray app is the default desktop surface: pair one active relay, chat from the dashboard, start/pause the daemon, open or manage remote TUI sessions in a real terminal, install desktop surface plugins such as Herm, view devices, inspect the task log, copy terminal/shim commands, run local diagnostics, edit compact settings, see the compact above-taskbar overlay pill, and emergency-stop from the tray or hotkey. The CLI remains the local line-mode thin-client: it handles pairing, persists the session token, resumes tmux-backed TUI sessions, and renders the agent's stream to plain stdout so `>`, `|`, and `jq` all work.
> **What this is not:** A local Hermes install. Point it at an existing Hermes-Relay server (`ws://host:8767`). For the full TUI with Ink, see the sibling package [`ui-tui`](../../hermes-agent-tui-smoke/ui-tui) in the hermes-agent fork.
## Desktop control posture
The Windows tray app is the primary desktop install surface for most users. It bundles the compiled CLI as a Tauri sidecar and uses the same `~/.hermes/remote-sessions.json` and relay session APIs as the CLI. The CLI and daemon remain first-class for macOS/Linux, headless hosts, scripts, terminals, and operators.
| Surface | Intended use | Default experience |
|---------|--------------|--------------------|
| Windows tray | Daily desktop use | Tauri tray + one active relay + compact status-only above-taskbar overlay pill + tray/hotkey pause/emergency stop |
| Tray dashboard | Management | Pair/replace, Chat, first-class embedded TUI tab, Terminal/CLI launcher and commands, surface plugins, diagnostics, devices/revoke, grants, task log, compact settings, collapsed Advanced controls |
| CLI/daemon | Operator and headless use | Commands, flags, scripts, and JSON policy |
Tauri is optional outside Windows. The CLI and `hermes-relay daemon` must continue to work without a native app installed.
Tray behavior is intentionally split: left-click the tray icon to open the dashboard, right-click it for the native management menu, and use the compact overlay pill as a click-through status indicator only. The dashboard defaults to a dark graphite UI, keeps Start, Pause, and Emergency Stop in the sticky topbar so daemon controls stay reachable while views scroll, and exposes one active desktop relay instance at a time. Pairing is a first-class flow with pasted pairing invites, manual code, stored-session selection, LAN/Tailscale/manual route preview, and explicit replacement confirmation before changing the active relay. A raw relay URL in Advanced is not treated as paired until the matching stored session token exists, so unpaired installs keep daemon, devices, and TUI actions gated with "Pair first" UI instead of silently falling back to localhost.
The Chat tab is the first-run conversational surface. If the desktop is already paired, Chat streams through the saved relay and reuses the same CLI/gateway session path as `hermes-relay chat --json`. If the desktop is not paired, Chat offers a direct Hermes gateway/API URL (`http://host:8642`) with an optional API bearer for that tray session; only the URL is saved in `~/.hermes/desktop-control.json`. Relay pairing remains the fuller desktop-control path for daemon, terminal, devices, grants, and local tools, while direct gateway mode is for chat-only WebAPI access.
The TUI tab owns the experimental embedded xterm/PTY surface inside the dashboard, while Terminal / CLI opens the remote TUI in a real terminal and turns the active relay into copyable standard `hermes-relay` commands for remote TUI, one-shot chat, headless daemon, TUI sessions, status, tool inventory, and doctor. The Plugins view registers terminal surface plugins that can be installed, updated, launched externally, or embedded in the same xterm/PTY surface. The Sessions view lists global server-side `hermes-*` tmux sessions, resumes named sessions, creates new sessions, kills stale sessions, and copies the matching CLI commands. Those commands use the saved active relay by default; `--remote` is shown only as an explicit one-off override. Diagnostics renders local install/session checks and can run `hermes-relay doctor --json` through the bundled sidecar. Settings keep daily controls short and put raw relay override, computer-use flag, emergency hotkey, and blocklist under a collapsed Advanced disclosure. The sidebar is navigation-only and uses the same Chevron Compass brand mark as the Android/docs surfaces. The pill sizes itself to the current state (`Observing`, `TUI active`, `Tools`, `Gateway`, `Approval`, `Paused`, `Offline`, or `Unavailable`) instead of reserving dashboard-width space. Start, pause, emergency stop, chat turns, embedded/external TUI session actions, plugin launches, grant resolution, settings saves, doctor runs, and log clears all emit a shared dashboard refresh event so the dashboard and overlay pill stay on the same activity state.
## Install
### GitHub Release binary (recommended, no Node required)
```sh
# macOS / Linux
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.sh | sh
```
### GitHub Release install (recommended, no Node required)
```powershell
# Windows
# Windows tray app (default)
irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
```
Downloads a prebuilt single-file binary from GitHub Releases (Bun `--compile`, ~60–110 MB per platform) into `~/.hermes/bin/`. No Node.js install needed. SHA256 verified against `SHA256SUMS.txt` before install; version-aware so a re-install prints `upgrading X → Y`. Pin a specific release with `HERMES_RELAY_VERSION=desktop-v0.3.0-alpha.1`, override the install dir with `HERMES_RELAY_INSTALL_DIR=...`.
```powershell
# Windows CLI only
$env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
```
After install, either `hermes-relay <prompt>` or the short alias `hermes <prompt>` works — the installer drops a `hermes` symlink (POSIX) / `hermes.cmd` shim (Windows) next to the binary for muscle-memory parity with the upstream hermes-agent CLI. Collision-safe: if a `hermes` already exists in the install dir (e.g. from a local hermes-agent install), the installer leaves it untouched and prints a skip notice.
```sh
# macOS / Linux CLI
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.sh | sh
```
> **Experimental.** Binaries are currently unsigned — Windows SmartScreen and macOS Gatekeeper may warn on first launch. The installers print the `Unblock-File` / `xattr -dr com.apple.quarantine` escape hatches. Code signing lands before v1.0.
Windows downloads and verifies `hermes-relay-desktop-windows-x64-setup.exe`, then launches the tray installer. The tray app bundles the compiled CLI sidecar so pair, daemon start/stop, devices, revoke, and task log work without a separate PATH install. CLI-only installs download the prebuilt single-file binary from GitHub Releases (Bun `--compile`, ~60–110 MB per platform) into `~/.hermes/bin/`. Pin a specific release with `HERMES_RELAY_VERSION=desktop-v0.3.0-alpha.1`; CLI-only installs can override the install dir with `HERMES_RELAY_INSTALL_DIR=...`.
After install, use `hermes-relay <prompt>`. The shorter `hermes <prompt>` alias is optional because it can shadow a real local hermes-agent install. Enable it only when you want hermes-relay to be the `hermes` command for tools like Orca:
```powershell
# Windows enable / disable
$env:HERMES_RELAY_HERMES_ALIAS='enable'; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/hermes-alias.ps1 | iex
$env:HERMES_RELAY_HERMES_ALIAS='disable'; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/hermes-alias.ps1 | iex
```
```sh
# macOS / Linux enable / disable
HERMES_RELAY_HERMES_ALIAS=enable curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/hermes-alias.sh | sh
HERMES_RELAY_HERMES_ALIAS=disable curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/hermes-alias.sh | sh
```
Both alias managers are collision-safe: they create/remove only the hermes-relay-owned alias and refuse to overwrite or delete an unrelated `hermes` command.
> **Experimental.** Assets are currently unsigned — Windows SmartScreen and macOS Gatekeeper may warn on first launch. The CLI installers print the `Unblock-File` / `xattr -dr com.apple.quarantine` escape hatches. Code signing lands before v1.0.
### Local clone + npm link
@@ -40,9 +77,18 @@ npm link
The package name in `package.json` is workspace metadata only today. The desktop CLI is not published to npm.
For tray development on Windows:
```sh
npm run tray:dev
npm run tray:check
npm run tray:test
npm run tray:build
```
## Uninstall
Three tiers — same shape on both platforms. Default keeps `~/.hermes/remote-sessions.json` so a future re-install pairs seamlessly.
Uninstall modes use the same shape on both platforms. Default keeps `~/.hermes/remote-sessions.json` so a future re-install pairs seamlessly.
### curl / irm
@@ -62,15 +108,15 @@ irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scri
$env:HERMES_RELAY_UNINSTALL_PURGE=1; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/uninstall.ps1 | iex
```
### Tiers
### Modes
| Flag | What it removes |
|-------------------|---------------------------------------------------------------------------------------------------------------|
| *(default)* | `~/.hermes/bin/hermes-relay[.exe]` plus the `hermes` / `hermes.cmd` alias (only if it points at our binary) and the Windows user-PATH entry. Preserves `~/.hermes/remote-sessions.json`. |
| *(default)* | `~/.hermes/bin/hermes-relay[.exe]` plus any hermes-relay-owned optional `hermes` / `hermes.cmd` alias and the Windows user-PATH entry. Preserves `~/.hermes/remote-sessions.json`. |
| `--purge` | Also deletes `~/.hermes/remote-sessions.json` — bearer tokens, cert pins, tools-consent flag. |
| `--service` | Stub. Prints the commands to remove a manually-installed systemd unit / launchd plist / Windows service. |
Tiers combine: `--purge --service` runs both.
Modes combine: `--purge --service` runs both.
**Heads-up about `--purge`:** `remote-sessions.json` is shared with the Ink TUI and Android desktop tooling. Wiping it signs those surfaces out too. Use `--purge` when giving the machine away — not for routine cleanup.
@@ -82,26 +128,68 @@ Tiers combine: `--purge --service` runs both.
## First-time pairing
On the Hermes host, mint a one-time pairing code:
On the Hermes host, mint a one-time pairing invite:
```sh
# on the relay host
hermes-pair # or: python -m plugin.pair --register-code
# → prints e.g. "CODE: F3W7EY (TTL: 10m)"
hermes-pair
# -> prints "Copy/paste pairing invite" with hermes-relay://pair?payload=...
```
On your machine, pair:
In the tray app, open **Pair** and paste the `hermes-relay://pair?...` URL into
**Paste invite**. From a terminal, use the same invite directly:
```sh
hermes-relay pair --remote ws://172.16.24.250:8767
# prompts for the code, then:
hermes-relay pair --pair-qr 'hermes-relay://pair?payload=...' --grant-tools
# ✓ Paired. Token stored in ~/.hermes/remote-sessions.json
# Server: 0.6.0
# Relay: ws://172.16.24.250:8767
```
Manual URL + six-character code pairing still works with
`hermes-relay pair --remote ws://172.16.24.250:8767`, but the invite URL is
the preferred path because it carries endpoint candidates and the correct
relay one-shot code.
Now subsequent `hermes-relay ...` calls reuse the stored session token. Tokens live at `~/.hermes/remote-sessions.json` (mode 0600) — same file the Ink TUI uses, so pairing once from either surface works for both.
### Tray Chat
Open **Chat** after pairing to stream a desktop chat turn through the saved relay. The tray spawns the bundled CLI sidecar in JSON mode, so relay auth, session resume, gateway events, and stop/cancel behavior stay aligned with `hermes-relay chat`.
If you are not paired, switch the Chat route to **Gateway/API** and enter a Hermes WebAPI base URL such as `http://host:8642`. The tray probes `/api/sessions/*/chat/stream` first and falls back to `/v1/runs` when that is the available upstream path. API key is optional and kept in memory for the current tray session; only the gateway URL is saved.
### Tray TUI, Terminal / CLI and Diagnostics
After pairing from the tray, open **TUI** to run the embedded dashboard terminal, or open **Terminal** to launch the remote TUI in a real terminal and copy commands that target the active desktop relay:
```powershell
hermes-relay
hermes-relay sessions list
hermes-relay chat "summarize my current project"
hermes-relay daemon --log-human
hermes-relay status
hermes-relay tools
hermes-relay doctor --json
```
The copied commands intentionally use the standard `hermes-relay` command rather than a bundled app path. The CLI reads the tray-selected active relay from `~/.hermes/desktop-control.json`, then falls back to the single stored session or an interactive picker. Bare `hermes-relay` resumes the active TUI tmux session stored in `~/.hermes/desktop-sessions.json`, falling back to `default`; `hermes-relay sessions list/resume/new/kill` is the explicit management path. Use `--remote ws://host:8767` only when you want to override the saved active relay for a single command. If the tray can only see its sidecar, Terminal shows a CLI install nudge and a copyable CLI-only installer command. The TUI tab hosts the embedded terminal: xterm.js renders inside the dashboard, Rust owns the local PTY, and the PTY runs the same `hermes-relay --session <name>` / `--new` path as an external terminal. Keep the external terminal button as the fallback when testing PTY focus, resize, or chord behavior. Terminal still shows shim state, session store, desktop config path, active route, daemon state, and whether experimental computer-use is enabled.
Open **Diagnostics** for local state checks: active relay, route, CLI shim availability, daemon status, desktop-tool consent, overlay visibility, blocklist count, session store, config path, and pending grants. **Run Doctor** executes the sidecar's `doctor --json` command and renders a redacted summary in the dashboard.
### Surface plugins
The tray **Plugins** view and `hermes-relay plugins` command expose installable terminal dashboard surfaces. The first built-in plugin is [Herm](https://github.com/liftaris/herm), packaged as `herm-tui`.
```powershell
hermes-relay plugins status herm
hermes-relay plugins install herm
hermes-relay plugins launch herm
hermes-relay plugins resume herm
```
Herm uses `bun add -g herm-tui` when Bun is available and falls back to `npm install -g herm-tui`; launch uses the installed `herm` binary or `bunx herm-tui` / `npx --yes herm-tui` when available. The tray can also embed Herm in the dashboard PTY, with `resume` mapped to `herm -c`.
### Pair + grant tools in one shot (CLI-only daemon bring-up)
If you plan to run `daemon` (headless tool serving), tack `--grant-tools` onto `pair` to capture the per-URL desktop-tool consent in the same step. That removes the historical `pair` → `shell` (consent prompt) → `daemon` dance:
@@ -110,7 +198,7 @@ If you plan to run `daemon` (headless tool serving), tack `--grant-tools` onto `
hermes-relay pair --remote ws://172.16.24.250:8767 --grant-tools
# ...prompts for code, then prompts for tool consent, stamps it on the stored session.
hermes-relay daemon --remote ws://172.16.24.250:8767
hermes-relay daemon
# ...starts headless; consent gate already satisfied.
```
@@ -130,6 +218,8 @@ hermes-relay [shell] Pipe the full Hermes CLI over a PTY (default
hermes-relay chat [<prompt>] Structured-event chat (REPL or one-shot, scriptable)
hermes-relay "<prompt>" One-shot structured chat (shortcut for chat "...")
hermes-relay pair [CODE] Pair with the relay and store a session token
hermes-relay plugins List/install/update/launch desktop surface plugins
hermes-relay sessions List / resume / create / kill TUI tmux sessions
hermes-relay status Show stored sessions + grants + TTL
hermes-relay tools List tools available on the server
hermes-relay devices List / revoke / extend server-side paired devices
@@ -144,16 +234,16 @@ hermes-relay --help Full help
$ hermes-relay
Connecting...
Connected via Tailscale (secure) — server 0.6.0
Attached (tmux session "shell-9f2a1c30") — re-attached to existing session.
Attached (tmux session "default") — re-attached to existing session.
Escape: Ctrl+A then . (detach, preserves tmux) · Ctrl+A then k (kill tmux) · Ctrl+A Ctrl+A (literal Ctrl+A)
Desktop tools: 5 handlers advertised (read_file, write_file, terminal, search_files, patch)
Desktop tools: 23 handlers advertised
[Axiom-Labs banner, Victor, "Hermes Agent v0.10.0 · claude-opus-4-7 ..."]
❯
```
- **`Ctrl+A .`** — detach cleanly; tmux session survives on the server, next `hermes-relay shell` re-attaches.
- **`Ctrl+A .`** — detach cleanly; tmux session survives on the server, next bare `hermes-relay` re-attaches.
- **`Ctrl+A k`** — kill the tmux session; next run gets a fresh hermes.
- **`Ctrl+A Ctrl+A`** — forward a literal `Ctrl+A` (for nested tmux).
- **Ctrl+C** passes through to `hermes` — interrupts the agent, not the client.
@@ -161,26 +251,33 @@ Desktop tools: 5 handlers advertised (read_file, write_file, terminal, search_fi
- **`--exec <cmd>`** — exec something else instead (e.g. `--exec btop`).
- **`--session <name>`** — override the tmux session name for deterministic resume.
### Multi-endpoint pairing (ADR 24)
If your Hermes server is reachable via multiple routes (LAN + Tailscale + a public URL), the QR payload `hermes-pair` produces carries all of them. Pass the raw payload string to `--pair-qr` and the CLI probes in priority order, picks the first reachable endpoint, and records which route it used — subsequent connects show `Connected via LAN (plain)` / `Connected via Tailscale (secure)` etc.
### Sessions — tmux continuity
```sh
# Paste the full QR payload (the string inside the QR code, not the URL):
hermes-relay pair --pair-qr '{"hermes":3,"host":"192.168.1.10","port":8642,"key":"ABC123",...}'
# Or via env:
HERMES_RELAY_PAIR_QR='<payload>' hermes-relay shell
hermes-relay sessions list
hermes-relay sessions resume default
hermes-relay sessions new
hermes-relay sessions kill default
```
Priority is strict — reachability only breaks ties within a priority tier. 4-second per-candidate timeout, 60-second reachability cache. Signature verification is TODO (matches Android).
The relay discovers background server-side tmux sessions named `hermes-*`, so `sessions list` shows sessions even when the current WebSocket did not create them. Bare `hermes-relay` is still the normal path: it resumes the active session recorded in `~/.hermes/desktop-sessions.json` and falls back to `default`. On reattach, the server captures recent tmux scrollback and replays it before live output so the reopened terminal has context immediately.
### Multi-endpoint pairing (ADR 24)
If your Hermes server is reachable via multiple routes (LAN + Tailscale + a public URL), the pairing invite encoded by the host QR carries all of them. Pass the printed `hermes-relay://pair?...` URL, raw JSON payload, or base64 payload to `--pair-qr` and the CLI probes in priority order, picks the first reachable endpoint, and records which route it used — subsequent connects show `Connected via LAN (plain)` / `Connected via Tailscale (secure)` etc.
```sh
# Paste the full pairing invite URL printed by hermes-pair:
hermes-relay pair --pair-qr 'hermes-relay://pair?payload=...'
# Or via env:
HERMES_RELAY_PAIR_QR='hermes-relay://pair?payload=...' hermes-relay shell
```
Priority is strict — reachability only breaks ties within a priority level. 4-second per-candidate timeout, 60-second reachability cache. Signature verification is TODO (matches Android).
### Local tool access for the agent
When you run `chat` or `shell` with tools consented, the CLI advertises five handlers to the remote Hermes agent so it can operate on YOUR machine (not the server):
- `desktop_read_file` / `desktop_write_file` / `desktop_patch` — file I/O in the CWD
- `desktop_terminal` — shell exec via `bash -lc` (30s timeout, SIGKILL on abort)
- `desktop_search_files` — ripgrep (with pure-Node fallback)
When you run `chat`, `shell`, or `daemon` with tools consented, the CLI advertises its desktop handlers to the remote Hermes agent so it can operate on YOUR machine (not the server). The regular surface covers file I/O, shell/PowerShell, process lookup, long-running jobs, transfers, clipboard, screenshots, and editor handoff. Computer-use tools are separate and experimental.
First connect per relay URL prompts:
@@ -198,6 +295,20 @@ You can also grant consent up front during pairing — `hermes-relay pair --gran
The server-side plugin (`plugin/tools/desktop_tool.py`) registers `desktop_*` tools with Hermes so the agent discovers them naturally; `check_fn` returns 503 when no client is connected so the LLM learns which tools it currently has.
### Experimental computer-use tools
`desktop_computer_status`, `desktop_computer_screenshot`, `desktop_computer_action`, `desktop_computer_grant_request`, and `desktop_computer_cancel` are registered server-side but the desktop client advertises and serves them only when explicitly enabled:
```sh
hermes-relay chat --experimental-computer-use
hermes-relay shell --experimental-computer-use
HERMES_RELAY_EXPERIMENTAL_COMPUTER_USE=1 hermes-relay daemon
```
They still require normal desktop-tool consent. Observe grants allow screenshots; assist/control grants require a visible local approval prompt before host input can run. The tray-managed daemon uses a local grant bridge: a pending assist/control grant opens the Grant Requests view, where it can be approved or rejected. Headless CLI daemon mode still fails closed unless an interactive local prompt provider is active.
Default computer-use policy blocks password managers, credential prompts, banking/payment/crypto surfaces, OS security/admin settings, and private-key/token material. `~/.hermes/desktop-control.json` lets operators tighten or extend that baseline.
### Devices — server-side session management
```sh
@@ -211,7 +322,7 @@ Talks to the relay's `GET/DELETE/PATCH /sessions` endpoints (same port as WSS, a
### Chat — REPL
```sh
hermes-relay --remote ws://172.16.24.250:8767
hermes-relay
```
```
@@ -235,7 +346,7 @@ Ctrl+C during a turn interrupts that turn (via `session.interrupt`). Ctrl+C at t
### Chat — one-shot
```sh
hermes-relay "summarize the last commit" --remote ws://172.16.24.250:8767
hermes-relay "summarize the last commit"
```
Stderr gets diagnostics; stdout gets the agent's reply — so `... > out.txt` captures just the answer.
@@ -259,7 +370,7 @@ hermes-relay --json "ls ~/" | jq -c '{type, name: .payload.name, text: .payload.
### Inspect tool access
```sh
hermes-relay tools --remote ws://172.16.24.250:8767
hermes-relay tools
```
```
@@ -289,6 +400,8 @@ Pass `--verbose` to list every tool inside each toolset.
| `--quiet, -q` | — | Suppress status lines and tool decorations |
| `--no-color` | `NO_COLOR` | Disable ANSI colors |
| `--non-interactive` | — | Never prompt; fail if credentials missing |
| `--experimental-computer-use` | `HERMES_RELAY_EXPERIMENTAL_COMPUTER_USE=1` | Advertise experimental `desktop_computer_*` tools after desktop-tool consent |
| `--no-computer-use` | — | Suppress computer-use advertisement even if env enables it |
Precedence for credentials: `--token` → `HERMES_RELAY_TOKEN` → `--code` → `HERMES_RELAY_CODE` → stored session → interactive prompt.
@@ -306,6 +419,7 @@ What's shipped in `desktop-v0.3.0-alpha.1`: remote chat + tool-event rendering,
What's next (see [ROADMAP.md](../ROADMAP.md#desktop-track) for the full track):
- Service installers — `install-service-{win,linux,mac}` to register the daemon with `sc.exe` / systemd user unit / `launchd` so it auto-starts on login.
- Optional Tauri v2 tray/overlay app — dark first-class pairing, one active desktop relay, visible status-only observing/control chip, task log, grant prompts, Devices/Revoke/Settings menu, and pause/emergency stop from the tray or hotkey on top of the CLI daemon.
- Multi-client server-side routing — today a connected desktop client is single-slot; allow laptop + home-desktop + work-box attached simultaneously with per-client tool dispatch via a new hermes-agent `ContextVar`.
- Code signing — Windows EV cert + Apple Developer ID + notarization to silence SmartScreen/Gatekeeper.
- npm registry publication — future v1.0 distribution work. Until then, use GitHub Release binaries or a local clone with `npm link`.
+237
View File
@@ -12,7 +12,10 @@
"hermes-relay": "bin/hermes-relay.js"
},
"devDependencies": {
"@tauri-apps/cli": "^2.11.2",
"@types/node": "^22.0.0",
"@xterm/addon-fit": "^0.11.0",
"@xterm/xterm": "^6.0.0",
"rimraf": "^5.0.0",
"tsx": "^4.19.0",
"typescript": "^5.7.0"
@@ -492,6 +495,223 @@
"node": ">=14"
}
},
"node_modules/@tauri-apps/cli": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli/-/cli-2.11.2.tgz",
"integrity": "sha512-bk3HemqvGRoy+5D/dVMUQHKMYLglD0jVnMm/0iGMH6ufZ+p8r14m6BpIixwij3PBvZdvORUp1YifTD8QxVZ1Nw==",
"dev": true,
"license": "Apache-2.0 OR MIT",
"bin": {
"tauri": "tauri.js"
},
"engines": {
"node": ">= 10"
},
"funding": {
"type": "opencollective",
"url": "https://opencollective.com/tauri"
},
"optionalDependencies": {
"@tauri-apps/cli-darwin-arm64": "2.11.2",
"@tauri-apps/cli-darwin-x64": "2.11.2",
"@tauri-apps/cli-linux-arm-gnueabihf": "2.11.2",
"@tauri-apps/cli-linux-arm64-gnu": "2.11.2",
"@tauri-apps/cli-linux-arm64-musl": "2.11.2",
"@tauri-apps/cli-linux-riscv64-gnu": "2.11.2",
"@tauri-apps/cli-linux-x64-gnu": "2.11.2",
"@tauri-apps/cli-linux-x64-musl": "2.11.2",
"@tauri-apps/cli-win32-arm64-msvc": "2.11.2",
"@tauri-apps/cli-win32-ia32-msvc": "2.11.2",
"@tauri-apps/cli-win32-x64-msvc": "2.11.2"
}
},
"node_modules/@tauri-apps/cli-darwin-arm64": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-darwin-arm64/-/cli-darwin-arm64-2.11.2.tgz",
"integrity": "sha512-+4UZzLt+eOAEQCwgd+TqKgyUJMrvx+BgdXLLaqJYmPqzP+nE6YZr/hY6CWLYGQb8jFn99jEkmC6uA3tNvamA1w==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-darwin-x64": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-darwin-x64/-/cli-darwin-x64-2.11.2.tgz",
"integrity": "sha512-VjYYtZUPqDMLutSfJEyxFE3Bz+DPi7c8wC3imckgvciLDZLq4qwKJxBicg0BXGhXjJsl8vKWgWRFNMPELQ+Xyg==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-linux-arm-gnueabihf": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-linux-arm-gnueabihf/-/cli-linux-arm-gnueabihf-2.11.2.tgz",
"integrity": "sha512-yMemD6f4i95AQriS8EazyOFzbE34yjnP16i3IOzpHGQvBoy2DjypFMFBq0NtPuITURv/cOGguRtHR5d79/9CSA==",
"cpu": [
"arm"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-linux-arm64-gnu": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-linux-arm64-gnu/-/cli-linux-arm64-gnu-2.11.2.tgz",
"integrity": "sha512-cgI91D2wL8GSgoWwZXDqt+DwnuZCP2/bz03QAE4TrhgAKIsrB4hX26W/H1EONPUUNkqrsgeCD0wU6pcNjV/5kw==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-linux-arm64-musl": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-linux-arm64-musl/-/cli-linux-arm64-musl-2.11.2.tgz",
"integrity": "sha512-X1rm0BERqAAggtYTESSgXrS3sz4Sb/OiPiz54UqISlXW+GkR3vNIGnsy/lejNmoXGVqri3Q53BCfQiclOIyRPw==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-linux-riscv64-gnu": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-linux-riscv64-gnu/-/cli-linux-riscv64-gnu-2.11.2.tgz",
"integrity": "sha512-usbMLJbT3KtkOrBMDVeGYNM35aTHXx38SJSzTMSqqjeUIOQ+iVPjb2yAGNAE+KqmBbAx4FOFIyMeKXx2M/JKGQ==",
"cpu": [
"riscv64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-linux-x64-gnu": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-linux-x64-gnu/-/cli-linux-x64-gnu-2.11.2.tgz",
"integrity": "sha512-Ru4gwJKPG0ctVGchRGpRup4Y4lW2SSfFnrbQcyHhCliKy4g8Qz97TrUgCur4CbWyAgKxvGh3SjrkA0LDYzDGiw==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-linux-x64-musl": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-linux-x64-musl/-/cli-linux-x64-musl-2.11.2.tgz",
"integrity": "sha512-eUm7T6clN1MMmNSRQ9gaWsQdyehQx2Gmn5hht/QUlqZQI/qcP2OJK5dnaxqwFzCr2HdsEo9ydxaqcS1oJzMvUw==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-win32-arm64-msvc": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-win32-arm64-msvc/-/cli-win32-arm64-msvc-2.11.2.tgz",
"integrity": "sha512-HeeZW80jU+gVTOEX4X/hC6NVSAdDVXajwP5fxIZ/3z9WvUC7qrudX2GMTilYq6Dg0e0sk0XgsAJD1hZ5wPBXUA==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-win32-ia32-msvc": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-win32-ia32-msvc/-/cli-win32-ia32-msvc-2.11.2.tgz",
"integrity": "sha512-YhjQNZcXfbkCLyazSv1nPnJ9iRFE1wm6kc51FDbU10/Dk09io+6PAGMLjkxnX2GdM0qMnDmTjstY8mTDVvtKeA==",
"cpu": [
"ia32"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@tauri-apps/cli-win32-x64-msvc": {
"version": "2.11.2",
"resolved": "https://registry.npmjs.org/@tauri-apps/cli-win32-x64-msvc/-/cli-win32-x64-msvc-2.11.2.tgz",
"integrity": "sha512-d2JchlFIpZevZVReyqhQOekJmb1UH3rhZ5VX6sH3ty9ETE0TKQavpihvoScUXfKKpW6HZC0MrFGRU0ZtD+w3gA==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 OR MIT",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">= 10"
}
},
"node_modules/@types/node": {
"version": "22.19.17",
"resolved": "https://registry.npmjs.org/@types/node/-/node-22.19.17.tgz",
@@ -502,6 +722,23 @@
"undici-types": "~6.21.0"
}
},
"node_modules/@xterm/addon-fit": {
"version": "0.11.0",
"resolved": "https://registry.npmjs.org/@xterm/addon-fit/-/addon-fit-0.11.0.tgz",
"integrity": "sha512-jYcgT6xtVYhnhgxh3QgYDnnNMYTcf8ElbxxFzX0IZo+vabQqSPAjC3c1wJrKB5E19VwQei89QCiZZP86DCPF7g==",
"dev": true,
"license": "MIT"
},
"node_modules/@xterm/xterm": {
"version": "6.0.0",
"resolved": "https://registry.npmjs.org/@xterm/xterm/-/xterm-6.0.0.tgz",
"integrity": "sha512-TQwDdQGtwwDt+2cgKDLn0IRaSxYu1tSUjgKarSDkUM0ZNiSRXFpjxEsvc/Zgc5kq5omJ+V0a8/kIM2WD3sMOYg==",
"dev": true,
"license": "MIT",
"workspaces": [
"addons/*"
]
},
"node_modules/ansi-regex": {
"version": "6.2.2",
"resolved": "https://registry.npmjs.org/ansi-regex/-/ansi-regex-6.2.2.tgz",
+11
View File
@@ -31,6 +31,14 @@
"build:bin:mac-x64": "npm run gen:version && bun build --compile --minify --sourcemap --target=bun-darwin-x64 src/cli.ts --outfile dist/bin/hermes-relay-darwin-x64",
"build:bin:mac-arm": "npm run gen:version && bun build --compile --minify --sourcemap --target=bun-darwin-arm64 src/cli.ts --outfile dist/bin/hermes-relay-darwin-arm64",
"build:sums": "cd dist/bin && sha256sum hermes-relay-* > SHA256SUMS.txt",
"pretray:dev": "node scripts/prepare-tray-sidecar.mjs",
"tray:dev": "tauri dev --config tray/src-tauri/tauri.conf.json",
"pretray:check": "node scripts/prepare-tray-sidecar.mjs --stub",
"tray:check": "cargo check --manifest-path tray/src-tauri/Cargo.toml",
"pretray:test": "node scripts/prepare-tray-sidecar.mjs --stub",
"tray:test": "cargo test --manifest-path tray/src-tauri/Cargo.toml",
"pretray:build": "node scripts/prepare-tray-sidecar.mjs",
"tray:build": "tauri build --config tray/src-tauri/tauri.conf.json",
"smoke": "npm run build:bin:win && node -e \"const{execFileSync}=require('child_process');const bin='./dist/bin/hermes-relay-win-x64.exe';const pkg=require('./package.json');for(const a of [['--version'],['--help'],['doctor'],['workspace']]){const out=execFileSync(bin,a,{encoding:'utf8'});if(!out||out.length<10)throw new Error('smoke FAIL: '+bin+' '+a.join(' ')+' produced no output');console.log('smoke OK: '+a.join(' ')+' ('+out.split('\\n')[0]+')')}const updOut=execFileSync(bin,['update','--check','--json'],{encoding:'utf8'});const parsed=JSON.parse(updOut);if(parsed.current!==pkg.version)throw new Error('smoke FAIL: update --check --json current='+parsed.current+' != package.json version='+pkg.version);console.log('smoke OK: update --check --json (current='+parsed.current+', up_to_date='+parsed.up_to_date+')')\"",
"prepublishOnly": "npm run build",
"dev": "tsx src/cli.ts",
@@ -65,7 +73,10 @@
],
"license": "MIT",
"devDependencies": {
"@tauri-apps/cli": "^2.11.2",
"@types/node": "^22.0.0",
"@xterm/addon-fit": "^0.11.0",
"@xterm/xterm": "^6.0.0",
"rimraf": "^5.0.0",
"tsx": "^4.19.0",
"typescript": "^5.7.0"
+88
View File
@@ -0,0 +1,88 @@
# Manage the optional `hermes` alias for hermes-relay on Windows.
#
# Status:
# irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/hermes-alias.ps1 | iex
#
# Enable:
# $env:HERMES_RELAY_HERMES_ALIAS='enable'; irm .../hermes-alias.ps1 | iex
#
# Disable:
# $env:HERMES_RELAY_HERMES_ALIAS='disable'; irm .../hermes-alias.ps1 | iex
#
# Override install dir:
# $env:HERMES_RELAY_INSTALL_DIR='C:\tools\hermes\bin'; irm .../hermes-alias.ps1 | iex
#Requires -Version 5.1
$ErrorActionPreference = 'Stop'
$dir = if ($env:HERMES_RELAY_INSTALL_DIR) { $env:HERMES_RELAY_INSTALL_DIR } else { Join-Path $HOME '.hermes\bin' }
$action = if ($args.Count -gt 0) { $args[0] } elseif ($env:HERMES_RELAY_HERMES_ALIAS) { $env:HERMES_RELAY_HERMES_ALIAS } else { 'status' }
$action = $action.ToLowerInvariant()
$target = Join-Path $dir 'hermes-relay.exe'
$shim = Join-Path $dir 'hermes.cmd'
function Say($msg) { Write-Host " $msg" }
function Die($msg) { Write-Host "hermes-alias.ps1: $msg" -ForegroundColor Red; exit 1 }
function Is-RelayShim {
param([string]$Path)
if (-not (Test-Path -LiteralPath $Path)) { return $false }
$content = Get-Content -LiteralPath $Path -Raw -ErrorAction SilentlyContinue
return $content -match 'hermes-relay\.exe'
}
function Show-Status {
Say "install dir : $dir"
Say "relay binary: $(if (Test-Path -LiteralPath $target) { $target } else { 'missing' })"
if (Test-Path -LiteralPath $shim) {
if (Is-RelayShim $shim) {
Say "hermes alias: enabled ($shim -> hermes-relay.exe)"
} else {
Say "hermes alias: present but not managed by hermes-relay ($shim)"
}
} else {
Say 'hermes alias: disabled'
}
$resolved = Get-Command hermes -ErrorAction SilentlyContinue
if ($resolved) {
Say "shell resolves: $($resolved.Source)"
} else {
Say 'shell resolves: not found in this process PATH'
}
}
switch ($action) {
{ $_ -in @('status', 'check', 'show') } {
Show-Status
}
{ $_ -in @('enable', 'on', 'true', 'yes', '1') } {
if (-not (Test-Path -LiteralPath $target)) {
Die "cannot enable alias because $target does not exist; install the CLI first"
}
if ((Test-Path -LiteralPath $shim) -and -not (Is-RelayShim $shim)) {
Die "refusing to overwrite existing non-hermes-relay shim at $shim"
}
New-Item -ItemType Directory -Force -Path $dir | Out-Null
$shimBody = @'
@echo off
"%~dp0hermes-relay.exe" %*
'@
Set-Content -LiteralPath $shim -Value $shimBody -Encoding ASCII -Force
Say "enabled hermes alias at $shim"
}
{ $_ -in @('disable', 'off', 'false', 'no', '0', 'remove') } {
if (Test-Path -LiteralPath $shim) {
if (Is-RelayShim $shim) {
Remove-Item -LiteralPath $shim -Force
Say "disabled hermes alias by removing $shim"
} else {
Say "left existing non-hermes-relay shim untouched at $shim"
}
} else {
Say 'hermes alias already disabled'
}
}
default {
Die "unknown action '$action' (use status, enable, or disable)"
}
}
+75
View File
@@ -0,0 +1,75 @@
#!/usr/bin/env sh
# Manage the optional `hermes` alias for hermes-relay on macOS/Linux.
#
# Status:
# curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/hermes-alias.sh | sh
#
# Enable:
# HERMES_RELAY_HERMES_ALIAS=enable curl -fsSL .../hermes-alias.sh | sh
#
# Disable:
# HERMES_RELAY_HERMES_ALIAS=disable curl -fsSL .../hermes-alias.sh | sh
set -eu
INSTALL_DIR="${HERMES_RELAY_INSTALL_DIR:-$HOME/.hermes/bin}"
ACTION="${1:-${HERMES_RELAY_HERMES_ALIAS:-status}}"
ACTION="$(printf '%s' "$ACTION" | tr '[:upper:]' '[:lower:]')"
TARGET="$INSTALL_DIR/hermes-relay"
ALIAS="$INSTALL_DIR/hermes"
say() { printf ' %s\n' "$*"; }
die() { printf 'hermes-alias.sh: %s\n' "$*" >&2; exit 1; }
is_relay_alias() {
[ -L "$ALIAS" ] && [ "$(readlink "$ALIAS")" = "hermes-relay" ]
}
show_status() {
say "install dir : $INSTALL_DIR"
if [ -f "$TARGET" ] || [ -L "$TARGET" ]; then
say "relay binary: $TARGET"
else
say "relay binary: missing"
fi
if is_relay_alias; then
say "hermes alias: enabled ($ALIAS -> hermes-relay)"
elif [ -e "$ALIAS" ] || [ -L "$ALIAS" ]; then
say "hermes alias: present but not managed by hermes-relay ($ALIAS)"
else
say "hermes alias: disabled"
fi
if command -v hermes >/dev/null 2>&1; then
say "shell resolves: $(command -v hermes)"
else
say "shell resolves: not found in this process PATH"
fi
}
case "$ACTION" in
status|check|show)
show_status
;;
enable|on|true|yes|1)
[ -f "$TARGET" ] || [ -L "$TARGET" ] || die "cannot enable alias because $TARGET does not exist; install the CLI first"
if [ -e "$ALIAS" ] || [ -L "$ALIAS" ]; then
is_relay_alias || die "refusing to overwrite existing non-hermes-relay command at $ALIAS"
fi
mkdir -p "$INSTALL_DIR"
ln -sf "hermes-relay" "$ALIAS"
say "enabled hermes alias at $ALIAS"
;;
disable|off|false|no|0|remove)
if is_relay_alias; then
rm -f "$ALIAS"
say "disabled hermes alias by removing $ALIAS"
elif [ -e "$ALIAS" ] || [ -L "$ALIAS" ]; then
say "left existing non-hermes-relay command untouched at $ALIAS"
else
say "hermes alias already disabled"
fi
;;
*)
die "unknown action '$ACTION' (use status, enable, or disable)"
;;
esac
+120 -26
View File
@@ -1,16 +1,22 @@
# hermes-relay desktop CLI installer — Windows (PowerShell 5.1+).
# hermes-relay desktop installer - Windows (PowerShell 5.1+).
#
# irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
#
# Downloads a prebuilt binary from GitHub Releases — no Node.js required.
# Downloads the tray app installer from GitHub Releases by default - no Node.js required.
# Install only the CLI binary instead:
# $env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm ... | iex
# Pin a specific release:
# $env:HERMES_RELAY_VERSION='desktop-v0.3.0-alpha.1'; irm ... | iex
# Override install dir:
# Override CLI install dir:
# $env:HERMES_RELAY_INSTALL_DIR='C:\tools\hermes\bin'; irm ... | iex
# Optional `hermes` alias for Orca/upstream-style workflows:
# $env:HERMES_RELAY_HERMES_ALIAS='enable'; $env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm ... | iex
# Disable an existing hermes-relay-owned alias without reinstalling:
# $env:HERMES_RELAY_HERMES_ALIAS='disable'; irm .../hermes-alias.ps1 | iex
#
# **Experimental phase** — binaries are unsigned. Windows SmartScreen may
# warn on first launch. This script documents the `Unblock-File` escape
# hatch on completion.
# **Experimental phase** - assets are unsigned. Windows SmartScreen may warn
# on first launch. The CLI branch documents the `Unblock-File` escape hatch on
# completion; the tray branch runs the downloaded NSIS installer.
#Requires -Version 5.1
$ErrorActionPreference = 'Stop'
@@ -18,6 +24,8 @@ $ErrorActionPreference = 'Stop'
$repo = if ($env:HERMES_RELAY_REPO) { $env:HERMES_RELAY_REPO } else { 'Codename-11/hermes-relay' }
$version = if ($env:HERMES_RELAY_VERSION) { $env:HERMES_RELAY_VERSION } else { 'latest' }
$dir = if ($env:HERMES_RELAY_INSTALL_DIR) { $env:HERMES_RELAY_INSTALL_DIR } else { Join-Path $HOME '.hermes\bin' }
$surface = if ($env:HERMES_RELAY_INSTALL_SURFACE) { $env:HERMES_RELAY_INSTALL_SURFACE.ToLowerInvariant() } else { 'tray' }
$aliasMode = if ($env:HERMES_RELAY_HERMES_ALIAS) { $env:HERMES_RELAY_HERMES_ALIAS.ToLowerInvariant() } else { 'skip' }
function Say($msg) { Write-Host " $msg" }
function Die($msg) { Write-Host "install.ps1: $msg" -ForegroundColor Red; exit 1 }
@@ -69,12 +77,35 @@ foreach ($cmd in 'Invoke-WebRequest','Get-FileHash') {
}
}
function Is-RelayShim {
param([string]$Path)
if (-not (Test-Path -LiteralPath $Path)) { return $false }
$content = Get-Content -LiteralPath $Path -Raw -ErrorAction SilentlyContinue
return $content -match 'hermes-relay\.exe'
}
function Alias-Enabled {
param([string]$Mode)
return $Mode -in @('enable', 'on', 'true', 'yes', '1')
}
function Alias-Disabled {
param([string]$Mode)
return $Mode -in @('disable', 'off', 'false', 'no', '0', 'remove')
}
if ($surface -ne 'tray' -and $surface -ne 'cli') {
Die "HERMES_RELAY_INSTALL_SURFACE must be 'tray' or 'cli'"
}
if ($aliasMode -notin @('skip', '', 'enable', 'on', 'true', 'yes', '1', 'disable', 'off', 'false', 'no', '0', 'remove')) {
Die "HERMES_RELAY_HERMES_ALIAS must be 'enable', 'disable', or unset"
}
if (-not [Environment]::Is64BitOperatingSystem) {
Die "32-bit Windows is not supported"
}
$arch = 'x64' # No ARM64 build yet; add hermes-relay-win-arm64 when cross-compile target lands.
$asset = "hermes-relay-win-$arch.exe"
$asset = if ($surface -eq 'tray') { "hermes-relay-desktop-windows-$arch-setup.exe" } else { "hermes-relay-win-$arch.exe" }
# Resolve "latest" to a concrete tag. GitHub's /releases/latest/download/ URL
# always skips prereleases, which breaks install during any all-alpha window.
@@ -125,13 +156,64 @@ if ($version -eq 'latest') {
}
$base = "https://github.com/$repo/releases/download/$resolvedVersion"
Say "Hermes-Relay desktop CLI installer"
Say "Hermes-Relay desktop installer"
Say " surface : $surface"
Say " platform : win-$arch"
Say " asset : $asset"
Say " version : $resolvedVersion"
Say " install : $dir"
if ($surface -eq 'cli') { Say " install : $dir" }
Say ""
if ($surface -eq 'tray') {
$tmp = New-Item -ItemType Directory -Path (Join-Path $env:TEMP ("hermes-relay-" + [Guid]::NewGuid()))
try {
Say '-> downloading tray installer...'
$installer = Join-Path $tmp $asset
try {
Invoke-WebRequest -UseBasicParsing "$base/$asset" -OutFile $installer
} catch {
Die "download failed: $base/$asset (maybe no Windows tray installer for this version yet?)"
}
Say '-> downloading checksums...'
try {
Invoke-WebRequest -UseBasicParsing "$base/SHA256SUMS.txt" -OutFile (Join-Path $tmp 'SHA256SUMS.txt')
} catch {
Die "could not fetch SHA256SUMS.txt (release incomplete?)"
}
Say '-> verifying SHA256...'
$expectedLine = (Select-String -Path (Join-Path $tmp 'SHA256SUMS.txt') -Pattern " $asset$").Line
if (-not $expectedLine) { Die "SHA256SUMS.txt has no entry for $asset" }
$expected = $expectedLine.Split(' ')[0].ToLower()
$actual = (Get-FileHash $installer -Algorithm SHA256).Hash.ToLower()
if ($expected -ne $actual) { Die "checksum mismatch (expected $expected, got $actual) - refusing to install" }
Say ' ok'
if (Get-Command Unblock-File -ErrorAction SilentlyContinue) {
Unblock-File $installer -ErrorAction SilentlyContinue
}
$installerArgs = @()
if ($env:HERMES_RELAY_TRAY_SILENT -eq '1') {
$installerArgs += '/S'
}
Say '-> launching installer...'
$proc = Start-Process -FilePath $installer -ArgumentList $installerArgs -Wait -PassThru
if ($proc.ExitCode -ne 0) {
Die "tray installer exited with code $($proc.ExitCode)"
}
} finally {
Remove-Item -Recurse -Force $tmp -ErrorAction SilentlyContinue
}
Say ''
Say 'Installed Hermes Relay Desktop. Launch it from the Start menu to pair, manage devices, view the task log, pause, or emergency-stop the daemon.'
Say 'For CLI-only installs, rerun with:'
Say " `$env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm https://raw.githubusercontent.com/$repo/main/desktop/scripts/install.ps1 | iex"
return
}
$tmp = New-Item -ItemType Directory -Path (Join-Path $env:TEMP ("hermes-relay-" + [Guid]::NewGuid()))
try {
# Pre-install: detect an existing binary so the user can see upgrade vs
@@ -185,28 +267,36 @@ try {
New-Item -ItemType Directory -Force -Path $dir | Out-Null
Copy-Item -Force (Join-Path $tmp $asset) $target
# Create a `hermes.cmd` alias so muscle memory from the upstream hermes-agent
# CLI (also called `hermes`) just works. Using a .cmd shim (not a symlink)
# because Windows symlinks require admin or Developer Mode. .cmd also works
# from any shell — cmd.exe, PowerShell, Git Bash, WSL interop — whereas a
# .ps1 shim would only fire from PowerShell.
# Optional `hermes.cmd` alias for Orca/upstream-style workflows. It is not
# created by default because it can shadow a real local hermes-agent install.
$hermesShim = Join-Path $dir 'hermes.cmd'
$existingShim = $null
if (Test-Path $hermesShim) {
$existingShim = Get-Content $hermesShim -Raw -ErrorAction SilentlyContinue
}
if (-not (Test-Path $hermesShim) -or ($existingShim -match 'hermes-relay\.exe')) {
# NB: the closing '@ of a PowerShell here-string MUST sit at column 0.
# Indenting it is a parse error, so we pull the here-string flush-left
# and scope-hide it in a subexpression.
if (Alias-Enabled $aliasMode) {
if ((Test-Path -LiteralPath $hermesShim) -and -not (Is-RelayShim $hermesShim)) {
Say "-> hermes.cmd already exists at $hermesShim (left untouched)"
} else {
# NB: the closing '@ of a PowerShell here-string MUST sit at column 0.
# Indenting it is a parse error, so we pull the here-string flush-left
# and scope-hide it in a subexpression.
$shimBody = @'
@echo off
"%~dp0hermes-relay.exe" %*
'@
Set-Content -Path $hermesShim -Value $shimBody -Encoding ASCII -Force
Say "-> created hermes.cmd -> hermes-relay.exe alias"
Set-Content -Path $hermesShim -Value $shimBody -Encoding ASCII -Force
Say "-> created hermes.cmd -> hermes-relay.exe alias"
}
} elseif (Alias-Disabled $aliasMode) {
if ((Test-Path -LiteralPath $hermesShim) -and (Is-RelayShim $hermesShim)) {
Remove-Item -LiteralPath $hermesShim -Force
Say "-> removed hermes.cmd alias"
} elseif (Test-Path -LiteralPath $hermesShim) {
Say "-> hermes.cmd exists but is not managed by hermes-relay (left untouched)"
} else {
Say "-> hermes.cmd alias already disabled"
}
} elseif ((Test-Path -LiteralPath $hermesShim) -and (Is-RelayShim $hermesShim)) {
Say "-> existing hermes.cmd alias left enabled (disable with HERMES_RELAY_HERMES_ALIAS='disable' and hermes-alias.ps1)"
} else {
Say "-> hermes.cmd already exists at $hermesShim (skipped alias creation)"
Say "-> skipped optional hermes.cmd alias (enable with HERMES_RELAY_HERMES_ALIAS='enable')"
}
# Post-install: confirm the NEW binary reports a sensible version. Don't
@@ -250,6 +340,10 @@ if ($installedVersion) {
} else {
Say 'Installed. Try:'
}
Say ' hermes-relay --help # (or the short alias: hermes --help)'
Say ' hermes-relay --help'
Say ' hermes-relay pair --remote ws://<host>:8767'
Say ''
Say 'Optional Orca/upstream-style alias:'
Say " `$env:HERMES_RELAY_HERMES_ALIAS='enable'; irm https://raw.githubusercontent.com/$repo/main/desktop/scripts/hermes-alias.ps1 | iex"
Say " `$env:HERMES_RELAY_HERMES_ALIAS='disable'; irm https://raw.githubusercontent.com/$repo/main/desktop/scripts/hermes-alias.ps1 | iex"
Say ''
+44 -15
View File
@@ -8,6 +8,8 @@
# HERMES_RELAY_VERSION=desktop-v0.3.0-alpha.1 curl -fsSL ... | sh
# Override install dir:
# HERMES_RELAY_INSTALL_DIR=/opt/hermes curl -fsSL ... | sh
# Optional `hermes` alias for Orca/upstream-style workflows:
# HERMES_RELAY_HERMES_ALIAS=enable curl -fsSL ... | sh
#
# **Experimental phase** — binaries are unsigned. macOS will quarantine;
# a `xattr -dr com.apple.quarantine <path>` one-liner is the escape hatch
@@ -18,6 +20,7 @@ set -eu
REPO="${HERMES_RELAY_REPO:-Codename-11/hermes-relay}"
VERSION="${HERMES_RELAY_VERSION:-latest}"
INSTALL_DIR="${HERMES_RELAY_INSTALL_DIR:-$HOME/.hermes/bin}"
HERMES_ALIAS_MODE="$(printf '%s' "${HERMES_RELAY_HERMES_ALIAS:-skip}" | tr '[:upper:]' '[:lower:]')"
say() { printf ' %s\n' "$*"; }
die() { printf 'install.sh: %s\n' "$*" >&2; exit 1; }
@@ -64,6 +67,11 @@ have curl || die "curl is required"
have uname || die "uname is required"
have install || die "install(1) is required"
case "$HERMES_ALIAS_MODE" in
skip|''|enable|on|true|yes|1|disable|off|false|no|0|remove) ;;
*) die "HERMES_RELAY_HERMES_ALIAS must be 'enable', 'disable', or unset" ;;
esac
# sha256sum on Linux, shasum on macOS — provide a shim.
if have sha256sum; then
sha_check() { grep " $1\$" SHA256SUMS.txt | sha256sum -c -; }
@@ -168,21 +176,38 @@ say " ok"
mkdir -p "$INSTALL_DIR"
install -m 0755 "$asset" "$target"
# Create a `hermes` alias next to `hermes-relay` so muscle memory from the
# upstream hermes-agent CLI (also called `hermes`) just works. Don't clobber
# a local hermes install: only create the symlink if nothing is there yet,
# or if an existing symlink already points at our binary. `-e` returns false
# for dangling symlinks — we want to overwrite those, so the `[ ! -e ]`
# branch will recreate when readlink resolves to a missing target.
# Optional `hermes` alias for Orca/upstream-style workflows. It is not created
# by default because it can shadow a real local hermes-agent install.
hermes_target="$INSTALL_DIR/hermes"
if [ ! -e "$hermes_target" ]; then
ln -sf "$(basename "$target")" "$hermes_target"
say "-> created hermes -> hermes-relay alias"
elif [ -L "$hermes_target" ] && [ "$(readlink "$hermes_target")" = "hermes-relay" ]; then
: # already points at us, no-op
else
say "-> hermes already exists at $hermes_target (skipped alias creation)"
fi
case "$HERMES_ALIAS_MODE" in
enable|on|true|yes|1)
if [ ! -e "$hermes_target" ] && [ ! -L "$hermes_target" ]; then
ln -sf "hermes-relay" "$hermes_target"
say "-> created hermes -> hermes-relay alias"
elif [ -L "$hermes_target" ] && [ "$(readlink "$hermes_target")" = "hermes-relay" ]; then
say "-> hermes alias already points at hermes-relay"
else
say "-> hermes already exists at $hermes_target (left untouched)"
fi
;;
disable|off|false|no|0|remove)
if [ -L "$hermes_target" ] && [ "$(readlink "$hermes_target")" = "hermes-relay" ]; then
rm -f "$hermes_target"
say "-> removed hermes alias"
elif [ -e "$hermes_target" ] || [ -L "$hermes_target" ]; then
say "-> hermes exists but is not managed by hermes-relay (left untouched)"
else
say "-> hermes alias already disabled"
fi
;;
*)
if [ -L "$hermes_target" ] && [ "$(readlink "$hermes_target")" = "hermes-relay" ]; then
say "-> existing hermes alias left enabled (disable with HERMES_RELAY_HERMES_ALIAS=disable and hermes-alias.sh)"
else
say "-> skipped optional hermes alias (enable with HERMES_RELAY_HERMES_ALIAS=enable)"
fi
;;
esac
# Post-install: confirm the NEW binary reports a sensible version. Don't
# fail the install on mismatch — the user may have pinned to a pre-release
@@ -236,6 +261,10 @@ if [ -n "$installed_version" ]; then
else
say "Installed. Try:"
fi
say " hermes-relay --help # (or the short alias: hermes --help)"
say " hermes-relay --help"
say " hermes-relay pair --remote ws://<host>:8767"
say ""
say "Optional Orca/upstream-style alias:"
say " HERMES_RELAY_HERMES_ALIAS=enable curl -fsSL https://raw.githubusercontent.com/$REPO/main/desktop/scripts/hermes-alias.sh | sh"
say " HERMES_RELAY_HERMES_ALIAS=disable curl -fsSL https://raw.githubusercontent.com/$REPO/main/desktop/scripts/hermes-alias.sh | sh"
say ""
+100
View File
@@ -0,0 +1,100 @@
#!/usr/bin/env node
import { copyFileSync, existsSync, mkdirSync, statSync, writeFileSync } from 'node:fs';
import { dirname, join, resolve } from 'node:path';
import { fileURLToPath } from 'node:url';
import { spawnSync } from 'node:child_process';
const desktopRoot = resolve(dirname(fileURLToPath(import.meta.url)), '..');
const vendorFiles = [
{
source: join('node_modules', '@xterm', 'xterm', 'lib', 'xterm.mjs'),
target: join('tray', 'ui', 'vendor', 'xterm', 'xterm.mjs')
},
{
source: join('node_modules', '@xterm', 'xterm', 'css', 'xterm.css'),
target: join('tray', 'ui', 'vendor', 'xterm', 'xterm.css')
},
{
source: join('node_modules', '@xterm', 'addon-fit', 'lib', 'addon-fit.mjs'),
target: join('tray', 'ui', 'vendor', 'xterm', 'addon-fit.mjs')
}
];
for (const file of vendorFiles) {
const source = join(desktopRoot, file.source);
if (!existsSync(source)) {
throw new Error(`missing tray vendor asset: ${source}. Run npm install in desktop/.`);
}
const targetPath = join(desktopRoot, file.target);
mkdirSync(dirname(targetPath), { recursive: true });
copyFileSync(source, targetPath);
}
const targets = {
'win32:x64': {
buildScript: 'build:bin:win',
source: join('dist', 'bin', 'hermes-relay-win-x64.exe'),
sidecar: join('tray', 'src-tauri', 'bin', 'hermes-relay-x86_64-pc-windows-msvc.exe')
},
'linux:x64': {
buildScript: 'build:bin:linux',
source: join('dist', 'bin', 'hermes-relay-linux-x64'),
sidecar: join('tray', 'src-tauri', 'bin', 'hermes-relay-x86_64-unknown-linux-gnu')
},
'darwin:x64': {
buildScript: 'build:bin:mac-x64',
source: join('dist', 'bin', 'hermes-relay-darwin-x64'),
sidecar: join('tray', 'src-tauri', 'bin', 'hermes-relay-x86_64-apple-darwin')
},
'darwin:arm64': {
buildScript: 'build:bin:mac-arm',
source: join('dist', 'bin', 'hermes-relay-darwin-arm64'),
sidecar: join('tray', 'src-tauri', 'bin', 'hermes-relay-aarch64-apple-darwin')
}
};
const target = targets[`${process.platform}:${process.arch}`];
if (!target) {
throw new Error(`unsupported tray sidecar platform: ${process.platform}/${process.arch}`);
}
const stubOnly = process.argv.includes('--stub');
const sidecar = join(desktopRoot, target.sidecar);
mkdirSync(dirname(sidecar), { recursive: true });
if (stubOnly) {
if (!existsSync(sidecar)) {
writeFileSync(sidecar, '');
console.log(`tray sidecar check stub ready: ${sidecar}`);
} else {
console.log(`tray sidecar already present: ${sidecar}`);
}
process.exit(0);
}
const npmExecPath = process.env.npm_execpath;
const npmCommand = npmExecPath ? process.execPath : (process.platform === 'win32' ? 'npm.cmd' : 'npm');
const npmArgs = npmExecPath ? [npmExecPath, 'run', target.buildScript] : ['run', target.buildScript];
const build = spawnSync(npmCommand, npmArgs, {
cwd: desktopRoot,
stdio: 'inherit'
});
if (build.error || build.status !== 0) {
const detail = build.error ? `: ${build.error.message}` : '';
throw new Error(`failed to build tray sidecar with npm run ${target.buildScript}${detail}`);
}
const source = join(desktopRoot, target.source);
if (!existsSync(source)) {
throw new Error(`expected sidecar source was not produced: ${source}`);
}
copyFileSync(source, sidecar);
if (process.platform !== 'win32') {
const mode = statSync(sidecar).mode | 0o755;
await import('node:fs').then(({ chmodSync }) => chmodSync(sidecar, mode));
}
console.log(`tray sidecar ready: ${sidecar}`);
+15 -3
View File
@@ -52,10 +52,22 @@ async function main() {
// ── Experimental computer-use advertisement and fail-closed action ───
const handlerSet = await import(pathToFileURL(path.join(distRoot, '..', 'handlerSet.js')).href)
const defaultAdvertised = handlerSet.advertisedDesktopTools()
const explicitlyDisabled = handlerSet.advertisedDesktopTools({ computerUse: false })
if (!defaultAdvertised.includes('desktop_computer_action')) {
throw new Error('computer-use tools should advertise with normal desktop tools')
const defaultHandlers = handlerSet.desktopHandlers({ computerUse: false })
const enabledHandlers = handlerSet.desktopHandlers({ computerUse: true })
const flagAdvertised = handlerSet.advertisedDesktopTools({ computerUse: true })
if (defaultAdvertised.includes('desktop_computer_action')) {
throw new Error('computer-use tools should not advertise without the experimental flag')
}
if ('desktop_computer_action' in defaultHandlers) {
throw new Error('computer-use handlers should not be served without the experimental flag')
}
if (!flagAdvertised.includes('desktop_computer_action')) {
throw new Error('computer-use tools should advertise when the experimental flag is enabled')
}
if (!('desktop_computer_action' in enabledHandlers)) {
throw new Error('computer-use handlers should be served when the experimental flag is enabled')
}
const explicitlyDisabled = handlerSet.advertisedDesktopTools({ computerUse: false })
if (explicitlyDisabled.includes('desktop_computer_action')) {
throw new Error('computer-use tools should be removable for explicit disable/test paths')
}
+47 -11
View File
@@ -4,15 +4,19 @@
// subcommands because a thin client has actual verbs (pair, status, tools).
import { chatCommand } from './commands/chat.js'
import { chatWorkerCommand } from './commands/chatWorker.js'
import { daemonCommand } from './commands/daemon.js'
import { devicesCommand } from './commands/devices.js'
import { doctorCommand } from './commands/doctor.js'
import { pairCommand } from './commands/pair.js'
import { pasteCommand } from './commands/paste.js'
import { pluginsCommand } from './commands/plugins.js'
import { sessionsCommand } from './commands/sessions.js'
import { shellCommand } from './commands/shell.js'
import { statusCommand } from './commands/status.js'
import { toolsCommand } from './commands/tools.js'
import { updateCommand } from './commands/update.js'
import { voiceCommand } from './commands/voice.js'
import { workspaceCommand } from './commands/workspace.js'
import { finalizePendingUpdate } from './updater.js'
import { VERSION } from './version.js'
@@ -56,6 +60,10 @@ const BOOLEAN_FLAGS = new Set([
'allow-tools',
'allow-computer-use',
'experimental-computer-use',
'no-computer-use',
'no-voice',
'no-tray',
'no-open',
'check',
'yes',
'new',
@@ -119,15 +127,19 @@ function parseArgs(argv: string[]): ParsedArgs {
const KNOWN_COMMANDS = new Set([
'chat',
'chat-worker',
'daemon',
'devices',
'doctor',
'paste',
'pair',
'sessions',
'shell',
'plugins',
'status',
'tools',
'update',
'voice',
'workspace',
'help'
])
@@ -140,22 +152,27 @@ Usage:
hermes-relay "<prompt>" One-shot structured chat (shortcut for chat "...")
hermes-relay pair [CODE] Pair with the relay and store a session token
hermes-relay paste Stage clipboard image for /paste in the TUI
hermes-relay plugins List/install/update/launch desktop surface plugins
hermes-relay sessions List / resume / create / kill TUI tmux sessions
hermes-relay status Show stored sessions + grants + TTL
hermes-relay tools List tools available on the server
hermes-relay devices List / revoke / extend server-side paired devices
hermes-relay daemon Run headless — expose desktop tools even when no shell is open
hermes-relay doctor Diagnostic report: version, paths, sessions, daemon status
hermes-relay update Check for and install the latest desktop-v* release
hermes-relay voice Show native Hermes voice config (STT/TTS/realtime providers)
hermes-relay voice mode Push-to-talk in a browser tab (proxied through this CLI)
hermes-relay workspace Print local workspace context (cwd, git, editor, shell) — --json for scripting
hermes-relay help Show this help
hermes-relay --version Print version and exit
Flags:
--remote <url> Relay WSS URL (env: HERMES_RELAY_URL)
--remote <url> Override saved active relay (env: HERMES_RELAY_URL)
--code <code> Pairing code (6 chars) (env: HERMES_RELAY_CODE)
--token <token> Session token (skips pairing) (env: HERMES_RELAY_TOKEN)
--pair-qr <payload> Full QR payload (multi-endpoint pairing, ADR 24;
probes endpoints and picks highest-priority reachable)
--pair-qr <payload> Full QR payload or hermes-relay://pair invite URL
(multi-endpoint pairing, ADR 24; probes endpoints
and picks highest-priority reachable)
(env: HERMES_RELAY_PAIR_QR)
--session <id> chat: resume session (legacy alias for --conversation);
shell: tmux session name (distinct — tmux, not hermes)
@@ -165,9 +182,12 @@ Flags:
--raw shell: skip auto-exec; drop into bare tmux/bash
--watch-editor shell/chat: poll tmux/$VSCODE and send active_editor hints every 5s
--no-tools chat/shell: disable local tool handlers (fs, exec, search)
Computer-use tools are experimental but ride with normal
desktop tools; host input still requires task grant plus
a visible local yes/no grant approval prompt.
--experimental-computer-use
chat/shell/daemon: advertise experimental desktop_computer_*
tools after normal desktop-tool consent. Host input still
requires task grant plus visible local approval.
Env: HERMES_RELAY_EXPERIMENTAL_COMPUTER_USE=1
--no-computer-use Disable computer-use advertisement even if env enabled.
--grant-tools pair: prompt for desktop-tool consent during pairing (TTY required;
lets you go straight from \`pair\` to \`daemon\` with no \`shell\` round-trip)
--auto-grant-tools pair: stamp tool consent without prompting — explicit non-interactive
@@ -188,11 +208,11 @@ Examples:
hermes-relay pair --remote ws://172.16.24.250:8767
# ...prompts for code, stores a token in ~/.hermes/remote-sessions.json
# REPL — reuses the stored token
hermes-relay --remote ws://172.16.24.250:8767
# REPL — reuses the tray-selected active relay or stored token
hermes-relay
# One-shot
hermes-relay "what files are in ~/.hermes?" --remote ws://172.16.24.250:8767
hermes-relay "what files are in ~/.hermes?"
# Pipe JSON events for scripting
hermes-relay --json "summarize the last commit" | jq -c '.type'
@@ -201,15 +221,23 @@ Examples:
hermes-relay tools --verbose
# Run the tool router headless so the agent can reach you without an open shell
hermes-relay daemon --remote ws://172.16.24.250:8767
hermes-relay daemon
# ...writes JSON-line lifecycle events to stderr; redirect or pipe to jq
# Inspect and resume server-side tmux TUI sessions
hermes-relay sessions list
hermes-relay sessions resume default
hermes-relay plugins install herm
hermes-relay plugins launch herm
# Two-command bring-up: pair with consent, then run headless. No \`shell\` round-trip.
hermes-relay pair --remote ws://172.16.24.250:8767 --grant-tools
hermes-relay daemon --remote ws://172.16.24.250:8767
hermes-relay daemon
Config files:
~/.hermes/remote-sessions.json session tokens (mode 0600)
~/.hermes/desktop-control.json tray-selected active relay
~/.hermes/desktop-sessions.json active TUI tmux session per relay
`
export async function main(argv = process.argv): Promise<number> {
@@ -252,6 +280,8 @@ export async function main(argv = process.argv): Promise<number> {
switch (args.command) {
case 'chat':
return chatCommand(args)
case 'chat-worker':
return chatWorkerCommand(args)
case 'daemon':
return daemonCommand(args)
case 'devices':
@@ -262,6 +292,10 @@ export async function main(argv = process.argv): Promise<number> {
return pairCommand(args)
case 'paste':
return pasteCommand(args)
case 'plugins':
return pluginsCommand(args)
case 'sessions':
return sessionsCommand(args)
case 'shell':
return shellCommand(args)
case 'status':
@@ -270,6 +304,8 @@ export async function main(argv = process.argv): Promise<number> {
return toolsCommand(args)
case 'update':
return updateCommand(args)
case 'voice':
return voiceCommand(args)
case 'workspace':
return workspaceCommand(args)
default:
+22 -7
View File
@@ -31,7 +31,11 @@ import { CliRenderer } from '../renderer.js'
import { fetchRecentSessions, pickSession } from '../sessionPicker.js'
import { ensureToolsConsent } from '../tools/consent.js'
import { configureComputerUseRuntime } from '../tools/computerGrants.js'
import { DESKTOP_HANDLERS, advertisedDesktopTools } from '../tools/handlerSet.js'
import {
advertisedDesktopTools,
desktopHandlers,
shouldAdvertiseComputerUse
} from '../tools/handlerSet.js'
import { DesktopToolRouter } from '../tools/router.js'
import { RelayTransport } from '../transport/RelayTransport.js'
@@ -400,21 +404,23 @@ export async function chatCommand(args: ParsedArgs): Promise<number> {
if (!toolsDisabled) {
const consent = await ensureToolsConsent(url)
if (consent.consented) {
const computerUseEnabled = shouldAdvertiseComputerUse(args.flags)
configureComputerUseRuntime({
url,
computerUseConsented: true,
computerUseConsented: computerUseEnabled,
consentSource: consent.source ?? 'stored'
})
const advertisedTools = advertisedDesktopTools()
const advertisedTools = advertisedDesktopTools({ computerUse: computerUseEnabled })
toolRouter = new DesktopToolRouter({
consentGranted: true,
handlers: DESKTOP_HANDLERS,
handlers: desktopHandlers({ computerUse: computerUseEnabled }),
advertisedTools: [...advertisedTools]
})
toolRouter.attach(relay)
process.stderr.write(
`Desktop tools: ${advertisedTools.length} handlers advertised (computer-use experimental; control requires grant approval)\n`
)
const computerUseNote = computerUseEnabled
? ' (computer-use experimental; control requires grant approval)'
: ''
process.stderr.write(`Desktop tools: ${advertisedTools.length} handlers advertised${computerUseNote}\n`)
} else if (consent.reason) {
process.stderr.write(`Desktop tools: disabled (${consent.reason})\n`)
}
@@ -499,6 +505,15 @@ export async function chatCommand(args: ParsedArgs): Promise<number> {
if (session.model) {
process.stderr.write(`Session ${session.sessionId.slice(0, 8)}… on ${session.model}\n`)
}
renderer.handle({
type: 'session.info',
session_id: session.sessionId,
payload: {
model: session.model ?? 'default',
skills: {},
tools: {}
}
})
// Mode detection — one-shot vs piped vs REPL.
+349
View File
@@ -0,0 +1,349 @@
import type { ParsedArgs } from '../cli.js'
import { rpcErrorMessage } from '../lib/rpc.js'
type JsonRecord = Record<string, unknown>
interface DirectApiChatOptions {
baseUrl: string
apiKey: string | null
prompt: string
sessionId: string | null
fresh: boolean
}
interface SseMessage {
event: string | null
data: string
}
function flag(args: ParsedArgs, name: string): string | null {
const value = args.flags[name]
return typeof value === 'string' ? value : null
}
function normalizeBaseUrl(raw: string): string {
const trimmed = raw.trim()
if (!trimmed) {
throw new Error('gateway URL is required')
}
const withScheme = /^[a-z][a-z0-9+.-]*:\/\//i.test(trimmed) ? trimmed : `http://${trimmed}`
const parsed = new URL(withScheme)
parsed.pathname = parsed.pathname.replace(/\/+$/, '')
parsed.search = ''
parsed.hash = ''
return parsed.toString().replace(/\/+$/, '')
}
function headers(apiKey: string | null, accept = 'application/json'): Record<string, string> {
const out: Record<string, string> = {
Accept: accept,
'Content-Type': 'application/json'
}
if (apiKey?.trim()) {
out.Authorization = `Bearer ${apiKey.trim()}`
}
return out
}
function asRecord(value: unknown): JsonRecord {
return value && typeof value === 'object' && !Array.isArray(value) ? (value as JsonRecord) : {}
}
function stringField(value: unknown): string | null {
return typeof value === 'string' && value.trim() ? value.trim() : null
}
function pickSessionId(value: unknown): string | null {
const root = asRecord(value)
const session = asRecord(root.session)
return (
stringField(root.session_id) ??
stringField(root.id) ??
stringField(session.id) ??
stringField(session.session_id)
)
}
function emit(type: string, payload: JsonRecord = {}, sessionId?: string | null): void {
const event: JsonRecord = { type, payload }
if (sessionId) {
event.session_id = sessionId
}
process.stdout.write(`${JSON.stringify(event)}\n`)
}
async function readJsonResponse(response: Response): Promise<unknown> {
const text = await response.text()
if (!text.trim()) {
return {}
}
return JSON.parse(text) as unknown
}
async function routeExists(baseUrl: string, apiKey: string | null, path: string): Promise<boolean> {
try {
const response = await fetch(`${baseUrl}${path}`, {
method: 'HEAD',
headers: headers(apiKey)
})
return response.status === 200 || response.status === 401 || response.status === 403 || response.status === 405
} catch {
return false
}
}
async function createSession(baseUrl: string, apiKey: string | null): Promise<string> {
const response = await fetch(`${baseUrl}/api/sessions`, {
method: 'POST',
headers: headers(apiKey),
body: JSON.stringify({ title: 'Hermes Relay Desktop Chat' })
})
if (!response.ok) {
throw new Error(`create session failed: HTTP ${response.status} ${response.statusText}`)
}
const id = pickSessionId(await readJsonResponse(response))
if (!id) {
throw new Error('create session response did not include a session id')
}
return id
}
async function* sseMessages(response: Response): AsyncGenerator<SseMessage> {
if (!response.body) {
throw new Error('stream response has no body')
}
const reader = response.body.getReader()
const decoder = new TextDecoder()
let buffer = ''
let event: string | null = null
let data: string[] = []
const flush = function* (): Generator<SseMessage> {
if (data.length > 0) {
yield { event, data: data.join('\n') }
}
event = null
data = []
}
for (;;) {
const { done, value } = await reader.read()
buffer += decoder.decode(value ?? new Uint8Array(), { stream: !done })
let newline = buffer.indexOf('\n')
while (newline >= 0) {
const rawLine = buffer.slice(0, newline)
buffer = buffer.slice(newline + 1)
const line = rawLine.endsWith('\r') ? rawLine.slice(0, -1) : rawLine
if (!line) {
yield* flush()
} else if (line.startsWith('event:')) {
event = line.slice('event:'.length).trim() || null
} else if (line.startsWith('data:')) {
data.push(line.slice('data:'.length).trimStart())
}
newline = buffer.indexOf('\n')
}
if (done) {
break
}
}
if (buffer.trim()) {
data.push(buffer.trim())
}
yield* flush()
}
function eventText(record: JsonRecord): string | null {
const message = asRecord(record.message)
return (
stringField(record.delta) ??
stringField(record.text) ??
stringField(record.content) ??
stringField(record.thinking_delta) ??
stringField(record.thinking) ??
stringField(message.content)
)
}
function eventToolName(record: JsonRecord): string {
return (
stringField(record.tool_name) ??
stringField(record.tool) ??
stringField(record.name) ??
stringField(record.call_id) ??
stringField(record.tool_call_id) ??
'tool'
)
}
function mapSseEvent(kind: string, record: JsonRecord, sessionId: string | null): void {
const sid = pickSessionId(record) ?? sessionId
if (sid) {
emit('session.info', { id: sid }, sid)
}
switch (kind) {
case 'response.output_text.delta':
case 'message.delta':
case 'assistant.delta':
case 'content_delta':
case 'delta': {
const text = eventText(record)
if (text) {
emit('message.delta', { text }, sid)
}
return
}
case 'tool.progress':
case 'thinking_delta':
case 'reasoning_delta':
case 'reasoning.available': {
const text = eventText(record)
if (text) {
emit('reasoning.delta', { text }, sid)
}
return
}
case 'tool.pending':
case 'tool.started':
case 'tool_start':
case 'tool_started': {
const name = eventToolName(record)
emit('tool.start', { tool_id: name, name }, sid)
return
}
case 'tool.completed':
case 'tool_result':
case 'tool_completed': {
const name = eventToolName(record)
const error = stringField(record.error) ?? undefined
const summary = stringField(record.result_preview) ?? stringField(record.summary) ?? stringField(record.message)
emit('tool.complete', { tool_id: name, name, error, summary }, sid)
return
}
case 'response.created':
case 'run.started':
case 'session.created':
case 'message.started':
return
case 'assistant.completed':
case 'response.completed':
case 'run.completed':
case 'content_complete':
case 'complete':
case 'completed':
case 'done':
emit('message.complete', {}, sid)
return
case 'error':
case 'run.failed':
case 'tool.failed':
emit('error', { message: stringField(record.error) ?? stringField(record.message) ?? 'gateway stream error' }, sid)
return
default:
emit('status.update', { text: kind }, sid)
return
}
}
async function streamSseResponse(response: Response, sessionId: string | null): Promise<void> {
if (!response.ok) {
const body = await response.text().catch(() => '')
throw new Error(`chat stream failed: HTTP ${response.status} ${response.statusText}${body ? ` - ${body.slice(0, 240)}` : ''}`)
}
let completed = false
for await (const message of sseMessages(response)) {
if (message.data === '[DONE]') {
completed = true
emit('message.complete', {}, sessionId)
continue
}
let record: JsonRecord
try {
record = asRecord(JSON.parse(message.data) as unknown)
} catch {
record = { text: message.data }
}
const kind = message.event ?? stringField(record.type) ?? stringField(record.event) ?? 'message.delta'
if (kind === 'done' || kind === 'response.completed' || kind === 'run.completed' || kind === 'message.complete') {
completed = true
}
mapSseEvent(kind, record, sessionId)
}
if (!completed) {
emit('message.complete', {}, sessionId)
}
}
async function streamSessionChat(opts: DirectApiChatOptions, sessionId: string): Promise<void> {
emit('session.info', { id: sessionId, source: 'gateway_sessions' }, sessionId)
const response = await fetch(`${opts.baseUrl}/api/sessions/${encodeURIComponent(sessionId)}/chat/stream`, {
method: 'POST',
headers: headers(opts.apiKey, 'text/event-stream'),
body: JSON.stringify({ message: opts.prompt })
})
await streamSseResponse(response, sessionId)
}
async function streamRunChat(opts: DirectApiChatOptions): Promise<void> {
emit('session.info', { id: opts.sessionId ?? 'runs', source: 'gateway_runs' }, opts.sessionId)
const body: JsonRecord = {
model: 'default',
input: opts.prompt,
stream: true
}
const response = await fetch(`${opts.baseUrl}/v1/runs`, {
method: 'POST',
headers: headers(opts.apiKey, 'text/event-stream'),
body: JSON.stringify(body)
})
await streamSseResponse(response, opts.sessionId)
}
async function runDirectApiChat(opts: DirectApiChatOptions): Promise<void> {
const sessionsChat = await routeExists(opts.baseUrl, opts.apiKey, '/api/sessions/probe/chat/stream')
if (sessionsChat) {
const sessionId = opts.fresh || !opts.sessionId ? await createSession(opts.baseUrl, opts.apiKey) : opts.sessionId
await streamSessionChat(opts, sessionId)
return
}
const runs = await routeExists(opts.baseUrl, opts.apiKey, '/v1/runs')
if (runs) {
await streamRunChat(opts)
return
}
throw new Error('gateway is reachable but no supported streaming chat endpoint was found (/api/sessions/*/chat/stream or /v1/runs)')
}
export async function chatWorkerCommand(args: ParsedArgs): Promise<number> {
const mode = args.positional.shift() ?? ''
if (mode !== 'api') {
process.stderr.write('usage: hermes-relay chat-worker api --gateway-url <url> [--session <id>] [--new] <prompt>\n')
return 2
}
const gatewayUrl = flag(args, 'gateway-url') ?? process.env.HERMES_RELAY_GATEWAY_URL ?? ''
const prompt = args.positional.join(' ').trim()
if (!prompt) {
process.stderr.write('error: prompt is required\n')
return 2
}
try {
await runDirectApiChat({
baseUrl: normalizeBaseUrl(gatewayUrl),
apiKey: process.env.HERMES_RELAY_GATEWAY_API_KEY ?? null,
prompt,
sessionId: flag(args, 'session'),
fresh: !!args.flags.new
})
return 0
} catch (error) {
const message = rpcErrorMessage(error)
emit('error', { message })
process.stderr.write(`error: ${message}\n`)
return 1
}
}
+132 -8
View File
@@ -31,18 +31,28 @@
// after the shell detaches; see roadmap for pause-while-interactive).
// - --log-file <path>: for now, redirect stderr if you need a file.
import { promises as fs } from 'node:fs'
import * as os from 'node:os'
import * as path from 'node:path'
import type { ParsedArgs } from '../cli.js'
import { rpcErrorMessage } from '../lib/rpc.js'
import { GatewayClient } from '../gatewayClient.js'
import type { GatewayEvent, SessionCreateResponse } from '../gatewayTypes.js'
import { rpcErrorMessage, asRpcResult } from '../lib/rpc.js'
import { resolveFirstRunUrl } from '../relayUrlPrompt.js'
import { getSession } from '../remoteSessions.js'
import {
DESKTOP_HANDLERS,
advertisedDesktopTools
advertisedDesktopTools,
desktopHandlers,
shouldAdvertiseComputerUse
} from '../tools/handlerSet.js'
import { configureComputerUseRuntime } from '../tools/computerGrants.js'
import { DesktopToolRouter } from '../tools/router.js'
import { RelayTransport } from '../transport/RelayTransport.js'
import { setupGracefulExit } from '../lib/gracefulExit.js'
import { startVoiceServer, type VoiceServer } from '../voiceServer.js'
const VOICE_DISCOVERY_FILE = 'desktop-voice.json'
type LogLevel = 'info' | 'warn' | 'error'
@@ -231,16 +241,17 @@ export async function daemonCommand(args: ParsedArgs): Promise<number> {
// or redirected daemon still fails host input closed because no visible
// local grant approval prompt can run.
const interactive = !!process.stdin.isTTY && !!process.stderr.isTTY
const computerUseEnabled = shouldAdvertiseComputerUse(args.flags)
configureComputerUseRuntime({
url,
computerUseConsented: true,
computerUseConsented: computerUseEnabled,
consentSource: consented ? 'stored' : 'override'
})
const advertisedTools = advertisedDesktopTools()
const advertisedTools = advertisedDesktopTools({ computerUse: computerUseEnabled })
const router = new DesktopToolRouter({
consentGranted: true,
interactive,
handlers: DESKTOP_HANDLERS,
handlers: desktopHandlers({ computerUse: computerUseEnabled }),
advertisedTools: [...advertisedTools]
})
router.attach(relay)
@@ -248,14 +259,66 @@ export async function daemonCommand(args: ParsedArgs): Promise<number> {
log.info({
event: 'ready',
advertised_tools: [...advertisedTools],
experimental_computer_use: true,
experimental_computer_use: computerUseEnabled,
interactive
})
// ── Voice server ──────────────────────────────────────────────────
// Hosts the same loopback HTTP voice surface that `voice mode` starts
// ad-hoc, but kept alive for the whole daemon lifetime. The tray reads
// ~/.hermes/desktop-voice.json to find the URL. Failures here are
// non-fatal — the tool router is the daemon's primary job, voice is
// a bonus.
const noVoice = !!args.flags['no-voice']
let voiceServer: VoiceServer | null = null
let voiceSessionId: string | null = null
if (!noVoice) {
const gateway = new GatewayClient(relay)
try {
// Attach the gateway.ready listener BEFORE drain — drain replays
// events buffered since the transport started, and gateway.ready
// already arrived (the router attached above is silent on it).
const ready = waitForGatewayReady(gateway, 30_000)
gateway.start()
gateway.drain()
await ready
voiceSessionId = await createVoiceSession(gateway)
voiceServer = await startVoiceServer({
token,
relayUrl: url,
gateway,
sessionId: voiceSessionId
})
await writeVoiceDiscovery(voiceServer.url, voiceSessionId, log)
log.info({
event: 'voice_ready',
url: voiceServer.url,
session_id: voiceSessionId.slice(0, 8)
})
} catch (e) {
log.warn({
event: 'voice_unavailable',
message: rpcErrorMessage(e)
})
voiceServer = null
}
}
// Graceful shutdown: detach router (stops heartbeats), kill transport
// (closes the WSS), then let setupGracefulExit's failsafe exit us.
const cleanup = () => {
const cleanup = async () => {
log.info({ event: 'shutdown' })
try {
if (voiceServer) await voiceServer.close()
} catch {
/* ignore */
}
try {
await removeVoiceDiscovery()
} catch {
/* ignore */
}
try {
router.detach()
} catch {
@@ -284,3 +347,64 @@ export default daemonCommand
// Small utility function re-exported for tests that need to stub the logger.
export type { LogFields }
export { makeLogger as __makeLoggerForTests, rpcErrorMessage as __rpcErrorMessageForTests }
// ── Voice-server helpers ───────────────────────────────────────────────
function voiceDiscoveryPath(): string {
return path.join(os.homedir(), '.hermes', VOICE_DISCOVERY_FILE)
}
async function writeVoiceDiscovery(
url: string,
sessionId: string,
log: { info: (f: LogFields) => void; warn: (f: LogFields) => void; error: (f: LogFields) => void }
): Promise<void> {
const payload = {
url,
pid: process.pid,
session_id: sessionId,
started_at: Math.floor(Date.now() / 1000)
}
const filePath = voiceDiscoveryPath()
try {
await fs.mkdir(path.dirname(filePath), { recursive: true })
await fs.writeFile(filePath, JSON.stringify(payload, null, 2) + '\n', { mode: 0o600 })
} catch (e) {
log.warn({ event: 'voice_discovery_write_failed', message: rpcErrorMessage(e), path: filePath })
}
}
async function removeVoiceDiscovery(): Promise<void> {
const filePath = voiceDiscoveryPath()
try {
await fs.unlink(filePath)
} catch (e) {
// ENOENT is fine; nothing else should bubble up — cleanup is best-effort.
if ((e as NodeJS.ErrnoException)?.code !== 'ENOENT') throw e
}
}
function waitForGatewayReady(gateway: GatewayClient, timeoutMs: number): Promise<void> {
return new Promise((resolve, reject) => {
const timer = setTimeout(() => {
gateway.off('event', handler)
reject(new Error(`gateway.ready timeout after ${timeoutMs}ms`))
}, timeoutMs)
timer.unref?.()
const handler = (ev: GatewayEvent) => {
if (ev.type === 'gateway.ready') {
clearTimeout(timer)
gateway.off('event', handler)
resolve()
}
}
gateway.on('event', handler)
})
}
async function createVoiceSession(gateway: GatewayClient): Promise<string> {
const raw = await gateway.request<SessionCreateResponse>('session.create', { cols: 80 })
const r = asRpcResult<SessionCreateResponse>(raw)
if (!r?.session_id) throw new Error('voice session.create returned no session_id')
return r.session_id
}
+7 -2
View File
@@ -17,11 +17,13 @@
// hermes-relay devices revoke <prefix> DELETE — destroys the token
// hermes-relay devices extend <prefix> [--ttl <seconds>] PATCH — defaults to 24h
//
// If no `--remote` is passed we default to the single stored relay (if
// exactly one exists). Otherwise --remote is required.
// If no `--remote` is passed we default to the tray-selected active relay,
// then the single stored relay if exactly one exists. Otherwise --remote is
// required.
import { humanExpiry } from '../banner.js'
import type { ParsedArgs } from '../cli.js'
import { getActiveDesktopRelayUrl } from '../desktopConfig.js'
import { getSession, listSessions } from '../remoteSessions.js'
const DEFAULT_EXTEND_TTL_SECONDS = 24 * 3600
@@ -77,10 +79,13 @@ async function resolveRemoteAndToken(
const stored = await listSessions()
const urls = Object.keys(stored)
const activeDesktopUrl = await getActiveDesktopRelayUrl()
let url: string
if (argUrl || envUrl) {
url = argUrl ?? envUrl!
} else if (activeDesktopUrl) {
url = activeDesktopUrl
} else if (urls.length === 1) {
url = urls[0]!
} else if (urls.length === 0) {
+4
View File
@@ -20,11 +20,15 @@ import { fileURLToPath } from 'node:url'
import { humanExpiry } from '../banner.js'
import type { ParsedArgs } from '../cli.js'
import { listSessions } from '../remoteSessions.js'
import { VERSION } from '../version.js'
import { detectWorkspaceContext, type WorkspaceContext } from '../workspaceContext.js'
const __dirname = dirname(fileURLToPath(import.meta.url))
function readVersion(): string {
if (VERSION) {
return VERSION
}
// Same trick as cli.ts — dist/commands/doctor.js → dist → pkg root.
try {
const pkgPath = join(__dirname, '..', '..', 'package.json')
+13 -5
View File
@@ -17,7 +17,11 @@ import {
promptForPairingCode,
validatePairingPayloadString
} from '../pairing.js'
import { payloadToCandidates, probeCandidatesByPriority } from '../pairingQr.js'
import {
payloadToRelayCandidates,
probeCandidatesByPriority,
relayPairingCodeFromPayload
} from '../pairingQr.js'
import { resolveFirstRunUrl } from '../relayUrlPrompt.js'
import { saveSession } from '../remoteSessions.js'
import { ensureToolsConsent } from '../tools/consent.js'
@@ -49,9 +53,13 @@ async function resolvePairTarget(args: ParsedArgs): Promise<PairTarget | { error
if (!validated.ok) {
return { error: `invalid --pair-qr payload: ${validated.reason}` }
}
const candidates = payloadToCandidates(validated.payload)
if (candidates.length === 0) {
return { error: 'pairing payload had no endpoints to probe' }
let candidates
let pairingCode
try {
candidates = payloadToRelayCandidates(validated.payload)
pairingCode = relayPairingCodeFromPayload(validated.payload)
} catch (e) {
return { error: e instanceof Error ? e.message : String(e) }
}
process.stderr.write(`Probing ${candidates.length} endpoint(s)...\n`)
let winner
@@ -65,7 +73,7 @@ async function resolvePairTarget(args: ParsedArgs): Promise<PairTarget | { error
)
return {
url: winner.relay.url,
code: validated.payload.key.toUpperCase(),
code: pairingCode,
endpointRole: winner.role
}
}
+4
View File
@@ -17,6 +17,7 @@
import type { ParsedArgs } from '../cli.js'
import { captureClipboardImage } from '../chatAttach.js'
import { getActiveDesktopRelayUrl } from '../desktopConfig.js'
import { getSession, listSessions } from '../remoteSessions.js'
function wsToHttp(url: string): string {
@@ -42,9 +43,12 @@ async function resolveRemoteAndToken(args: ParsedArgs): Promise<{ url: string; t
const stored = await listSessions()
const urls = Object.keys(stored)
const activeDesktopUrl = await getActiveDesktopRelayUrl()
let url: string
if (argUrl || envUrl) {
url = argUrl ?? envUrl!
} else if (activeDesktopUrl) {
url = activeDesktopUrl
} else if (urls.length === 1) {
url = urls[0]!
} else if (urls.length === 0) {
+111
View File
@@ -0,0 +1,111 @@
import type { ParsedArgs } from '../cli.js'
import {
getSurfacePlugin,
listSurfacePluginStatuses,
resolvePluginPlan,
runPluginPlan,
surfacePluginStatus,
type SurfacePluginStatus
} from '../surfacePlugins.js'
function renderStatus(status: SurfacePluginStatus): string {
const plugin = status.descriptor
const lines = [
`${plugin.id} - ${plugin.name}`,
` state: ${status.installed ? 'installed' : status.available ? 'fallback available' : 'missing'}`,
` package: ${plugin.packageName}`,
` command: ${status.installed ? status.command : (status.launch?.display ?? status.command)}`,
` source: ${plugin.sourceUrl}`,
` setup: ${status.setupHint}`
]
if (status.version) {
lines.push(` version: ${status.version}`)
}
if (status.installer) {
lines.push(` install: ${status.installer.display}`)
}
if (status.fallback && !status.installed) {
lines.push(` fallback: ${status.fallback.display}`)
}
lines.push(` tabs: ${plugin.tabs.join(', ')}`)
lines.push(` actions: ${plugin.sessionActions.map((action) => action.label).join(', ')}`)
return lines.join('\n')
}
function resolvePluginOrPrint(id: string): ReturnType<typeof getSurfacePlugin> {
const plugin = getSurfacePlugin(id)
if (!plugin) {
process.stderr.write(`plugins: unknown plugin "${id}". Try: herm\n`)
}
return plugin
}
export async function pluginsCommand(args: ParsedArgs): Promise<number> {
const sub = args.positional[0] ?? 'list'
const id = args.positional[1] ?? 'herm'
const wantJson = !!args.flags.json
if (sub === 'list' || sub === 'status') {
const statuses =
sub === 'status' && args.positional[1]
? (() => {
const plugin = resolvePluginOrPrint(id)
return plugin ? [surfacePluginStatus(plugin)] : null
})()
: listSurfacePluginStatuses()
if (!statuses) {
return 2
}
if (wantJson) {
process.stdout.write(JSON.stringify(statuses, null, 2) + '\n')
return 0
}
process.stdout.write('Desktop surface plugins:\n\n')
process.stdout.write(statuses.map(renderStatus).join('\n\n') + '\n')
return 0
}
if (sub === 'install' || sub === 'update' || sub === 'launch' || sub === 'resume') {
const plugin = resolvePluginOrPrint(id)
if (!plugin) {
return 2
}
const plan = resolvePluginPlan(plugin, sub)
if (!plan) {
const message =
sub === 'install' || sub === 'update'
? `plugins: install Bun or npm first, then rerun \`hermes-relay plugins ${sub} ${plugin.id}\`.\n`
: `plugins: ${plugin.name} is not installed and no bunx/npx fallback is available.\n`
if (wantJson) {
process.stdout.write(
JSON.stringify({ plugin: plugin.id, action: sub, ok: false, error: message.trim() }, null, 2) +
'\n'
)
} else {
process.stderr.write(message)
}
return 1
}
if (wantJson && (sub === 'launch' || sub === 'resume')) {
process.stdout.write(
JSON.stringify({ plugin: plugin.id, action: sub, ok: true, command: plan.display }, null, 2) +
'\n'
)
return 0
}
if (!wantJson) {
process.stderr.write(`plugins: ${sub} ${plugin.name} via ${plan.display}\n`)
}
const code = await runPluginPlan(plan)
if (wantJson) {
process.stdout.write(
JSON.stringify({ plugin: plugin.id, action: sub, ok: code === 0, code, command: plan.display }, null, 2) +
'\n'
)
}
return code
}
process.stderr.write('unknown plugins sub-verb. Try: list | status [id] | install <id> | update <id> | launch <id> | resume <id>\n')
return 2
}
+218
View File
@@ -0,0 +1,218 @@
import type { ParsedArgs } from '../cli.js'
import {
clearActiveTerminalSession,
getActiveTerminalSession
} from '../terminalSessionStore.js'
import { connectAndAuth, shellCommand } from './shell.js'
const COMMAND_TIMEOUT_MS = 15_000
interface TerminalSessionInfo {
name: string
tmux_name?: string
pid?: number
shell?: string
attached?: number
windows?: number
created_at?: number
live?: boolean
owned_by_client?: boolean
}
function asTerminalSession(raw: unknown): TerminalSessionInfo | null {
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
return null
}
const item = raw as Record<string, unknown>
const name = typeof item.name === 'string' ? item.name : ''
if (!name) {
return null
}
return {
name,
tmux_name: typeof item.tmux_name === 'string' ? item.tmux_name : undefined,
pid: typeof item.pid === 'number' ? item.pid : undefined,
shell: typeof item.shell === 'string' ? item.shell : undefined,
attached: typeof item.attached === 'number' ? item.attached : undefined,
windows: typeof item.windows === 'number' ? item.windows : undefined,
created_at: typeof item.created_at === 'number' ? item.created_at : undefined,
live: typeof item.live === 'boolean' ? item.live : undefined,
owned_by_client: typeof item.owned_by_client === 'boolean' ? item.owned_by_client : undefined
}
}
function waitForSessions(
relay: import('../transport/RelayTransport.js').RelayTransport
): Promise<TerminalSessionInfo[]> {
return new Promise((resolve, reject) => {
const timer = setTimeout(() => {
relay.onChannel('terminal', null)
reject(new Error(`terminal.sessions timeout after ${COMMAND_TIMEOUT_MS}ms`))
}, COMMAND_TIMEOUT_MS)
timer.unref?.()
relay.onChannel('terminal', (type, payload) => {
if (type === 'terminal.sessions') {
clearTimeout(timer)
relay.onChannel('terminal', null)
const sessions = Array.isArray(payload.sessions)
? payload.sessions.map(asTerminalSession).filter((item): item is TerminalSessionInfo => item !== null)
: []
resolve(sessions)
return
}
if (type === 'terminal.error') {
clearTimeout(timer)
relay.onChannel('terminal', null)
reject(new Error(typeof payload.message === 'string' ? payload.message : 'terminal error'))
}
})
})
}
function waitForDetached(
relay: import('../transport/RelayTransport.js').RelayTransport
): Promise<string> {
return new Promise((resolve, reject) => {
const timer = setTimeout(() => {
relay.onChannel('terminal', null)
reject(new Error(`terminal.kill timeout after ${COMMAND_TIMEOUT_MS}ms`))
}, COMMAND_TIMEOUT_MS)
timer.unref?.()
relay.onChannel('terminal', (type, payload) => {
if (type === 'terminal.detached') {
clearTimeout(timer)
relay.onChannel('terminal', null)
resolve(typeof payload.session_name === 'string' ? payload.session_name : '')
return
}
if (type === 'terminal.error') {
clearTimeout(timer)
relay.onChannel('terminal', null)
reject(new Error(typeof payload.message === 'string' ? payload.message : 'terminal error'))
}
})
})
}
function formatCreated(value: number | undefined): string {
if (!value) {
return 'unknown'
}
return new Date(value * 1000).toLocaleString()
}
async function listTerminalSessions(args: ParsedArgs): Promise<number> {
const authed = await connectAndAuth({
...args,
flags: { ...args.flags, 'non-interactive': args.flags['non-interactive'] ?? true }
})
const { relay, url } = authed
try {
const wait = waitForSessions(relay)
relay.sendChannel('terminal', 'terminal.list', {})
const sessions = (await wait).sort((a, b) => a.name.localeCompare(b.name))
const active = await getActiveTerminalSession(url)
if (args.flags.json) {
process.stdout.write(JSON.stringify({ url, active, sessions }, null, 2) + '\n')
return 0
}
if (sessions.length === 0) {
process.stdout.write(`No Hermes TUI sessions on ${url}.\n`)
process.stdout.write('Run `hermes-relay` to start the default session.\n')
return 0
}
process.stdout.write(`Hermes TUI sessions on ${url}:\n\n`)
for (const session of sessions) {
const activeMark = active?.name === session.name ? ' * active' : ''
const attached = session.attached ?? (session.live ? 1 : 0)
process.stdout.write(` ${session.name}${activeMark}\n`)
process.stdout.write(` tmux: ${session.tmux_name ?? `hermes-${session.name}`}\n`)
process.stdout.write(` attached: ${attached}\n`)
if (session.windows !== undefined) {
process.stdout.write(` windows: ${session.windows}\n`)
}
process.stdout.write(` created: ${formatCreated(session.created_at)}\n\n`)
}
process.stdout.write(
' Resume with `hermes-relay sessions resume <name>`, or run bare `hermes-relay` for the active/default session.\n'
)
return 0
} finally {
relay.kill()
}
}
async function killTerminalSession(args: ParsedArgs): Promise<number> {
const name = args.positional[0]
if (!name) {
process.stderr.write('error: `sessions kill` needs a tmux session name. Run `hermes-relay sessions list` first.\n')
return 2
}
const authed = await connectAndAuth({
...args,
flags: { ...args.flags, 'non-interactive': args.flags['non-interactive'] ?? true }
})
const { relay, url } = authed
try {
const wait = waitForDetached(relay)
relay.sendChannel('terminal', 'terminal.kill', { session_name: name })
await wait
await clearActiveTerminalSession(url, name)
process.stdout.write(`Killed Hermes TUI session "${name}" on ${url}.\n`)
return 0
} catch (e) {
process.stderr.write(`error: ${e instanceof Error ? e.message : String(e)}\n`)
return 1
} finally {
relay.kill()
}
}
export async function sessionsCommand(args: ParsedArgs): Promise<number> {
const sub = args.positional[0] ?? 'list'
if (sub === 'list') {
if (args.positional[0] === 'list') {
args.positional.shift()
}
try {
return await listTerminalSessions(args)
} catch (e) {
process.stderr.write(`error: ${e instanceof Error ? e.message : String(e)}\n`)
return 1
}
}
if (sub === 'resume') {
args.positional.shift()
const name = args.positional[0]
return shellCommand({
...args,
command: 'shell',
flags: name ? { ...args.flags, session: name } : { ...args.flags }
})
}
if (sub === 'new') {
args.positional.shift()
const name = args.positional[0]
return shellCommand({
...args,
command: 'shell',
flags: name ? { ...args.flags, new: true, session: name } : { ...args.flags, new: true }
})
}
if (sub === 'kill') {
args.positional.shift()
return killTerminalSession(args)
}
process.stderr.write('unknown sessions sub-verb. Try: list | resume [name] | new [name] | kill <name>\n')
return 2
}
+51 -15
View File
@@ -57,10 +57,19 @@ import { resolveFirstRunUrl } from '../relayUrlPrompt.js'
import { deleteSession, getSession, saveSession } from '../remoteSessions.js'
import { stageClipboardImageToInbox } from './paste.js'
import { fetchRecentSessions, pickSession } from '../sessionPicker.js'
import {
clearActiveTerminalSession,
getActiveTerminalSession,
saveActiveTerminalSession
} from '../terminalSessionStore.js'
import { ensureToolsConsent } from '../tools/consent.js'
import { configureComputerUseRuntime } from '../tools/computerGrants.js'
import { setComputerActionPromptCoordinator } from '../tools/computerActionApproval.js'
import { DESKTOP_HANDLERS, advertisedDesktopTools } from '../tools/handlerSet.js'
import {
advertisedDesktopTools,
desktopHandlers,
shouldAdvertiseComputerUse
} from '../tools/handlerSet.js'
import { DesktopToolRouter } from '../tools/router.js'
import { RelayTransport } from '../transport/RelayTransport.js'
@@ -92,13 +101,13 @@ function resolveRemoteOrNull(args: ParsedArgs): string | null {
return url ? url.trim() : null
}
interface AuthedRelay {
export interface AuthedRelay {
relay: RelayTransport
url: string
endpointRole: string | null
}
async function connectAndAuth(args: ParsedArgs): Promise<AuthedRelay> {
export async function connectAndAuth(args: ParsedArgs): Promise<AuthedRelay> {
let urlFlag = resolveRemoteOrNull(args)
const argCode = typeof args.flags.code === 'string' ? args.flags.code : undefined
const argToken = typeof args.flags.token === 'string' ? args.flags.token : undefined
@@ -174,6 +183,7 @@ interface AttachedInfo {
shell?: string
tmuxAvailable?: boolean
reattach?: boolean
replay?: string
}
/** Wait for the server's `terminal.attached` ack (or error) after we've
@@ -199,7 +209,8 @@ function waitForAttached(relay: RelayTransport): Promise<AttachedInfo> {
pid: typeof payload.pid === 'number' ? payload.pid : undefined,
shell: typeof payload.shell === 'string' ? payload.shell : undefined,
tmuxAvailable: typeof payload.tmux_available === 'boolean' ? payload.tmux_available : undefined,
reattach: typeof payload.reattach === 'boolean' ? payload.reattach : undefined
reattach: typeof payload.reattach === 'boolean' ? payload.reattach : undefined,
replay: typeof payload.replay === 'string' ? payload.replay : undefined
})
return
}
@@ -347,6 +358,14 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
endpointRole: bannerRole
}) + '\n'
)
const storedTerminalSession = sessionNameArg ? null : await getActiveTerminalSession(url)
const explicitConversation = typeof args.flags.conversation === 'string'
const forceFreshTerminalForConversation = explicitConversation && !sessionNameArg && !args.flags.new
const terminalSessionName =
sessionNameArg ??
((args.flags.new || forceFreshTerminalForConversation)
? `session-${Date.now().toString(36)}`
: (storedTerminalSession?.name ?? 'default'))
// Conversation picker — runs on a TTY when the user didn't pass
// --conversation / --new, and the default exec is still `hermes` (so
@@ -354,7 +373,12 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
// We append `--resume <id>` to the exec command when the user picks
// one; 'cancel' exits cleanly; 'new' leaves the exec command alone.
// --raw / --exec-override / --new all skip the picker entirely.
const skipPicker = raw || execOverride !== null || !!args.flags.new
const skipPicker =
raw ||
execOverride !== null ||
!!args.flags.new ||
(!explicitConversation && (!!sessionNameArg || storedTerminalSession !== null))
let pickedConversationId: string | null = null
if (!skipPicker) {
const picked = await resolveHermesConversationId(relay, args, url)
if (picked === 'cancel') {
@@ -366,6 +390,7 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
}
return 0
}
pickedConversationId = picked
if (picked && postAttachExec === 'hermes') {
// Shell-quote defensively — session ids are typically [a-z0-9-] but
// we can't assume, and the tmux shell will bash-eval the exec line.
@@ -383,21 +408,23 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
if (!toolsDisabled) {
const consent = await ensureToolsConsent(url)
if (consent.consented) {
const computerUseEnabled = shouldAdvertiseComputerUse(args.flags)
configureComputerUseRuntime({
url,
computerUseConsented: true,
computerUseConsented: computerUseEnabled,
consentSource: consent.source ?? 'stored'
})
const advertisedTools = advertisedDesktopTools()
const advertisedTools = advertisedDesktopTools({ computerUse: computerUseEnabled })
toolRouter = new DesktopToolRouter({
consentGranted: true,
handlers: DESKTOP_HANDLERS,
handlers: desktopHandlers({ computerUse: computerUseEnabled }),
advertisedTools: [...advertisedTools]
})
toolRouter.attach(relay)
process.stderr.write(
`Desktop tools: ${advertisedTools.length} handlers advertised (computer-use experimental; control requires grant approval)\n`
)
const computerUseNote = computerUseEnabled
? ' (computer-use experimental; control requires grant approval)'
: ''
process.stderr.write(`Desktop tools: ${advertisedTools.length} handlers advertised${computerUseNote}\n`)
} else if (consent.reason) {
process.stderr.write(`Desktop tools: disabled (${consent.reason})\n`)
}
@@ -412,9 +439,7 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
const attachPromise = waitForAttached(relay)
const attachPayload: Record<string, unknown> = { cols, rows }
if (sessionNameArg) {
attachPayload.session_name = sessionNameArg
}
attachPayload.session_name = terminalSessionName
relay.sendChannel('terminal', 'terminal.attach', attachPayload)
let attached: AttachedInfo
@@ -436,12 +461,22 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
// recent), but echoing it makes the wire self-describing and survives a
// future multi-session client.
const sessionName = attached.sessionName
await saveActiveTerminalSession(url, {
name: sessionName,
conversationId: pickedConversationId,
endpointRole: bannerRole,
serverVersion: relay.serverVersion,
status: attached.reattach ? 'resumed' : 'attached'
})
const reattachMsg = attached.reattach ? ' — re-attached to existing session' : ''
process.stderr.write(
`Attached${attached.tmuxAvailable ? ` (tmux session "${sessionName}")` : ''}${reattachMsg}.\n` +
`${CHORD_HELP}\n\n`
)
if (attached.replay) {
process.stdout.write(attached.replay)
}
// Swap handler from attach-waiter to steady-state output pump. Re-registering
// replaces the previous listener, so `terminal.output` frames now flow to
@@ -557,6 +592,7 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
// Ctrl+A k → destructive kill (tmux session destroyed)
exiting = true
relay.sendChannel('terminal', 'terminal.kill', { session_name: sessionName })
void clearActiveTerminalSession(url, sessionName)
process.stderr.write('\n\x1b[90m[shell] killed tmux session "' + sessionName + '"\x1b[0m\n')
cleanup()
process.exit(0)
@@ -711,7 +747,7 @@ export async function shellCommand(args: ParsedArgs): Promise<number> {
// On a fresh tmux attach the login shell takes ~250-350ms to paint its first
// prompt; injecting `exec` too early means bash swallows the first keystroke
// and the command never runs. Skipped entirely when --raw is set.
if (postAttachExec) {
if (postAttachExec && !attached.reattach) {
setTimeout(() => {
if (exiting) {
return
+1 -1
View File
@@ -66,7 +66,7 @@ export async function statusCommand(args: ParsedArgs): Promise<number> {
process.stdout.write(` paired: ${age} ago\n`)
process.stdout.write(` token: ${tokenDisplay}\n`)
process.stdout.write(` expires: ${humanExpiry(rec.ttlExpiresAt)}\n`)
const computerUse = rec.toolsConsented ? 'with desktop tools' : 'no'
const computerUse = rec.toolsConsented ? 'feature-flagged opt-in' : 'no'
process.stdout.write(
` desktop: tools=${rec.toolsConsented ? 'yes' : 'no'}, computer-use=${computerUse}\n`
)
+529
View File
@@ -0,0 +1,529 @@
// voice — probe the relay's native Hermes voice passthrough.
//
// Hermes voice mode is configured server-side in `~/.hermes/config.yaml`
// (stt.provider / tts.provider). The relay exposes that configuration
// through three HTTP endpoints on the same port as WSS:
// GET /voice/config — basic STT+TTS provider snapshot
// GET /voice/realtime/config — realtime/WS speech-to-speech surface
// POST /voice/transcribe — multipart audio → text (future)
// POST /voice/synthesize — text → audio/mpeg bytes (future)
//
// `voice status` calls the two GET routes and prints what the user has
// configured. Auth is `Authorization: Bearer <session_token>` pulled from
// the local session store, mirroring `devices.ts`. Voice grants
// (`voice:config`, `voice:stt`, `voice:tts`, `voice:realtime`) are minted
// with no expiry at pair time (plugin/relay/auth.py:129), so any paired
// session can probe these without an extra consent step.
//
// Sub-verb shape mirrors `devices`:
// hermes-relay voice (alias for `voice status`)
// hermes-relay voice status basic + realtime snapshot
// hermes-relay voice status --json raw JSON for scripting
//
// Future sub-verbs (`voice say <text>`, `voice transcribe <file>`) slot in
// here without restructuring — the resolver and HTTP helper are reusable.
import type { ParsedArgs } from '../cli.js'
import { resolveCredentials } from '../credentials.js'
import { GatewayClient } from '../gatewayClient.js'
import type {
GatewayEvent,
SessionCreateResponse,
SessionResumeResponse
} from '../gatewayTypes.js'
import { getActiveDesktopRelayUrl } from '../desktopConfig.js'
import { setupGracefulExit } from '../lib/gracefulExit.js'
import { asRpcResult, rpcErrorMessage } from '../lib/rpc.js'
import { resolveFirstRunUrl } from '../relayUrlPrompt.js'
import { deleteSession, getSession, listSessions, saveSession } from '../remoteSessions.js'
import { RelayTransport } from '../transport/RelayTransport.js'
import { openInBrowser, startVoiceServer } from '../voiceServer.js'
import { discoverTray, notifyTrayShowVoice } from '../trayBridge.js'
const READY_TIMEOUT_MS = 60_000
interface VoiceProvider {
provider?: string | null
model?: string | null
voice?: string | null
voice_id?: string | null
enabled?: boolean
available?: boolean
}
interface VoiceConfigResponse {
success?: boolean
tts?: VoiceProvider | null
stt?: VoiceProvider | null
requirements?: Record<string, unknown> | null
error?: string
}
interface RealtimeProvider {
id: string
name?: string | null
status?: string | null
description?: string | null
supports_tts?: boolean
supports_stt?: boolean
supports_speech_to_speech?: boolean
supports_interruption?: boolean
}
interface RealtimeVoiceConfigResponse {
success?: boolean
enabled?: boolean
protocol?: string | null
default_provider?: string | null
default_model?: string | null
default_voice?: string | null
sample_rate?: number
providers?: RealtimeProvider[]
error?: string
}
/** `ws(s)://host:port` → `http(s)://host:port` — voice routes are HTTP. */
function wsToHttp(url: string): string {
const trimmed = url.trim()
if (trimmed.startsWith('wss://')) {
return 'https://' + trimmed.slice('wss://'.length)
}
if (trimmed.startsWith('ws://')) {
return 'http://' + trimmed.slice('ws://'.length)
}
return trimmed
}
async function resolveRemoteAndToken(
args: ParsedArgs
): Promise<{ url: string; token: string }> {
const argUrl = typeof args.flags.remote === 'string' ? args.flags.remote.trim() : null
const envUrl = process.env.HERMES_RELAY_URL?.trim()
const argToken = typeof args.flags.token === 'string' ? args.flags.token.trim() : null
const envToken = process.env.HERMES_RELAY_TOKEN?.trim()
if (argToken || envToken) {
const url = argUrl ?? envUrl
if (!url) {
throw new Error('--token supplied without --remote. Pass both, or set HERMES_RELAY_URL.')
}
return { url, token: (argToken ?? envToken)! }
}
const stored = await listSessions()
const urls = Object.keys(stored)
const activeDesktopUrl = await getActiveDesktopRelayUrl()
let url: string
if (argUrl || envUrl) {
url = argUrl ?? envUrl!
} else if (activeDesktopUrl) {
url = activeDesktopUrl
} else if (urls.length === 1) {
url = urls[0]!
} else if (urls.length === 0) {
throw new Error('No paired relays. Run `hermes-relay pair --remote ws://host:port` first.')
} else {
throw new Error(
`Multiple paired relays; pass --remote to pick one (${urls.join(', ')}).`
)
}
const rec = await getSession(url)
if (!rec) {
throw new Error(`No stored session for ${url}. Run \`hermes-relay pair --remote ${url}\` first.`)
}
return { url, token: rec.token }
}
async function getJson<T>(
httpUrl: string,
token: string
): Promise<{ status: number; body: T | { error?: string } | string | undefined }> {
const res = await fetch(httpUrl, {
method: 'GET',
headers: {
Authorization: `Bearer ${token}`,
Accept: 'application/json'
}
})
const text = await res.text()
let body: unknown
if (text.length > 0) {
try {
body = JSON.parse(text)
} catch {
body = text
}
}
return { status: res.status, body: body as T | { error?: string } | string | undefined }
}
function formatProvider(p: VoiceProvider | null | undefined, label: string): string {
if (!p || !p.provider) {
return ` ${label}: (not configured)`
}
const enabled = p.enabled === false ? '○' : '●'
const provider = p.provider
const model = p.model ? ` · ${p.model}` : ''
const voice = p.voice ?? p.voice_id
const voiceTag = voice ? ` · voice=${voice}` : ''
return ` ${enabled} ${label}: ${provider}${model}${voiceTag}`
}
function formatRealtime(rt: RealtimeVoiceConfigResponse | null): string[] {
if (!rt) {
return [' Realtime: (unavailable)']
}
if (rt.success === false) {
return [` Realtime: error — ${rt.error ?? 'unknown'}`]
}
const lines: string[] = []
const enabled = rt.enabled ? '●' : '○'
const provider = rt.default_provider ?? '(none)'
const model = rt.default_model ? ` · ${rt.default_model}` : ''
const voice = rt.default_voice ? ` · voice=${rt.default_voice}` : ''
const rate = rt.sample_rate ? ` @ ${rt.sample_rate}Hz` : ''
lines.push(` ${enabled} Realtime: ${provider}${model}${voice}${rate}`)
const providers = (rt.providers ?? []).filter((p) => p.status && p.status !== 'unavailable')
if (providers.length > 0) {
const labels = providers.map((p) => p.name ?? p.id)
lines.push(` available: ${labels.join(', ')}`)
}
return lines
}
async function voiceStatus(args: ParsedArgs): Promise<number> {
const { url, token } = await resolveRemoteAndToken(args)
const httpBase = wsToHttp(url)
// Run both probes in parallel — they're independent and fast.
const [basic, realtime] = await Promise.all([
getJson<VoiceConfigResponse>(`${httpBase}/voice/config`, token),
getJson<RealtimeVoiceConfigResponse>(`${httpBase}/voice/realtime/config`, token).catch(
() => ({ status: 0, body: undefined as undefined })
)
])
if (args.flags.json) {
const payload = {
url,
voice: basic.status === 200 ? basic.body : { status: basic.status, error: basic.body },
realtime:
realtime.status === 200
? realtime.body
: realtime.status === 0
? null
: { status: realtime.status, error: realtime.body }
}
process.stdout.write(JSON.stringify(payload, null, 2) + '\n')
// Non-zero on the primary probe so scripts can `voice status --json | jq`
// AND still rely on exit code as the canonical success signal.
return basic.status === 200 ? 0 : 1
}
if (basic.status !== 200) {
const detail =
typeof basic.body === 'string'
? basic.body
: JSON.stringify(basic.body)
if (basic.status === 401 || basic.status === 403) {
process.stderr.write(
`error: voice auth rejected (${basic.status}). Re-pair: hermes-relay pair --remote ${url}\n` +
` (${detail})\n`
)
} else if (basic.status === 404) {
process.stderr.write(
`error: relay at ${url} has no /voice/config — server is too old or voice plugin not loaded.\n`
)
} else {
process.stderr.write(
`error: GET /voice/config returned ${basic.status}: ${detail}\n`
)
}
return 1
}
const cfg = basic.body as VoiceConfigResponse
const rt = (realtime.status === 200 ? (realtime.body as RealtimeVoiceConfigResponse) : null)
process.stdout.write(`Voice on ${url}:\n\n`)
process.stdout.write(formatProvider(cfg.stt, 'STT') + '\n')
process.stdout.write(formatProvider(cfg.tts, 'TTS') + '\n')
for (const line of formatRealtime(rt)) {
process.stdout.write(line + '\n')
}
const sttOk = cfg.stt?.enabled !== false && !!cfg.stt?.provider
const ttsOk = cfg.tts?.enabled !== false && !!cfg.tts?.provider
process.stdout.write('\n')
if (sttOk && ttsOk) {
process.stdout.write(' Native Hermes voice is configured on the server. ✓\n')
} else if (!sttOk && !ttsOk) {
process.stdout.write(
' Neither STT nor TTS is configured.\n' +
' Edit ~/.hermes/config.yaml on the server (stt.provider / tts.provider) and restart.\n'
)
} else {
process.stdout.write(
` Partial: ${sttOk ? 'STT' : 'TTS'} is configured, ${sttOk ? 'TTS' : 'STT'} is not.\n` +
' See ~/.hermes/config.yaml on the server.\n'
)
}
process.stdout.write('\n ● = enabled ○ = available but off\n')
return 0
}
// ─────────────────────────────────────────────────────────────────────────
// voice mode — push-to-talk in a browser, proxied through this Node process
// to the relay's existing /voice/transcribe + prompt.submit + /voice/synthesize.
// ─────────────────────────────────────────────────────────────────────────
interface AuthedRelay {
relay: RelayTransport
url: string
endpointRole: string | null
token: string
}
/** Mirror of chat.ts's connectAndAuth but also exposes the final token so
* we can hand it to the voice server. Stays inline rather than refactoring
* chat.ts because the chat path doesn't need the token externally. */
async function connectAndAuth(args: ParsedArgs): Promise<AuthedRelay> {
let urlFlag =
(typeof args.flags.remote === 'string' ? args.flags.remote.trim() : null) ??
process.env.HERMES_RELAY_URL?.trim() ??
null
const argCode = typeof args.flags.code === 'string' ? args.flags.code : undefined
const argToken = typeof args.flags.token === 'string' ? args.flags.token : undefined
const argPairQr =
typeof args.flags['pair-qr'] === 'string' ? args.flags['pair-qr'] : process.env.HERMES_RELAY_PAIR_QR
const nonInteractive = !!args.flags['non-interactive']
if (!urlFlag && !argPairQr) {
urlFlag = await resolveFirstRunUrl({ nonInteractive })
}
const probeUrl = urlFlag ?? 'ws://pair-qr-pending'
for (let attempt = 0; attempt < 2; attempt++) {
const creds = await resolveCredentials(probeUrl, {
argCode,
argToken,
argPairQr,
nonInteractive
})
const url = (creds.resolvedEndpoint?.relay.url ?? urlFlag)!.trim()
const endpointRole = creds.resolvedEndpoint?.role ?? null
const cfg: ConstructorParameters<typeof RelayTransport>[0] = {
url,
deviceName: `hermes-relay-cli voice (${process.platform})`
}
if (creds.pairingCode) cfg.pairingCode = creds.pairingCode
if (creds.sessionToken) cfg.sessionToken = creds.sessionToken
const relay = new RelayTransport(cfg)
let mintedToken: string | null = creds.sessionToken ?? null
relay.onAuthSuccess((token, ver, meta) => {
mintedToken = token
void saveSession(url, token, ver, {
grants: meta.grants,
ttlExpiresAt: meta.ttlExpiresAt,
endpointRole
})
})
relay.start()
const outcome = await relay.whenAuthResolved()
if (outcome.ok && mintedToken) {
return { relay, url, endpointRole, token: mintedToken }
}
try { relay.kill() } catch { /* ignore */ }
if (creds.sessionToken) await deleteSession(url)
if (attempt === 1 || nonInteractive) {
throw new Error(`relay rejected credentials: ${outcome.ok ? 'no token issued' : outcome.reason}`)
}
process.stderr.write(`\nRelay rejected credentials: ${outcome.ok ? 'no token issued' : outcome.reason}\n`)
}
throw new Error('unreachable: connectAndAuth exhausted loop')
}
function waitForReady(gw: GatewayClient, timeoutMs = READY_TIMEOUT_MS): Promise<void> {
return new Promise((resolve, reject) => {
const timer = setTimeout(() => {
gw.off('event', handler)
reject(new Error(`gateway.ready timeout after ${timeoutMs}ms`))
}, timeoutMs)
const handler = (ev: GatewayEvent) => {
if (ev.type === 'gateway.ready') {
clearTimeout(timer)
gw.off('event', handler)
resolve()
}
}
gw.on('event', handler)
})
}
async function createOrResumeSession(
gw: GatewayClient,
resumeId: string | null
): Promise<{ sessionId: string; model: string | null }> {
const cols = process.stdout.columns ?? 80
if (resumeId) {
const raw = await gw.request<SessionResumeResponse>('session.resume', { session_id: resumeId, cols })
const r = asRpcResult<SessionResumeResponse>(raw)
if (!r?.session_id) throw new Error(`failed to resume session ${resumeId}`)
return { sessionId: r.session_id, model: r.info?.model ?? null }
}
const raw = await gw.request<SessionCreateResponse>('session.create', { cols })
const r = asRpcResult<SessionCreateResponse>(raw)
if (!r?.session_id) throw new Error('failed to create session')
return { sessionId: r.session_id, model: r.info?.model ?? null }
}
async function voiceMode(args: ParsedArgs): Promise<number> {
const noOpen = !!args.flags['no-open']
const noTray = !!args.flags['no-tray']
const portFlag = typeof args.flags.port === 'string' ? parseInt(args.flags.port, 10) : NaN
const port = Number.isFinite(portFlag) && portFlag >= 0 && portFlag <= 65535 ? portFlag : 0
const conversation =
(typeof args.flags.conversation === 'string' ? args.flags.conversation : null) ??
(typeof args.flags.session === 'string' ? args.flags.session : null)
// Tray-first short-circuit: if the tray app is running AND it has its
// own voice surface (the daemon's voice server is up + the tray UI has
// a voice tab), just ask the tray to focus voice and exit. We don't
// need to spin up our own gateway/session for that — the daemon already
// owns one. `--no-tray` opts out for testing the browser path.
// `--conversation` forces our own session, so we skip the short-circuit
// there too — the daemon's session is independent of any --conversation
// the user requested.
if (!noTray && !conversation) {
const ctl = await discoverTray()
if (ctl) {
const ok = await notifyTrayShowVoice(ctl)
if (ok) {
process.stderr.write('Tray-hosted voice mode focused. (Pass --no-tray to use the standalone browser path.)\n')
return 0
}
process.stderr.write('Tray IPC unreachable; falling back to standalone voice server.\n')
}
}
process.stderr.write(`Connecting to relay...\n`)
let authed: AuthedRelay
try {
authed = await connectAndAuth(args)
} catch (e) {
process.stderr.write(`error: ${rpcErrorMessage(e)}\n`)
return 1
}
const { relay, url, token } = authed
const gw = new GatewayClient(relay)
const tearDownState = { closed: false }
let voiceServerHandle: { url: string; close: () => Promise<void> } | null = null
const tearDown = async () => {
if (tearDownState.closed) return
tearDownState.closed = true
try { await voiceServerHandle?.close() } catch { /* ignore */ }
try { gw.kill() } catch { /* ignore */ }
}
setupGracefulExit({ cleanups: [tearDown] })
const ready = waitForReady(gw)
gw.start()
gw.drain()
try {
await ready
} catch (e) {
process.stderr.write(`error: ${rpcErrorMessage(e)}\n`)
await tearDown()
return 1
}
let session: { sessionId: string; model: string | null }
try {
session = await createOrResumeSession(gw, conversation)
} catch (e) {
process.stderr.write(`error: ${rpcErrorMessage(e)}\n`)
await tearDown()
return 1
}
if (session.model) {
process.stderr.write(`Session ${session.sessionId.slice(0, 8)}… on ${session.model}\n`)
}
try {
voiceServerHandle = await startVoiceServer({
token,
relayUrl: url,
gateway: gw,
sessionId: session.sessionId,
port
})
} catch (e) {
process.stderr.write(`error: failed to start voice server: ${rpcErrorMessage(e)}\n`)
await tearDown()
return 1
}
process.stdout.write(
`\nVoice mode ready.\n` +
` Open: ${voiceServerHandle.url}\n` +
` Relay: ${url}\n` +
` Press Ctrl+C to stop.\n\n`
)
if (!noOpen) {
openInBrowser(voiceServerHandle.url)
}
// Park until the gateway transport exits (user Ctrl+C also triggers
// tearDown via setupGracefulExit). Returning here ends the process.
await new Promise<void>((resolve) => {
gw.on('exit', () => resolve())
})
await tearDown()
return 0
}
export async function voiceCommand(args: ParsedArgs): Promise<number> {
const sub = args.positional[0] ?? 'status'
if (sub === 'status') {
if (args.positional.length > 0 && args.positional[0] === 'status') {
args.positional.shift()
}
try {
return await voiceStatus(args)
} catch (e) {
process.stderr.write(`error: ${e instanceof Error ? e.message : String(e)}\n`)
return 1
}
}
if (sub === 'mode') {
args.positional.shift()
try {
return await voiceMode(args)
} catch (e) {
process.stderr.write(`error: ${e instanceof Error ? e.message : String(e)}\n`)
return 1
}
}
process.stderr.write(`unknown voice sub-verb "${sub}". Try: status | mode\n`)
return 2
}
+8 -3
View File
@@ -18,7 +18,12 @@
import type { EndpointCandidate } from './endpoint.js'
import { promptForPairingCode } from './pairing.js'
import { decodePairingPayload, payloadToCandidates, probeCandidatesByPriority } from './pairingQr.js'
import {
decodePairingPayload,
payloadToRelayCandidates,
probeCandidatesByPriority,
relayPairingCodeFromPayload
} from './pairingQr.js'
import { getSession } from './remoteSessions.js'
export interface Credentials {
@@ -62,10 +67,10 @@ export async function resolveCredentials(
const pairQr = opts.argPairQr?.trim() || envPairQr
if (pairQr) {
const payload = decodePairingPayload(pairQr)
const candidates = payloadToCandidates(payload)
const candidates = payloadToRelayCandidates(payload)
const winner = await probeCandidatesByPriority(candidates)
return {
pairingCode: payload.key.toUpperCase(),
pairingCode: relayPairingCodeFromPayload(payload),
resolvedEndpoint: winner,
}
}
+20
View File
@@ -0,0 +1,20 @@
import { promises as fs } from 'node:fs'
import { homedir } from 'node:os'
import { join } from 'node:path'
interface DesktopControlConfigFile {
relay_url?: unknown
}
const configPath = () => join(homedir(), '.hermes', 'desktop-control.json')
export async function getActiveDesktopRelayUrl(): Promise<string | null> {
try {
const raw = await fs.readFile(configPath(), 'utf8')
const parsed = JSON.parse(raw) as DesktopControlConfigFile
const relayUrl = typeof parsed.relay_url === 'string' ? parsed.relay_url.trim() : ''
return relayUrl || null
} catch {
return null
}
}
+75 -4
View File
@@ -42,8 +42,8 @@ export const PROBE_CACHE_TTL_MS = 60_000
* `transport_hint` / `code` are all optional so v1 QRs with only `url`
* still decode.
*
* `code` is not currently used by the CLI (the top-level `key` field
* carries the pairing code) but is preserved so we can pivot later.
* `code` is the relay one-shot pairing code. The top-level `key` field is
* the Hermes API bearer for direct HTTP chat and must stay separate.
*/
export interface PairingRelay {
url: string
@@ -72,6 +72,77 @@ export interface PairingPayload {
sig?: string
}
const PAIRING_CODE_RE = /^[A-Z0-9]{6}$/
function normalizeBase64Payload(raw: string): string {
return raw.replaceAll('-', '+').replaceAll('_', '/')
}
function extractInviteCandidate(raw: string): string | null {
const trimmed = raw.trim()
if (!trimmed) return null
const direct = trimmed.match(/^hermes-relay:\/\/pair(?:\/([^?\s#]+))?(?:\?([^\s#]+))?/i)
const embedded = direct ?? trimmed.match(/hermes-relay:\/\/pair(?:\/([^?\s#]+))?(?:\?([^\s#]+))?/i)
if (!embedded) return null
const candidate = embedded[0]
try {
const url = new URL(candidate)
const fromQuery = url.searchParams.get('payload') ?? url.searchParams.get('p')
if (fromQuery && fromQuery.trim()) return fromQuery.trim()
const fromPath = url.pathname.replace(/^\/+/, '')
if (fromPath) return decodeURIComponent(fromPath)
} catch {
// Fall through to regex query extraction for partial clipboard lines.
}
const query = embedded[2]
if (query) {
const params = new URLSearchParams(query)
const fromQuery = params.get('payload') ?? params.get('p')
if (fromQuery && fromQuery.trim()) return fromQuery.trim()
}
const pathPayload = embedded[1]
return pathPayload ? decodeURIComponent(pathPayload) : null
}
/**
* Return the raw JSON/base64 payload from either a QR payload or a
* paste-friendly ``hermes-relay://pair?payload=...`` invite URL.
*/
export function unwrapPairingPayload(raw: string): string {
const trimmed = raw.trim()
return extractInviteCandidate(trimmed) ?? trimmed
}
/**
* The relay one-shot code lives in ``relay.code``. Top-level ``key`` is the
* Hermes API bearer for direct HTTP chat and must never be used as a relay
* pairing code.
*/
export function relayPairingCodeFromPayload(payload: PairingPayload): string {
const code = payload.relay?.code?.trim().toUpperCase() ?? ''
if (!code) {
throw new Error(
'pairing invite has no relay.code. Start the relay on the server and mint a new invite.'
)
}
if (!PAIRING_CODE_RE.test(code)) {
throw new Error('pairing invite has an invalid relay.code; expected 6 chars of A-Z or 0-9')
}
return code
}
export function payloadToRelayCandidates(payload: PairingPayload): EndpointCandidate[] {
const candidates = payloadToCandidates(payload).filter((candidate) =>
typeof candidate.relay.url === 'string' && candidate.relay.url.trim().length > 0
)
if (candidates.length === 0) {
throw new Error(
'pairing invite has no relay URL. Start the relay on the server and mint a new invite.'
)
}
return candidates
}
/**
* Parse an `endpoints[i]` object from the wire. Returns null on
* malformed input so the outer parser can silently skip bad records
@@ -110,7 +181,7 @@ function parseCandidate(v: unknown): EndpointCandidate | null {
* directly to the user.
*/
export function decodePairingPayload(raw: string): PairingPayload {
const trimmed = raw.trim()
const trimmed = unwrapPairingPayload(raw)
if (trimmed.length === 0) {
throw new Error('empty pairing payload')
}
@@ -128,7 +199,7 @@ export function decodePairingPayload(raw: string): PairingPayload {
}
try {
// Accept both standard and URL-safe base64 variants.
const normalized = trimmed.replaceAll('-', '+').replaceAll('_', '/')
const normalized = normalizeBase64Payload(trimmed)
text = Buffer.from(normalized, 'base64').toString('utf8')
parsed = JSON.parse(text)
} catch {
+16 -6
View File
@@ -20,6 +20,7 @@
import { createInterface } from 'node:readline/promises'
import { getActiveDesktopRelayUrl } from './desktopConfig.js'
import { listSessions } from './remoteSessions.js'
const URL_RE = /^wss?:\/\/\S+$/
@@ -102,11 +103,14 @@ export interface ResolveFirstRunUrlOptions {
* When neither --remote nor HERMES_RELAY_URL nor --pair-qr is set, figure out
* what URL the user wants to talk to:
*
* 1. If `~/.hermes/remote-sessions.json` has exactly one entry → return it
* 1. If the desktop tray has an active relay in
* `~/.hermes/desktop-control.json` → return it (the GUI-selected relay
* wins even when multiple CLI sessions are stored).
* 2. If `~/.hermes/remote-sessions.json` has exactly one entry → return it
* and note the pick to stderr (zero-friction re-use).
* 2. If it has multiple entries → show a numbered list, let the user pick
* 3. If it has multiple entries → show a numbered list, let the user pick
* or type "n" to enter a new URL.
* 3. If it has zero entries → print the first-run banner and prompt for
* 4. If it has zero entries → print the first-run banner and prompt for
* a URL directly.
*
* Non-interactive callers (daemon, CI scripts with --non-interactive) never
@@ -118,8 +122,14 @@ export async function resolveFirstRunUrl(
): Promise<string> {
const sessions = await listSessions()
const urls = Object.keys(sessions)
const activeDesktopUrl = await getActiveDesktopRelayUrl()
// Case (1): exactly one stored session — auto-pick. Both interactive and
if (activeDesktopUrl) {
process.stderr.write(`Using active desktop relay ${activeDesktopUrl}\n`)
return activeDesktopUrl
}
// Case (2): exactly one stored session — auto-pick. Both interactive and
// non-interactive callers benefit (it's the happy path for repeat users).
if (urls.length === 1) {
const url = urls[0]!
@@ -149,7 +159,7 @@ export async function resolveFirstRunUrl(
)
}
// Case (3): zero stored sessions — first-run path. Show a welcoming banner
// Case (4): zero stored sessions — first-run path. Show a welcoming banner
// before the URL prompt so a brand-new user knows they're in the right place.
if (urls.length === 0) {
const banner =
@@ -159,7 +169,7 @@ export async function resolveFirstRunUrl(
return promptForRelayUrl()
}
// Case (2): multiple stored sessions — numbered picker with "n" for new URL.
// Case (3): multiple stored sessions — numbered picker with "n" for new URL.
process.stderr.write('\nStored sessions:\n')
urls.forEach((u, i) => {
process.stderr.write(` ${i + 1}. ${u}\n`)
+257
View File
@@ -0,0 +1,257 @@
import { spawn, spawnSync } from 'node:child_process'
export interface SurfacePluginCommand {
id: string
label: string
description: string
}
export interface SurfacePluginDescriptor {
id: string
name: string
description: string
sourceUrl: string
packageName: string
binaryName: string
tabs: string[]
commands: SurfacePluginCommand[]
keybindings: SurfacePluginCommand[]
statusCards: SurfacePluginCommand[]
sessionActions: SurfacePluginCommand[]
}
export interface PluginCommandPlan {
program: string
args: string[]
display: string
mode: string
}
export interface SurfacePluginStatus {
descriptor: SurfacePluginDescriptor
installed: boolean
available: boolean
version: string | null
command: string
installer: PluginCommandPlan | null
update: PluginCommandPlan | null
fallback: PluginCommandPlan | null
launch: PluginCommandPlan | null
resume: PluginCommandPlan | null
setupHint: string
}
export const BUILTIN_SURFACE_PLUGINS: SurfacePluginDescriptor[] = [
{
id: 'herm',
name: 'Herm',
description: 'OpenTUI dashboard for Hermes Agent, packaged as herm-tui.',
sourceUrl: 'https://github.com/liftaris/herm',
packageName: 'herm-tui',
binaryName: 'herm',
tabs: [
'chat',
'sessions',
'context',
'agents',
'analytics',
'skills',
'cron',
'toolsets',
'config',
'env',
'memory',
'kanban'
],
commands: [
{
id: 'install',
label: 'Install',
description: 'Install herm-tui globally with Bun when available, otherwise npm.'
},
{
id: 'update',
label: 'Update',
description: 'Re-run the package manager install to refresh herm-tui.'
},
{
id: 'launch',
label: 'Launch',
description: 'Open a fresh Herm dashboard session.'
},
{
id: 'resume',
label: 'Resume',
description: 'Open Herm with -c to continue the last dashboard session.'
}
],
keybindings: [
{
id: 'palette',
label: 'Ctrl+K',
description: 'Open the Herm command palette.'
},
{
id: 'keys',
label: '/keys',
description: 'Show or edit Herm keybindings.'
}
],
statusCards: [
{
id: 'install',
label: 'Install state',
description: 'Reports whether the herm binary is on PATH.'
},
{
id: 'runtime',
label: 'Runtime',
description: 'Reports Bun/npm fallback availability.'
},
{
id: 'source',
label: 'Source',
description: 'Links the built-in plugin to liftaris/herm.'
}
],
sessionActions: [
{
id: 'fresh',
label: 'Fresh session',
description: 'Run herm without resume flags.'
},
{
id: 'resume',
label: 'Resume last',
description: 'Run herm -c.'
}
]
}
]
function quoteArg(arg: string): string {
if (/^[A-Za-z0-9_./:@=-]+$/.test(arg)) {
return arg
}
return `"${arg.replaceAll('"', '\\"')}"`
}
function makePlan(program: string, args: string[], mode: string): PluginCommandPlan {
return {
program,
args,
mode,
display: [program, ...args].map(quoteArg).join(' ')
}
}
function commandOk(program: string, args: string[] = ['--version']): boolean {
const result = spawnSync(program, args, {
stdio: 'ignore',
shell: process.platform === 'win32'
})
return !result.error && result.status === 0
}
function commandOutput(program: string, args: string[] = ['--version']): string | null {
const result = spawnSync(program, args, {
encoding: 'utf8',
shell: process.platform === 'win32'
})
if (result.error || result.status !== 0) {
return null
}
const output = `${result.stdout ?? ''}${result.stderr ?? ''}`
.split(/\r?\n/)
.map((line) => line.trim())
.find(Boolean)
return output ?? null
}
function installPlan(plugin: SurfacePluginDescriptor): PluginCommandPlan | null {
if (commandOk('bun')) {
return makePlan('bun', ['add', '-g', plugin.packageName], 'bun')
}
if (commandOk('npm')) {
return makePlan('npm', ['install', '-g', plugin.packageName], 'npm')
}
return null
}
function fallbackPlan(plugin: SurfacePluginDescriptor, resume = false): PluginCommandPlan | null {
const args = resume ? [plugin.packageName, '-c'] : [plugin.packageName]
if (commandOk('bunx')) {
return makePlan('bunx', args, 'bunx')
}
if (commandOk('npx')) {
return makePlan('npx', ['--yes', ...args], 'npx')
}
return null
}
function launchPlan(plugin: SurfacePluginDescriptor, resume = false): PluginCommandPlan | null {
const args = resume ? ['-c'] : []
if (commandOk(plugin.binaryName)) {
return makePlan(plugin.binaryName, args, 'installed')
}
return fallbackPlan(plugin, resume)
}
export function getSurfacePlugin(id: string): SurfacePluginDescriptor | null {
return BUILTIN_SURFACE_PLUGINS.find((plugin) => plugin.id === id) ?? null
}
export function listSurfacePluginStatuses(): SurfacePluginStatus[] {
return BUILTIN_SURFACE_PLUGINS.map(surfacePluginStatus)
}
export function surfacePluginStatus(plugin: SurfacePluginDescriptor): SurfacePluginStatus {
const installed = commandOk(plugin.binaryName)
const installer = installPlan(plugin)
const launch = launchPlan(plugin, false)
const resume = launchPlan(plugin, true)
const fallback = fallbackPlan(plugin, false)
const version = installed ? commandOutput(plugin.binaryName) : null
const setupHint = installed
? `${plugin.binaryName} is available on PATH.`
: installer
? `Install with ${installer.display}.`
: fallback
? `Use fallback launch with ${fallback.display}.`
: 'Install Bun or npm, then install herm-tui.'
return {
descriptor: plugin,
installed,
available: launch !== null,
version,
command: plugin.binaryName,
installer,
update: installer,
fallback,
launch,
resume,
setupHint
}
}
export function resolvePluginPlan(
plugin: SurfacePluginDescriptor,
action: 'install' | 'update' | 'launch' | 'resume'
): PluginCommandPlan | null {
if (action === 'install' || action === 'update') {
return installPlan(plugin)
}
return launchPlan(plugin, action === 'resume')
}
export async function runPluginPlan(plan: PluginCommandPlan): Promise<number> {
return new Promise((resolve, reject) => {
const child = spawn(plan.program, plan.args, {
stdio: 'inherit',
shell: process.platform === 'win32'
})
child.on('error', reject)
child.on('close', (code) => resolve(code ?? 0))
})
}
+154
View File
@@ -0,0 +1,154 @@
import { promises as fs } from 'node:fs'
import { homedir } from 'node:os'
import { dirname, join } from 'node:path'
export interface ActiveTerminalSession {
name: string
conversationId?: string | null
endpointRole?: string | null
serverVersion?: string | null
status?: string | null
lastAttachedAt: number
}
interface StoredActiveTerminalSession {
name: string
conversation_id?: string | null
endpoint_role?: string | null
server_version?: string | null
status?: string | null
last_attached_at: number
}
interface StoredFile {
version: number
active_by_relay: Record<string, StoredActiveTerminalSession>
}
const STORE_VERSION = 1
const defaultPath = () => join(homedir(), '.hermes', 'desktop-sessions.json')
const emptyFile = (): StoredFile => ({ version: STORE_VERSION, active_by_relay: {} })
let pathOverride: string | null = null
export const setTerminalSessionStorePath = (path: string | null) => {
pathOverride = path
}
export const terminalSessionStorePath = () => pathOverride ?? defaultPath()
const toRecord = (raw: StoredActiveTerminalSession): ActiveTerminalSession | null => {
if (!raw || typeof raw.name !== 'string' || !raw.name.trim()) {
return null
}
return {
name: raw.name,
conversationId: raw.conversation_id ?? null,
endpointRole: raw.endpoint_role ?? null,
serverVersion: raw.server_version ?? null,
status: raw.status ?? null,
lastAttachedAt: Number.isFinite(raw.last_attached_at) ? raw.last_attached_at : 0
}
}
const fromRecord = (record: ActiveTerminalSession): StoredActiveTerminalSession => ({
name: record.name,
conversation_id: record.conversationId ?? null,
endpoint_role: record.endpointRole ?? null,
server_version: record.serverVersion ?? null,
status: record.status ?? null,
last_attached_at: record.lastAttachedAt
})
const readFile = async (): Promise<StoredFile> => {
try {
const raw = await fs.readFile(terminalSessionStorePath(), 'utf8')
const parsed = JSON.parse(raw) as unknown
if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) {
return emptyFile()
}
const obj = parsed as Record<string, unknown>
const active = obj.active_by_relay
if (!active || typeof active !== 'object' || Array.isArray(active)) {
return emptyFile()
}
return {
version: STORE_VERSION,
active_by_relay: active as Record<string, StoredActiveTerminalSession>
}
} catch {
return emptyFile()
}
}
const writeFile = async (file: StoredFile): Promise<void> => {
const path = terminalSessionStorePath()
await fs.mkdir(dirname(path), { recursive: true, mode: 0o700 })
const tmp = `${path}.tmp-${process.pid}-${Date.now()}`
await fs.writeFile(tmp, JSON.stringify(file, null, 2), { mode: 0o600 })
await fs.rename(tmp, path)
}
export const getActiveTerminalSession = async (url: string): Promise<ActiveTerminalSession | null> => {
try {
const file = await readFile()
const raw = file.active_by_relay[url]
return raw ? toRecord(raw) : null
} catch {
return null
}
}
export const saveActiveTerminalSession = async (
url: string,
record: Omit<ActiveTerminalSession, 'lastAttachedAt'> & { lastAttachedAt?: number }
): Promise<void> => {
try {
const name = record.name.trim()
if (!url.trim() || !name) {
return
}
const file = await readFile()
file.active_by_relay[url] = fromRecord({
...record,
name,
lastAttachedAt: record.lastAttachedAt ?? Date.now()
})
await writeFile(file)
} catch {
/* persistence failures must not break the terminal session */
}
}
export const clearActiveTerminalSession = async (url: string, name?: string): Promise<void> => {
try {
const file = await readFile()
const current = file.active_by_relay[url]
if (!current) {
return
}
if (name && current.name !== name) {
return
}
delete file.active_by_relay[url]
await writeFile(file)
} catch {
/* fail closed */
}
}
export const listActiveTerminalSessions = async (): Promise<Record<string, ActiveTerminalSession>> => {
try {
const file = await readFile()
const out: Record<string, ActiveTerminalSession> = {}
for (const [url, raw] of Object.entries(file.active_by_relay)) {
const record = toRecord(raw)
if (record) {
out[url] = record
}
}
return out
} catch {
return {}
}
}
@@ -1,4 +1,6 @@
import { createInterface } from 'node:readline/promises'
import { mkdir, readFile, rename, rm, writeFile } from 'node:fs/promises'
import { join } from 'node:path'
export interface ComputerActionApprovalRequest {
action: string
@@ -41,6 +43,73 @@ function cleanAnswer(raw: string): string {
.trim()
}
function sleep(ms: number): Promise<void> {
return new Promise(resolve => setTimeout(resolve, ms))
}
function safeBridgeId(): string {
return `grant-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`
}
async function writeJsonAtomic(path: string, value: unknown): Promise<void> {
const tmp = `${path}.tmp-${process.pid}-${Date.now()}`
await writeFile(tmp, `${JSON.stringify(value, null, 2)}\n`, 'utf8')
await rename(tmp, path)
}
async function approveComputerGrantViaBridge(
request: ComputerGrantApprovalRequest,
bridgeDir: string
): Promise<ComputerActionApprovalDecision> {
const id = safeBridgeId()
await mkdir(bridgeDir, { recursive: true })
const requestPath = join(bridgeDir, `request-${id}.json`)
const responsePath = join(bridgeDir, `response-${id}.json`)
const timeoutMs = Math.min(Math.max(request.durationSeconds * 1000, 30_000), 175_000)
try {
await writeJsonAtomic(requestPath, {
id,
kind: 'computer_grant_request',
mode: request.mode,
duration_seconds: request.durationSeconds,
reason: request.reason,
scope: request.scope,
created_at: new Date().toISOString()
})
const deadline = Date.now() + timeoutMs
while (Date.now() < deadline) {
try {
const parsed = JSON.parse(await readFile(responsePath, 'utf8')) as {
approved?: unknown
reason?: unknown
}
return {
approved: parsed.approved === true,
reason: typeof parsed.reason === 'string' ? parsed.reason : ''
}
} catch (err) {
const code = (err as NodeJS.ErrnoException).code
if (code && code !== 'ENOENT') {
return {
approved: false,
reason: `native grant approval bridge failed: ${String((err as Error).message ?? err)}`
}
}
}
await sleep(500)
}
return {
approved: false,
reason: 'native grant approval timed out'
}
} finally {
await rm(requestPath, { force: true }).catch(() => undefined)
await rm(responsePath, { force: true }).catch(() => undefined)
}
}
/** Legacy one-action approval prompt retained for compatibility with older
* callers. Current computer-use control approves the assist/control grant
* once, then actions run until the grant expires or is canceled. */
@@ -61,6 +130,10 @@ export async function approveComputerGrant(
request: ComputerGrantApprovalRequest
): Promise<ComputerActionApprovalDecision> {
if (!request.interactive) {
const bridgeDir = process.env.HERMES_RELAY_GRANT_BRIDGE_DIR
if (bridgeDir) {
return approveComputerGrantViaBridge(request, bridgeDir)
}
return {
approved: false,
reason: 'non-interactive mode - computer control grant approval requires a TTY'
+33 -8
View File
@@ -45,7 +45,7 @@ import {
import type { ToolHandler } from './router.js'
/** Experimental computer-use tools are registered in the local handler map
* and heartbeat-advertised with the rest of the desktop tools after normal
* but heartbeat-advertised only when explicitly feature-flagged after normal
* desktop-tool consent. Host input still fails closed unless a task-scoped
* grant exists and was approved from a visible local prompt. */
export const DESKTOP_COMPUTER_USE_TOOLS: readonly string[] = Object.freeze([
@@ -112,27 +112,52 @@ export interface DesktopAdvertiseOptions {
computerUse?: boolean
}
function envEnabled(value: string | undefined): boolean {
if (!value) {
return false
}
return ['1', 'true', 'yes', 'on'].includes(value.trim().toLowerCase())
}
export function shouldAdvertiseComputerUse(
_flags?: Record<string, string | true>
flags: Record<string, string | true> = {},
env: NodeJS.ProcessEnv = process.env
): boolean {
return true
if (flags['no-computer-use'] === true) {
return false
}
if (flags['experimental-computer-use'] === true || flags['allow-computer-use'] === true) {
return true
}
return envEnabled(env.HERMES_RELAY_EXPERIMENTAL_COMPUTER_USE) ||
envEnabled(env.HERMES_RELAY_COMPUTER_USE)
}
export function desktopHandlers(
opts: DesktopAdvertiseOptions = {}
): Record<string, ToolHandler> {
if (opts.computerUse !== true) {
return BASE_DESKTOP_HANDLERS
}
return DESKTOP_HANDLERS
}
/** Stable list of advertised tool names — what the heartbeat claims to
* service. Computer-use tools are included by default with the rest of the
* desktop tool surface; callers may pass `computerUse:false` only for tests
* or a future explicit disable path. */
* service. Computer-use tools are feature-flagged so the regular desktop
* CLI/daemon surface stays primary and backward-compatible. */
export function advertisedDesktopTools(
opts: DesktopAdvertiseOptions = {}
): readonly string[] {
if (opts.computerUse === false) {
if (opts.computerUse !== true) {
const experimental = new Set(DESKTOP_COMPUTER_USE_TOOLS)
return Object.freeze(Object.keys(DESKTOP_HANDLERS).filter(name => !experimental.has(name)))
}
return Object.freeze(Object.keys(DESKTOP_HANDLERS))
}
export const DESKTOP_ADVERTISED_TOOLS: readonly string[] = advertisedDesktopTools()
export const DESKTOP_ADVERTISED_TOOLS: readonly string[] = advertisedDesktopTools({
computerUse: shouldAdvertiseComputerUse()
})
/** Short summary line used by chat.ts / shell.ts when announcing to the
* user that tools are wired. Centralizing the count avoids the "9 handlers"
+5 -8
View File
@@ -373,13 +373,6 @@ export const computerGrantRequestHandler: ToolHandler = async (args, ctx) => {
'Assist/control grants require local desktop-tool consent for this relay URL before task-scoped input grants can be created.'
)
}
if (!ctx.interactive) {
return failure(
'not_interactive',
'Assist/control grants require a visible local approval prompt. This desktop client is running non-interactively.',
{ requested_mode: mode }
)
}
const approval = await approveComputerGrant({
mode,
durationSeconds: normalizeComputerGrantDurationSeconds(args.duration_seconds),
@@ -388,7 +381,11 @@ export const computerGrantRequestHandler: ToolHandler = async (args, ctx) => {
interactive: ctx.interactive
})
if (!approval.approved) {
return failure('rejected', approval.reason, { requested_mode: mode })
return failure(
approval.reason.startsWith('non-interactive mode') ? 'not_interactive' : 'rejected',
approval.reason,
{ requested_mode: mode }
)
}
}
return {
+86
View File
@@ -0,0 +1,86 @@
// Foreign-process bridge to the tray app.
//
// The Tauri tray hosts a tiny localhost HTTP listener on a random port,
// recorded along with a token in `~/.hermes/desktop-tray-control.json`.
// Sibling processes (notably `hermes-relay voice mode`) read that file
// and POST `/voice/show` to bring the tray window forward + activate
// the voice tab — instead of opening a system browser.
//
// Why HTTP and not Tauri IPC: Tauri's `invoke()` is webview-only. A
// foreign process has no entry point into the Tauri runtime, so the
// tray opens its own loopback HTTP port. Token + loopback are the gate.
import { promises as fs } from 'node:fs'
import * as os from 'node:os'
import * as path from 'node:path'
interface TrayControl {
port: number
token: string
pid?: number
started_at?: number
}
const CONTROL_FILE = 'desktop-tray-control.json'
function controlPath(): string {
return path.join(os.homedir(), '.hermes', CONTROL_FILE)
}
/** Read the tray control file. Returns null if the tray isn't running or
* if the file is missing / malformed. Callers should treat null as the
* normal "tray unavailable" case, not an error. */
export async function discoverTray(): Promise<TrayControl | null> {
let raw: string
try {
raw = await fs.readFile(controlPath(), 'utf8')
} catch {
return null
}
let parsed: unknown
try {
parsed = JSON.parse(raw)
} catch {
return null
}
if (typeof parsed !== 'object' || parsed === null) return null
const o = parsed as Record<string, unknown>
const port = typeof o.port === 'number' ? o.port : NaN
const token = typeof o.token === 'string' ? o.token : ''
if (!Number.isFinite(port) || port <= 0 || port > 65535) return null
if (!token) return null
const out: TrayControl = { port, token }
if (typeof o.pid === 'number') out.pid = o.pid
if (typeof o.started_at === 'number') out.started_at = o.started_at
return out
}
/** Ask the tray to focus the voice tab. Returns true on 200, false otherwise
* (including network failure, 401, etc.). Best-effort — the caller falls
* back to a system-browser open when this returns false. */
export async function notifyTrayShowVoice(control: TrayControl, timeoutMs = 2000): Promise<boolean> {
const ctl = new AbortController()
const timer = setTimeout(() => ctl.abort(), timeoutMs)
try {
const res = await fetch(`http://127.0.0.1:${control.port}/voice/show`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ token: control.token }),
signal: ctl.signal
})
return res.status === 200
} catch {
return false
} finally {
clearTimeout(timer)
}
}
/** Convenience helper: discover + notify in one call. Returns true if the
* tray accepted the wakeup; false in every other case (no tray, mismatched
* token, dropped connection, etc.). Caller treats false as "open browser". */
export async function tryTrayShowVoice(): Promise<boolean> {
const ctl = await discoverTray()
if (!ctl) return false
return notifyTrayShowVoice(ctl)
}
+568
View File
@@ -0,0 +1,568 @@
// Embedded single-page UI for `hermes-relay voice mode`.
//
// Everything lives in one HTML string so the Node binary stays self-contained
// (no static asset directory to ship). Vanilla JS only — no framework, no
// build step, no external CDNs. The page talks ONLY to the local Node server
// over loopback; the relay bearer token never leaves the Node process.
//
// Wire protocol with the Node side:
// POST {base}/turn Content-Type: audio/<mime> Body: raw audio bytes
// → SSE stream:
// event: transcript data: {"text": "..."}
// event: delta data: {"text": "..."}
// event: complete data: {"text": "..."}
// event: error data: {"message": "..."}
// POST {base}/synthesize Content-Type: application/json
// Body: {"text": "..."}
// → audio/mpeg bytes (binary)
//
// The {base} prefix is `/v/<nonce>` — see voiceServer.ts for why.
export function renderVoicePage(opts: { base: string; relayUrl: string }): string {
const safeBase = JSON.stringify(opts.base)
const safeRelayUrl = escapeHtml(opts.relayUrl)
return `<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<title>Hermes Voice Mode</title>
<style>
:root {
color-scheme: dark;
--bg: #0b0d10;
--panel: #14181d;
--panel-2: #1b2028;
--fg: #e6e8eb;
--muted: #8a939c;
--accent: #5eead4;
--accent-fg: #042f2c;
--you: #93c5fd;
--agent: #fbcfe8;
--err: #fca5a5;
--border: #232a31;
}
html, body {
margin: 0; height: 100%; background: var(--bg); color: var(--fg);
font-family: ui-sans-serif, -apple-system, "Segoe UI", system-ui, sans-serif;
font-size: 15px;
}
body { display: grid; grid-template-rows: auto 1fr auto; }
header {
padding: 14px 20px; border-bottom: 1px solid var(--border);
display: flex; align-items: center; gap: 16px;
}
header h1 { font-size: 14px; margin: 0; font-weight: 600; letter-spacing: 0.04em; text-transform: uppercase; color: var(--muted); }
header .relay { font-size: 12px; color: var(--muted); margin-left: auto; }
#status {
font-size: 12px; padding: 4px 10px; border-radius: 999px;
background: var(--panel-2); color: var(--muted);
border: 1px solid var(--border);
}
#status[data-state="recording"] { color: var(--err); border-color: var(--err); }
#status[data-state="thinking"] { color: var(--accent); border-color: var(--accent); }
#status[data-state="speaking"] { color: var(--you); border-color: var(--you); }
#status[data-state="error"] { color: var(--err); border-color: var(--err); }
main {
padding: 20px; overflow-y: auto;
display: flex; flex-direction: column; gap: 12px;
}
.turn {
display: flex; flex-direction: column; gap: 6px;
padding: 12px 14px; border-radius: 10px; background: var(--panel);
border: 1px solid var(--border);
}
.turn .label { font-size: 11px; letter-spacing: 0.06em; text-transform: uppercase; color: var(--muted); }
.turn.you .label { color: var(--you); }
.turn.agent .label { color: var(--agent); }
.turn.error .label { color: var(--err); }
.turn .body { white-space: pre-wrap; line-height: 1.5; }
footer {
padding: 18px 20px; border-top: 1px solid var(--border);
display: flex; flex-direction: column; align-items: center; gap: 10px;
background: var(--panel);
}
.mic {
width: 96px; height: 96px; border-radius: 50%;
background: var(--accent); color: var(--accent-fg);
border: none; cursor: pointer; user-select: none;
font-size: 13px; font-weight: 600; letter-spacing: 0.04em;
box-shadow: 0 6px 20px rgba(94, 234, 212, 0.18);
transition: transform 60ms ease;
}
.mic:active, .mic[data-recording="true"] {
transform: scale(0.96);
background: var(--err); color: white;
box-shadow: 0 6px 20px rgba(252, 165, 165, 0.18);
}
.mic:disabled { opacity: 0.5; cursor: not-allowed; }
.hint { font-size: 12px; color: var(--muted); text-align: center; }
kbd {
background: var(--panel-2); border: 1px solid var(--border);
padding: 1px 6px; border-radius: 4px; font-size: 11px;
}
/* Three-layer SVG sine waveform (ported from conjure/AudioWaveform.tsx).
Dimensions match the conjure pill overlay (PillOverlayRoot). Color
shifts with state via CSS custom property set on the host. */
.wave-host {
position: relative; width: 256px; height: 64px;
--wave-color: var(--accent);
}
.wave-host[data-state="recording"] { --wave-color: var(--err); }
.wave-host[data-state="speaking"] { --wave-color: var(--you); }
.wave-host[data-state="thinking"] { --wave-color: var(--accent); }
.wave-host svg { width: 100%; height: 100%; display: block; }
.wave-host path { fill: none; stroke: var(--wave-color); stroke-width: 1.6; stroke-linecap: round; stroke-linejoin: round; }
</style>
</head>
<body>
<header>
<h1>Hermes Voice Mode</h1>
<span id="status" data-state="idle">Idle</span>
<span class="relay">${safeRelayUrl}</span>
</header>
<main id="log">
<div class="turn agent">
<div class="label">Hermes</div>
<div class="body">Hold the mic button (or press &amp; hold <kbd>Space</kbd>) and start talking. Release to send.</div>
</div>
</main>
<footer>
<div class="wave-host" id="waveHost" data-state="idle">
<svg viewBox="0 0 256 64" preserveAspectRatio="none" aria-hidden="true">
<path id="wave0" opacity="1"></path>
<path id="wave1" opacity="0.78"></path>
<path id="wave2" opacity="0.56"></path>
</svg>
</div>
<button id="mic" class="mic" disabled>Hold to talk</button>
<div class="hint">Mic permission required. The bearer token stays on the Node side.</div>
</footer>
<script>
(() => {
const BASE = ${safeBase};
const statusEl = document.getElementById('status');
const logEl = document.getElementById('log');
const micBtn = document.getElementById('mic');
const waveHost = document.getElementById('waveHost');
const wavePaths = [
document.getElementById('wave0'),
document.getElementById('wave1'),
document.getElementById('wave2')
];
let stream = null;
let recorder = null;
let recording = false;
let busy = false;
let chunks = [];
let currentAudio = null;
// ── Waveform — verbatim port of conjure/AudioWaveform.tsx ─────────
// Constants, WAVE_CONFIG, createWavePath, animation step, and both
// state-transition effects mirror the source 1:1. Dimensions match the
// conjure PillOverlayRoot usage (256×64, baselineOffset=0). Levels come
// from a single AnalyserNode tap on the mic stream — the TTS playback
// does NOT feed the wave (matches conjure: when speaking, the processing
// flag drives the constant PROCESSING_BASE_LEVEL breathe, no analyser).
const TAU = Math.PI * 2;
const LEVEL_SMOOTHING = 0.14;
const TARGET_DECAY_PER_FRAME = 0.988;
const WAVE_BASE_PHASE_STEP = 0.065;
const WAVE_PHASE_GAIN = 0.2;
const MIN_AMPLITUDE = 0.03;
const MAX_AMPLITUDE = 1.3;
const PROCESSING_BASE_LEVEL = 0.16;
const WAVE_CONFIG = [
{ frequency: 0.8, multiplier: 2.0, phaseOffset: 0, opacity: 1 },
{ frequency: 1.0, multiplier: 1.7, phaseOffset: 0.85, opacity: 0.78 },
{ frequency: 1.25, multiplier: 1.3, phaseOffset: 1.7, opacity: 0.56 }
];
const WAVE_WIDTH = 256;
const WAVE_HEIGHT = 64;
const BASELINE_OFFSET = 0;
// Animation state — mirrors conjure's animationStateRef.
const waveState = { phase: 0, currentLevel: 0, targetLevel: 0 };
// Phase flags — mirrors conjure's phaseStateRef. Read inside the rAF
// step so transitions take effect on the next frame without a restart.
const phaseState = { active: false, processing: false };
let waveRafId = null;
// Single mic analyser — matches conjure's levels prop coming from the
// recording capture only. TTS playback does NOT tap an analyser.
let audioCtx = null;
let micAnalyser = null;
let analyserBuf = null;
function ensureAudioCtx() {
if (!audioCtx) audioCtx = new (window.AudioContext || window.webkitAudioContext)();
return audioCtx;
}
function attachMicAnalyser(mediaStream) {
const ctx = ensureAudioCtx();
const src = ctx.createMediaStreamSource(mediaStream);
micAnalyser = ctx.createAnalyser();
micAnalyser.fftSize = 256;
micAnalyser.smoothingTimeConstant = 0.6;
src.connect(micAnalyser);
analyserBuf = new Uint8Array(micAnalyser.frequencyBinCount);
}
// ── Effect 1 — conjure: useEffect([active, processing]) ────────────
// When !active: target = processing ? max(target, BASE) : 0
// if !processing: currentLevel *= 0.4 (sharp drop on stop)
function applyTransitionEffect() {
if (!phaseState.active) {
waveState.targetLevel = phaseState.processing
? Math.max(waveState.targetLevel, PROCESSING_BASE_LEVEL)
: 0;
if (!phaseState.processing) {
waveState.currentLevel *= 0.4;
if (waveState.currentLevel < 0.0002) waveState.currentLevel = 0;
}
}
}
// ── Effect 2 — conjure: useEffect([active, processing, width, height])
// FULL idle branch — hard-reset state, snap paths flat, cancel rAF.
// Active branch — (re)start the rAF loop.
function applyIdleOrStartEffect() {
if (!(phaseState.active || phaseState.processing)) {
waveState.targetLevel = 0;
waveState.currentLevel = 0;
waveState.phase = 0;
const baseline = WAVE_HEIGHT / 2 + BASELINE_OFFSET;
const flat = 'M 0 ' + baseline + ' L ' + WAVE_WIDTH + ' ' + baseline;
for (let i = 0; i < WAVE_CONFIG.length; i++) {
wavePaths[i].setAttribute('d', flat);
wavePaths[i].setAttribute('opacity', WAVE_CONFIG[i].opacity.toString());
}
if (waveRafId != null) {
cancelAnimationFrame(waveRafId);
waveRafId = null;
}
return;
}
if (waveRafId == null) {
waveRafId = requestAnimationFrame(waveStep);
}
}
// ── Effect-equivalent: useEffect([levels, active]) ─────────────────
// Called each frame from the rAF step when mic is recording. Same
// average×0.75 + peak×0.65 → pow(0.7)*1.2 boost → target *= 0.35 + 0.65.
function feedMicLevels() {
if (!phaseState.active || !micAnalyser || !analyserBuf) return;
try {
micAnalyser.getByteFrequencyData(analyserBuf);
} catch { return; }
let sum = 0;
let peak = 0;
for (let i = 0; i < analyserBuf.length; i++) {
const v = analyserBuf[i] / 255;
sum += v;
if (v > peak) peak = v;
}
const average = sum / analyserBuf.length;
const combined = Math.min(1, average * 0.75 + peak * 0.65);
const boosted = Math.min(1, Math.pow(combined, 0.7) * 1.2);
waveState.targetLevel = Math.min(1, waveState.targetLevel * 0.35 + boosted * 0.65);
}
// ── rAF step — verbatim from conjure's step() closure. ────────────
function waveStep() {
feedMicLevels();
waveState.currentLevel +=
(waveState.targetLevel - waveState.currentLevel) * LEVEL_SMOOTHING;
if (waveState.currentLevel < 0.0002) waveState.currentLevel = 0;
waveState.targetLevel *= TARGET_DECAY_PER_FRAME;
if (waveState.targetLevel < 0.0005) waveState.targetLevel = 0;
const baseLevel =
phaseState.processing && !phaseState.active ? PROCESSING_BASE_LEVEL : 0;
const level = Math.max(baseLevel, waveState.currentLevel);
const advance = WAVE_BASE_PHASE_STEP + WAVE_PHASE_GAIN * level;
waveState.phase = (waveState.phase + advance) % TAU;
const baseline = WAVE_HEIGHT / 2 + BASELINE_OFFSET;
const waveHeight = WAVE_HEIGHT;
const waveWidth = WAVE_WIDTH;
for (let i = 0; i < WAVE_CONFIG.length; i++) {
const config = WAVE_CONFIG[i];
const amplitudeFactor = Math.min(
MAX_AMPLITUDE,
Math.max(MIN_AMPLITUDE, level * config.multiplier)
);
const amplitude = Math.max(1, waveHeight * 0.75 * amplitudeFactor);
const phase = waveState.phase + config.phaseOffset;
wavePaths[i].setAttribute('d', createWavePath(waveWidth, baseline, amplitude, config.frequency, phase));
wavePaths[i].setAttribute('opacity', config.opacity.toString());
}
waveRafId = requestAnimationFrame(waveStep);
}
// ── createWavePath — verbatim from conjure. ────────────────────────
function createWavePath(width, baseline, amplitude, frequency, phase) {
const segments = Math.max(72, Math.floor(width / 2));
let path = 'M 0 ' + (baseline + amplitude * Math.sin(phase));
for (let index = 1; index <= segments; index += 1) {
const t = index / segments;
const x = width * t;
const theta = frequency * t * TAU + phase;
const y = baseline + amplitude * Math.sin(theta);
path += ' L ' + x + ' ' + y;
}
return path;
}
// Map our 5-state UI to conjure's (active, processing) pair:
// recording → active=true, processing=false
// thinking → active=false, processing=true
// speaking → active=false, processing=true (TTS playback still
// breathes at PROCESSING_BASE_LEVEL — same as conjure)
// idle/error → both false
function setWaveState(uiState) {
waveHost.dataset.state = uiState;
phaseState.active = uiState === 'recording';
phaseState.processing = uiState === 'thinking' || uiState === 'speaking';
applyTransitionEffect();
applyIdleOrStartEffect();
}
function setStatus(state, label) {
statusEl.dataset.state = state;
statusEl.textContent = label;
setWaveState(state);
}
function appendTurn(role, text) {
const t = document.createElement('div');
t.className = 'turn ' + role;
const label = document.createElement('div');
label.className = 'label';
label.textContent = role === 'you' ? 'You' : role === 'agent' ? 'Hermes' : 'Error';
const body = document.createElement('div');
body.className = 'body';
body.textContent = text;
t.appendChild(label);
t.appendChild(body);
logEl.appendChild(t);
logEl.scrollTop = logEl.scrollHeight;
return body;
}
async function initMic() {
try {
stream = await navigator.mediaDevices.getUserMedia({ audio: true });
// Let the browser pick its best supported codec (Chrome=webm/opus,
// Firefox=webm/opus, Safari=mp4). Relay's transcribe accepts all three.
const mime = pickMime();
recorder = new MediaRecorder(stream, mime ? { mimeType: mime } : undefined);
recorder.ondataavailable = (e) => { if (e.data && e.data.size > 0) chunks.push(e.data); };
recorder.onstop = onRecorderStop;
attachMicAnalyser(stream);
micBtn.disabled = false;
setStatus('idle', 'Idle');
} catch (e) {
setStatus('error', 'Mic denied');
appendTurn('error', 'Could not access microphone: ' + (e && e.message ? e.message : String(e)));
micBtn.disabled = true;
}
}
function pickMime() {
const candidates = [
'audio/webm;codecs=opus',
'audio/webm',
'audio/ogg;codecs=opus',
'audio/mp4'
];
for (const m of candidates) {
if (window.MediaRecorder && MediaRecorder.isTypeSupported(m)) return m;
}
return null;
}
function startRecording() {
if (busy || recording || !recorder) return;
chunks = [];
recorder.start();
recording = true;
micBtn.dataset.recording = 'true';
micBtn.textContent = 'Release';
setStatus('recording', 'Recording');
if (currentAudio) {
try { currentAudio.pause(); } catch {}
currentAudio = null;
}
}
function stopRecording() {
if (!recording || !recorder) return;
recording = false;
micBtn.dataset.recording = 'false';
micBtn.textContent = 'Working...';
micBtn.disabled = true;
recorder.stop();
}
async function onRecorderStop() {
busy = true;
setStatus('thinking', 'Transcribing');
const blob = new Blob(chunks, { type: recorder.mimeType || 'audio/webm' });
chunks = [];
let youBody = null;
let agentBody = null;
let finalText = '';
try {
const res = await fetch(BASE + '/turn', {
method: 'POST',
headers: { 'Content-Type': blob.type },
body: blob
});
if (!res.ok || !res.body) {
const txt = await safeText(res);
throw new Error('turn failed (' + res.status + '): ' + txt);
}
const reader = res.body.pipeThrough(new TextDecoderStream()).getReader();
let buf = '';
while (true) {
const r = await reader.read();
if (r.done) break;
buf += r.value;
let idx;
while ((idx = buf.indexOf('\\n\\n')) >= 0) {
const frame = buf.slice(0, idx);
buf = buf.slice(idx + 2);
const ev = parseSseFrame(frame);
if (!ev) continue;
if (ev.event === 'transcript') {
const t = (ev.data && ev.data.text) || '';
youBody = appendTurn('you', t);
agentBody = appendTurn('agent', '');
setStatus('thinking', 'Thinking');
} else if (ev.event === 'delta') {
const t = (ev.data && ev.data.text) || '';
if (agentBody) agentBody.textContent += t;
finalText += t;
} else if (ev.event === 'complete') {
const t = (ev.data && ev.data.text) || finalText;
if (agentBody) agentBody.textContent = t;
finalText = t;
} else if (ev.event === 'error') {
const msg = (ev.data && ev.data.message) || 'unknown error';
appendTurn('error', msg);
setStatus('error', 'Error');
return;
}
}
}
} catch (e) {
appendTurn('error', e && e.message ? e.message : String(e));
setStatus('error', 'Error');
busy = false;
micBtn.disabled = false;
micBtn.textContent = 'Hold to talk';
return;
}
// Synthesize the final reply and play it.
if (finalText.trim().length > 0) {
setStatus('speaking', 'Speaking');
try {
const tts = await fetch(BASE + '/synthesize', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ text: finalText })
});
if (!tts.ok) {
const txt = await safeText(tts);
appendTurn('error', 'TTS failed (' + tts.status + '): ' + txt);
} else {
const audioBlob = await tts.blob();
const url = URL.createObjectURL(audioBlob);
currentAudio = new Audio(url);
// Conjure design: TTS playback does NOT feed the waveform. The
// 'speaking' state sets processing=true → wave breathes at the
// constant PROCESSING_BASE_LEVEL. Keep playback as plain audio.
await currentAudio.play().catch(() => {});
currentAudio.onended = () => {
URL.revokeObjectURL(url);
currentAudio = null;
};
}
} catch (e) {
appendTurn('error', 'TTS error: ' + (e && e.message ? e.message : String(e)));
}
}
setStatus('idle', 'Idle');
busy = false;
micBtn.disabled = false;
micBtn.textContent = 'Hold to talk';
}
async function safeText(res) {
try { return await res.text(); } catch { return '<no body>'; }
}
function parseSseFrame(frame) {
const lines = frame.split('\\n');
let event = 'message';
let dataLines = [];
for (const line of lines) {
if (line.startsWith('event: ')) event = line.slice(7).trim();
else if (line.startsWith('data: ')) dataLines.push(line.slice(6));
}
if (dataLines.length === 0) return null;
const raw = dataLines.join('\\n');
try { return { event, data: JSON.parse(raw) }; }
catch { return { event, data: { _raw: raw } }; }
}
// Pointer controls — desktop and touch.
micBtn.addEventListener('mousedown', startRecording);
micBtn.addEventListener('mouseup', stopRecording);
micBtn.addEventListener('mouseleave', () => { if (recording) stopRecording(); });
micBtn.addEventListener('touchstart', (e) => { e.preventDefault(); startRecording(); });
micBtn.addEventListener('touchend', (e) => { e.preventDefault(); stopRecording(); });
// Spacebar push-to-talk. Ignore when focus is in an editable element.
let spaceHeld = false;
document.addEventListener('keydown', (e) => {
if (e.code !== 'Space' || e.repeat) return;
const t = e.target;
if (t && (t.tagName === 'INPUT' || t.tagName === 'TEXTAREA' || t.isContentEditable)) return;
e.preventDefault();
if (!spaceHeld) { spaceHeld = true; startRecording(); }
});
document.addEventListener('keyup', (e) => {
if (e.code !== 'Space') return;
if (spaceHeld) { spaceHeld = false; stopRecording(); }
});
initMic();
})();
</script>
</body>
</html>
`
}
function escapeHtml(s: string): string {
return s
.replace(/&/g, '&amp;')
.replace(/</g, '&lt;')
.replace(/>/g, '&gt;')
.replace(/"/g, '&quot;')
.replace(/'/g, '&#39;')
}
+416
View File
@@ -0,0 +1,416 @@
// Loopback-only HTTP server that backs `hermes-relay voice mode`.
//
// The browser page (see voicePage.ts) does mic capture and playback in
// JavaScript and ships the audio bytes to this server. The server is a thin
// proxy onto the relay: it forwards audio to `/voice/transcribe`, submits
// the transcript via the existing gateway client (`prompt.submit`), streams
// `message.delta` / `message.complete` events back as SSE, and forwards the
// final reply text to `/voice/synthesize` returning the mp3 bytes verbatim.
//
// Design notes:
//
// * **Loopback only.** Server binds to 127.0.0.1 — any cross-host caller
// can't reach it without ssh-tunneling on purpose.
// * **Path-prefix nonce.** All endpoints sit under `/v/<nonce>` where
// <nonce> is 32 hex chars generated at startup. Other localhost callers
// (other browser tabs, other apps on the box) can't guess the prefix,
// so they can't drive the mic round-trip from the user's box.
// * **Token never crosses the wire.** The relay bearer token stays in
// this Node process. The page only ever sees `/v/<nonce>/...` paths.
// * **No CORS, no cookies, no auth.** Loopback + nonce is the gate. A
// cross-origin caller can't read SSE bodies or audio bodies anyway
// (no `Access-Control-Allow-Origin`).
//
// Concurrency: one in-flight turn at a time. If a turn POST arrives while
// another is running we 409 — the page won't trigger that because the mic
// button is locked while busy, but it's a defensive shape.
import { createServer, type IncomingMessage, type ServerResponse } from 'node:http'
import { Readable } from 'node:stream'
import { randomBytes } from 'node:crypto'
import { spawn } from 'node:child_process'
import type { AddressInfo } from 'node:net'
import { renderVoicePage } from './voicePage.js'
import type { GatewayClient } from './gatewayClient.js'
import type { GatewayEvent, PromptSubmitResponse } from './gatewayTypes.js'
import { rpcErrorMessage } from './lib/rpc.js'
const TURN_TIMEOUT_MS = 5 * 60_000
export interface VoiceServerOptions {
/** Relay bearer token. Used as `Authorization: Bearer <token>` on voice routes. */
token: string
/** Relay base URL — `ws(s)://host:port`. Converted to http(s) for the voice routes. */
relayUrl: string
/** Gateway client, already-connected and ready (caller awaited `gateway.ready`). */
gateway: GatewayClient
/** Hermes session id to drive — caller created or resumed it. */
sessionId: string
/** Bind port. 0 picks an available port. */
port?: number
}
export interface VoiceServer {
/** http://127.0.0.1:<port>/v/<nonce>/ — the URL to open in the browser. */
url: string
/** Stop the server and tear down all in-flight responses. */
close: () => Promise<void>
}
/** Start a loopback-only HTTP server that proxies voice round-trips to the
* relay. The returned `url` includes the nonce prefix. */
export async function startVoiceServer(opts: VoiceServerOptions): Promise<VoiceServer> {
const nonce = randomBytes(16).toString('hex')
const base = `/v/${nonce}`
const httpBase = wsToHttp(opts.relayUrl)
let inFlight: AbortController | null = null
const server = createServer((req, res) => {
handleRequest(req, res, { nonce, base, httpBase, opts, getInFlight: () => inFlight, setInFlight: (c) => { inFlight = c } })
.catch((e) => {
try {
if (!res.headersSent) {
res.writeHead(500, { 'Content-Type': 'text/plain' })
}
res.end(`internal error: ${e instanceof Error ? e.message : String(e)}\n`)
} catch {
/* socket already dead */
}
})
})
await new Promise<void>((resolve, reject) => {
server.once('error', reject)
server.listen(opts.port ?? 0, '127.0.0.1', () => resolve())
})
const addr = server.address() as AddressInfo
const url = `http://127.0.0.1:${addr.port}${base}/`
return {
url,
close: async () => {
try { inFlight?.abort() } catch { /* ignore */ }
await new Promise<void>((resolve) => server.close(() => resolve()))
}
}
}
interface HandlerCtx {
nonce: string
base: string
httpBase: string
opts: VoiceServerOptions
getInFlight: () => AbortController | null
setInFlight: (c: AbortController | null) => void
}
async function handleRequest(req: IncomingMessage, res: ServerResponse, ctx: HandlerCtx): Promise<void> {
const url = req.url ?? ''
// Strip the nonce prefix. Anything that doesn't start with it 404s — that's
// the entire localhost-tab-isolation guarantee.
if (!url.startsWith(ctx.base + '/') && url !== ctx.base) {
res.writeHead(404, { 'Content-Type': 'text/plain' })
res.end('not found\n')
return
}
const path = url === ctx.base ? '/' : url.slice(ctx.base.length)
if (req.method === 'GET' && (path === '/' || path === '')) {
const html = renderVoicePage({ base: ctx.base, relayUrl: ctx.opts.relayUrl })
res.writeHead(200, {
'Content-Type': 'text/html; charset=utf-8',
'Cache-Control': 'no-store',
'Referrer-Policy': 'no-referrer'
})
res.end(html)
return
}
if (req.method === 'POST' && path === '/turn') {
await handleTurn(req, res, ctx)
return
}
if (req.method === 'POST' && path === '/synthesize') {
await handleSynthesize(req, res, ctx)
return
}
res.writeHead(404, { 'Content-Type': 'text/plain' })
res.end('not found\n')
}
async function handleTurn(req: IncomingMessage, res: ServerResponse, ctx: HandlerCtx): Promise<void> {
if (ctx.getInFlight()) {
res.writeHead(409, { 'Content-Type': 'text/plain' })
res.end('another turn is in flight\n')
return
}
const abort = new AbortController()
ctx.setInFlight(abort)
res.writeHead(200, {
'Content-Type': 'text/event-stream; charset=utf-8',
'Cache-Control': 'no-store',
Connection: 'keep-alive',
'X-Accel-Buffering': 'no'
})
function sendEvent(event: string, data: Record<string, unknown>): void {
if (res.writableEnded) return
res.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`)
}
function endTurn(): void {
ctx.setInFlight(null)
try { if (!res.writableEnded) res.end() } catch { /* ignore */ }
}
// Drop the connection if the browser navigates away mid-turn.
req.on('close', () => { try { abort.abort() } catch { /* ignore */ } })
try {
const audioMime = (req.headers['content-type'] ?? 'audio/webm').toString().split(';', 1)[0]?.trim() || 'audio/webm'
const audioBytes = await readBody(req, 25 * 1024 * 1024)
if (audioBytes.length === 0) {
sendEvent('error', { message: 'empty audio body' })
endTurn()
return
}
// STT — forward to /voice/transcribe as multipart.
const transcript = await transcribeOnRelay({
httpBase: ctx.httpBase,
token: ctx.opts.token,
audio: audioBytes,
audioMime,
signal: abort.signal
})
sendEvent('transcript', { text: transcript })
if (transcript.trim().length === 0) {
sendEvent('complete', { text: '' })
sendEvent('error', { message: 'empty transcript — try speaking louder or closer' })
endTurn()
return
}
// Gateway turn — submit prompt and forward streaming events. We attach
// the listener BEFORE prompt.submit so we don't drop early deltas.
let finalText = ''
let completed = false
const settle = new Promise<void>((resolve, reject) => {
const timer = setTimeout(() => {
if (completed) return
cleanup()
reject(new Error(`turn timeout after ${TURN_TIMEOUT_MS}ms`))
}, TURN_TIMEOUT_MS)
timer.unref?.()
const onAbort = () => {
if (completed) return
cleanup()
// Best-effort interrupt — the gateway may have already finished.
ctx.opts.gateway.request('session.interrupt', { session_id: ctx.opts.sessionId }).catch(() => {})
reject(new Error('cancelled'))
}
const handler = (ev: GatewayEvent) => {
if (completed) return
if (ev.type === 'message.delta') {
const t = ev.payload?.text ?? ''
if (t) {
finalText += t
sendEvent('delta', { text: t })
}
} else if (ev.type === 'message.complete') {
const t = ev.payload?.text ?? finalText
finalText = t
completed = true
cleanup()
sendEvent('complete', { text: t })
resolve()
} else if (ev.type === 'error') {
completed = true
cleanup()
sendEvent('error', { message: ev.payload?.message ?? 'agent error' })
resolve()
}
}
function cleanup(): void {
clearTimeout(timer)
ctx.opts.gateway.off('event', handler)
abort.signal.removeEventListener('abort', onAbort)
}
ctx.opts.gateway.on('event', handler)
abort.signal.addEventListener('abort', onAbort)
ctx.opts.gateway
.request<PromptSubmitResponse>('prompt.submit', { session_id: ctx.opts.sessionId, text: transcript })
.catch((e: unknown) => {
if (completed) return
completed = true
cleanup()
sendEvent('error', { message: rpcErrorMessage(e) })
resolve()
})
})
await settle
} catch (e) {
if (!abort.signal.aborted) {
sendEvent('error', { message: e instanceof Error ? e.message : String(e) })
}
} finally {
endTurn()
}
}
async function handleSynthesize(req: IncomingMessage, res: ServerResponse, ctx: HandlerCtx): Promise<void> {
let text = ''
try {
const raw = await readBody(req, 1 * 1024 * 1024)
const parsed = JSON.parse(raw.toString('utf8'))
if (typeof parsed?.text !== 'string') {
res.writeHead(400, { 'Content-Type': 'text/plain' })
res.end('synthesize: body must be {"text": string}\n')
return
}
text = parsed.text
} catch (e) {
res.writeHead(400, { 'Content-Type': 'text/plain' })
res.end(`synthesize: bad body — ${e instanceof Error ? e.message : String(e)}\n`)
return
}
if (!text.trim()) {
res.writeHead(400, { 'Content-Type': 'text/plain' })
res.end('synthesize: empty text\n')
return
}
const ttsRes = await fetch(`${ctx.httpBase}/voice/synthesize`, {
method: 'POST',
headers: {
Authorization: `Bearer ${ctx.opts.token}`,
'Content-Type': 'application/json',
Accept: 'audio/mpeg'
},
body: JSON.stringify({ text })
})
if (!ttsRes.ok || !ttsRes.body) {
const detail = await safeText(ttsRes)
res.writeHead(ttsRes.status, { 'Content-Type': 'text/plain' })
res.end(`relay /voice/synthesize ${ttsRes.status}: ${detail}\n`)
return
}
res.writeHead(200, {
'Content-Type': ttsRes.headers.get('content-type') ?? 'audio/mpeg',
'Cache-Control': 'no-store'
})
// Stream the audio body straight through — no buffering. fetch's body is a
// web ReadableStream; Readable.fromWeb adapts it to a Node stream.
Readable.fromWeb(ttsRes.body as unknown as Parameters<typeof Readable.fromWeb>[0]).pipe(res)
}
async function transcribeOnRelay(opts: {
httpBase: string
token: string
audio: Buffer
audioMime: string
signal: AbortSignal
}): Promise<string> {
const form = new FormData()
const blob = new Blob([new Uint8Array(opts.audio)], { type: opts.audioMime })
form.append('audio', blob, `voice.${extFor(opts.audioMime)}`)
const sttRes = await fetch(`${opts.httpBase}/voice/transcribe`, {
method: 'POST',
headers: { Authorization: `Bearer ${opts.token}`, Accept: 'application/json' },
body: form,
signal: opts.signal
})
const txt = await sttRes.text()
if (!sttRes.ok) {
throw new Error(`relay /voice/transcribe ${sttRes.status}: ${txt.slice(0, 240)}`)
}
let body: { text?: string; success?: boolean; error?: string }
try {
body = JSON.parse(txt)
} catch {
throw new Error(`relay /voice/transcribe returned non-JSON: ${txt.slice(0, 240)}`)
}
if (body.success === false) {
throw new Error(`relay /voice/transcribe error: ${body.error ?? 'unknown'}`)
}
return body.text ?? ''
}
function readBody(req: IncomingMessage, maxBytes: number): Promise<Buffer> {
return new Promise((resolve, reject) => {
const chunks: Buffer[] = []
let total = 0
req.on('data', (chunk: Buffer) => {
total += chunk.length
if (total > maxBytes) {
reject(new Error(`body exceeds ${maxBytes} bytes`))
req.destroy()
return
}
chunks.push(chunk)
})
req.on('end', () => resolve(Buffer.concat(chunks)))
req.on('error', reject)
})
}
async function safeText(res: Response): Promise<string> {
try { return (await res.text()).slice(0, 240) } catch { return '<no body>' }
}
function wsToHttp(url: string): string {
const trimmed = url.trim()
if (trimmed.startsWith('wss://')) return 'https://' + trimmed.slice('wss://'.length)
if (trimmed.startsWith('ws://')) return 'http://' + trimmed.slice('ws://'.length)
return trimmed
}
function extFor(mime: string): string {
const base = mime.split(';', 1)[0]?.trim().toLowerCase() ?? ''
if (base.includes('webm')) return 'webm'
if (base.includes('ogg')) return 'ogg'
if (base.includes('mp4') || base.includes('m4a') || base.includes('aac')) return 'm4a'
if (base.includes('wav')) return 'wav'
if (base.includes('mpeg') || base.includes('mp3')) return 'mp3'
return 'webm'
}
/** Open `url` in the user's default browser. Best-effort, non-blocking. */
export function openInBrowser(url: string): void {
try {
if (process.platform === 'win32') {
// `start` is a cmd.exe builtin. Empty title prevents the URL being
// interpreted as a window title.
spawn('cmd', ['/c', 'start', '""', url], { detached: true, stdio: 'ignore' }).unref()
} else if (process.platform === 'darwin') {
spawn('open', [url], { detached: true, stdio: 'ignore' }).unref()
} else {
spawn('xdg-open', [url], { detached: true, stdio: 'ignore' }).unref()
}
} catch {
/* user can paste the URL manually — printed by the caller */
}
}
+4817
View File
File diff suppressed because it is too large Load Diff

Some files were not shown because too many files have changed in this diff Show More