Compare commits
@@ -0,0 +1 @@
|
||||
*.sh text eol=lf
|
||||
@@ -0,0 +1,90 @@
|
||||
name: Bug report
|
||||
description: Report a reproducible problem in Hermes-Relay.
|
||||
title: "[Bug]: "
|
||||
labels: ["bug"]
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Before submitting, remove secrets, access tokens, real hostnames/IPs, private deployment names, and personal names. Public example IPs such as `192.168.1.100` are fine.
|
||||
|
||||
- type: dropdown
|
||||
id: area
|
||||
attributes:
|
||||
label: Affected area
|
||||
description: Pick the closest surface.
|
||||
options:
|
||||
- Android app
|
||||
- Standard Hermes chat or voice
|
||||
- Relay plugin or server
|
||||
- Desktop CLI or tray
|
||||
- Dashboard plugin
|
||||
- Docs or installer
|
||||
- CI, release, or packaging
|
||||
- Unsure
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: summary
|
||||
attributes:
|
||||
label: What happened?
|
||||
description: State the behavior you saw and what you expected instead.
|
||||
placeholder: |
|
||||
Observed:
|
||||
|
||||
Expected:
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: steps
|
||||
attributes:
|
||||
label: Reproduction steps
|
||||
description: Include the smallest sequence that reproduces the issue.
|
||||
placeholder: |
|
||||
1. Pair or configure...
|
||||
2. Open...
|
||||
3. Tap or run...
|
||||
4. See...
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: environment
|
||||
attributes:
|
||||
label: Environment
|
||||
description: Include only the fields that apply.
|
||||
value: |
|
||||
- Hermes-Relay version/tag:
|
||||
- Install surface: Google Play / sideload APK / local build / plugin / desktop CLI
|
||||
- Android device and OS:
|
||||
- hermes-agent version or commit:
|
||||
- Connection mode: LAN / Tailscale / public TLS / other
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: logs
|
||||
attributes:
|
||||
label: Sanitized logs, screenshots, or traces
|
||||
description: Paste the smallest useful log excerpt. Remove tokens, private URLs, hostnames, IPs, and user-identifying data.
|
||||
render: shell
|
||||
|
||||
- type: textarea
|
||||
id: upstream
|
||||
attributes:
|
||||
label: Upstream or standard-path notes
|
||||
description: If relevant, note whether this reproduces against unmodified upstream hermes-agent or only with the relay plugin enabled.
|
||||
|
||||
- type: checkboxes
|
||||
id: checklist
|
||||
attributes:
|
||||
label: Checklist
|
||||
options:
|
||||
- label: I searched existing issues first.
|
||||
required: true
|
||||
- label: I removed secrets, tokens, private infrastructure, and personal names.
|
||||
required: true
|
||||
- label: I included the affected version or install surface where known.
|
||||
required: true
|
||||
@@ -0,0 +1,11 @@
|
||||
blank_issues_enabled: true
|
||||
contact_links:
|
||||
- name: Security guidance
|
||||
url: https://github.com/Codename-11/hermes-relay/blob/main/docs/security.md
|
||||
about: Review the security model before posting sensitive vulnerability details publicly.
|
||||
- name: User documentation
|
||||
url: https://codename-11.github.io/hermes-relay/
|
||||
about: Read setup, pairing, remote access, and troubleshooting docs.
|
||||
- name: Contributing guide
|
||||
url: https://github.com/Codename-11/hermes-relay/blob/main/CONTRIBUTING.md
|
||||
about: Review local setup, branch, commit, changelog, and test conventions.
|
||||
@@ -0,0 +1,64 @@
|
||||
name: Documentation or setup issue
|
||||
description: Report unclear, stale, or missing docs and setup guidance.
|
||||
title: "[Docs]: "
|
||||
labels: ["documentation"]
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Use this for docs, installer, setup, release-note, or contribution-guide problems. Remove private hostnames/IPs, tokens, and personal names before posting.
|
||||
|
||||
- type: dropdown
|
||||
id: area
|
||||
attributes:
|
||||
label: Documentation area
|
||||
options:
|
||||
- README
|
||||
- User docs site
|
||||
- Android setup
|
||||
- Relay plugin setup
|
||||
- Desktop CLI or tray setup
|
||||
- Release notes or changelog
|
||||
- Contributor docs
|
||||
- Other
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: input
|
||||
id: location
|
||||
attributes:
|
||||
label: Page, file, or section
|
||||
description: Link the page or name the file and heading.
|
||||
placeholder: user-docs/guide/getting-started.md, README install section, etc.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: issue
|
||||
attributes:
|
||||
label: What is wrong or missing?
|
||||
description: Explain what was unclear, outdated, misleading, or absent.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: expected
|
||||
attributes:
|
||||
label: Suggested correction
|
||||
description: Optional. Include the wording, command, screenshot need, or structure that would help.
|
||||
|
||||
- type: textarea
|
||||
id: context
|
||||
attributes:
|
||||
label: Context
|
||||
description: Optional. Include the version, install path, device, or command you were following.
|
||||
|
||||
- type: checkboxes
|
||||
id: checklist
|
||||
attributes:
|
||||
label: Checklist
|
||||
options:
|
||||
- label: I checked that this is not already covered in current docs.
|
||||
required: true
|
||||
- label: I removed secrets, private hostnames/IPs, internal deployment names, and personal names.
|
||||
required: true
|
||||
@@ -0,0 +1,78 @@
|
||||
name: Feature request
|
||||
description: Propose a product, workflow, or platform improvement.
|
||||
title: "[Feature]: "
|
||||
labels: ["enhancement"]
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Keep requests focused on user-visible outcomes. Do not include private infrastructure, secrets, personal names, or branch/workspace plumbing.
|
||||
|
||||
- type: dropdown
|
||||
id: area
|
||||
attributes:
|
||||
label: Affected area
|
||||
options:
|
||||
- Android app
|
||||
- Standard Hermes chat or voice
|
||||
- Relay plugin or server
|
||||
- Desktop CLI or tray
|
||||
- Dashboard plugin
|
||||
- Docs or installer
|
||||
- CI, release, or packaging
|
||||
- Unsure
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: problem
|
||||
attributes:
|
||||
label: Problem or workflow
|
||||
description: What is hard, missing, slow, confusing, or unsafe today?
|
||||
placeholder: Describe the concrete user workflow this would improve.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: proposal
|
||||
attributes:
|
||||
label: Proposed behavior
|
||||
description: Describe the outcome, not just an implementation detail.
|
||||
placeholder: After this change, a user should be able to...
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: standard_path
|
||||
attributes:
|
||||
label: Standard upstream compatibility
|
||||
description: If this touches chat, voice, dashboard, API routes, or server behavior, note whether it can work against unmodified upstream hermes-agent.
|
||||
placeholder: This should work on vanilla upstream because... / This requires the relay plugin because...
|
||||
|
||||
- type: textarea
|
||||
id: alternatives
|
||||
attributes:
|
||||
label: Alternatives considered
|
||||
description: Optional. Mention current workarounds or related approaches.
|
||||
|
||||
- type: textarea
|
||||
id: acceptance
|
||||
attributes:
|
||||
label: Acceptance criteria
|
||||
description: What would make the request complete?
|
||||
placeholder: |
|
||||
- Users can...
|
||||
- The app/server handles...
|
||||
- Documentation covers...
|
||||
|
||||
- type: checkboxes
|
||||
id: checklist
|
||||
attributes:
|
||||
label: Checklist
|
||||
options:
|
||||
- label: I searched existing issues first.
|
||||
required: true
|
||||
- label: I described the user outcome and affected surface.
|
||||
required: true
|
||||
- label: I removed private infrastructure details and personal names.
|
||||
required: true
|
||||
@@ -6,11 +6,20 @@
|
||||
|
||||
-
|
||||
|
||||
## Verification
|
||||
|
||||
<!-- List the checks you ran, or explain why a check is not applicable. -->
|
||||
|
||||
-
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] `./gradlew assembleDebug` succeeds
|
||||
- [ ] `./gradlew test` passes
|
||||
- [ ] Tested on emulator or device (if UI change)
|
||||
- [ ] Target branch is `dev` unless this is a release PR
|
||||
- [ ] Android changes: lint and focused unit tests ran, or rationale is listed above
|
||||
- [ ] Server changes: focused `python -m unittest ...` checks ran, or rationale is listed above
|
||||
- [ ] Desktop changes: `npm run build` or a narrower documented check ran, or rationale is listed above
|
||||
- [ ] Docs/site changes: docs build or link check ran, or rationale is listed above
|
||||
- [ ] UI changes were tested on emulator/device or desktop surface when applicable
|
||||
- [ ] Commit messages follow [Conventional Commits](https://www.conventionalcommits.org/)
|
||||
- [ ] CHANGELOG.md updated (if user-facing)
|
||||
- [ ] No credentials or secrets in committed files
|
||||
- [ ] Public writing hygiene checked: no secrets, private infrastructure, personal names, or AI/process narration
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
# GitHub Copilot instructions — Hermes-Relay
|
||||
|
||||
This file exists so GitHub Copilot (which reads `.github/copilot-instructions.md`,
|
||||
not `AGENTS.md`) picks up the project's agent guidance.
|
||||
|
||||
**Read [AGENTS.md](../AGENTS.md) first — it is the single source of truth**
|
||||
for agent guidance: the entry point, the non-negotiables, and the public-repo
|
||||
writing hygiene. It links on to `CLAUDE.md` for the deep reference
|
||||
(architecture, upstream Hermes API, repository layout, per-language code style,
|
||||
the dev loop, and the Key Files map). Follow those; don't restate them here.
|
||||
|
||||
Quick non-negotiables (the full list and rationale are in `AGENTS.md`):
|
||||
|
||||
- **Standard path = vanilla upstream only.** The default no-plugin connection
|
||||
must work against unmodified upstream hermes-agent; server-side needs go
|
||||
through upstream PRs or the optional relay plugin, never fork patches.
|
||||
- **Conventional Commits**, `main`/`dev` branching — feature branches off
|
||||
`dev`, `--no-ff` merges, tags cut from `main`.
|
||||
- **Android:** Jetpack Compose (no XML), kotlinx.serialization (no Gson),
|
||||
OkHttp (no Ktor), `wss://` only; run `./gradlew lint` before pushing Kotlin.
|
||||
- **Public repo:** no personal names, no private infrastructure, no
|
||||
AI/assistant self-narration in committed prose.
|
||||
@@ -3,7 +3,14 @@
|
||||
# Runs on pushes to main/dev and on PRs targeting main/dev, scoped to
|
||||
# Android-affecting paths so Python-only changes don't spin up the JVM.
|
||||
#
|
||||
# Pipeline: lint -> build + test (parallel) -> upload artifacts
|
||||
# Pipeline: lint, build, and focused tests run concurrently. PRs build debug
|
||||
# APKs before merge; dev pushes keep lint/tests only to avoid duplicate
|
||||
# post-merge packaging. Main pushes keep APK artifacts.
|
||||
#
|
||||
# A release-build smoke (bundleRelease assembleRelease) runs on dev/main pushes
|
||||
# and on the dev→main release PR so release-only breakage (R8/minify rules,
|
||||
# resource shrinking, bundletool OOM) is caught BEFORE the android-v* tag,
|
||||
# instead of mid-release. It is debug-signed, so it needs no signing secrets.
|
||||
|
||||
name: CI — Android
|
||||
|
||||
@@ -38,11 +45,12 @@ concurrency:
|
||||
|
||||
jobs:
|
||||
# ──────────────────────────────────────────────
|
||||
# Android Lint — gate for build and test jobs
|
||||
# Android Lint
|
||||
# ──────────────────────────────────────────────
|
||||
lint:
|
||||
name: Lint (Android)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v6
|
||||
@@ -55,25 +63,20 @@ jobs:
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
with:
|
||||
cache-read-only: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
|
||||
|
||||
# Prefer ktlintCheck if configured; fall back to Android lint
|
||||
- name: Run lint checks
|
||||
run: |
|
||||
if ./gradlew tasks --all 2>/dev/null | grep -q "ktlintCheck"; then
|
||||
echo "Running ktlintCheck..."
|
||||
./gradlew ktlintCheck
|
||||
else
|
||||
echo "ktlintCheck not found, falling back to Android lint..."
|
||||
./gradlew lint
|
||||
fi
|
||||
- name: Run Android lint
|
||||
run: ./gradlew lint --console=plain
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Android Build — assembleDebug + upload APK
|
||||
# Android Build — assembleDebug for PRs and main pushes
|
||||
# ──────────────────────────────────────────────
|
||||
build:
|
||||
name: Build (Android)
|
||||
needs: lint
|
||||
if: ${{ github.event_name == 'pull_request' || github.ref == 'refs/heads/main' }}
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v6
|
||||
@@ -86,12 +89,15 @@ jobs:
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
with:
|
||||
cache-read-only: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
|
||||
|
||||
- name: Build debug APK
|
||||
run: ./gradlew assembleDebug
|
||||
run: ./gradlew assembleDebug --console=plain
|
||||
|
||||
- name: Upload debug APK
|
||||
uses: actions/upload-artifact@v7
|
||||
if: ${{ github.ref == 'refs/heads/main' }}
|
||||
with:
|
||||
name: debug-apk
|
||||
# Product flavors (googlePlay, sideload) nest APKs under
|
||||
@@ -109,7 +115,6 @@ jobs:
|
||||
# ──────────────────────────────────────────────
|
||||
test:
|
||||
name: Test (Android)
|
||||
needs: lint
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
# Advisory on dev, strict on main. Evaluates to false (= strict) for
|
||||
@@ -128,6 +133,8 @@ jobs:
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
with:
|
||||
cache-read-only: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
|
||||
|
||||
# The broad Gradle `test` aggregate currently hangs in deferred JVM test
|
||||
# suites tracked by issue #32. Keep CI release-relevant until that suite is
|
||||
@@ -136,15 +143,52 @@ jobs:
|
||||
- name: Run focused Android unit tests
|
||||
run: |
|
||||
./gradlew :app:testSideloadDebugUnitTest \
|
||||
--tests com.hermesandroid.relay.network.RelayUrlDeriverTest \
|
||||
--tests com.hermesandroid.relay.network.ArchitectureBoundaryTest \
|
||||
--tests com.hermesandroid.relay.network.relay.RelayUrlDeriverTest \
|
||||
--tests com.hermesandroid.relay.viewmodel.ConnectionSwitchTest \
|
||||
--console=plain
|
||||
|
||||
# Upload test reports even if tests fail, for debugging
|
||||
# Upload reports only for failures. Successful PR report uploads add
|
||||
# noticeable latency and are rarely inspected.
|
||||
- name: Upload test reports
|
||||
uses: actions/upload-artifact@v7
|
||||
if: always()
|
||||
if: failure()
|
||||
with:
|
||||
name: test-reports
|
||||
path: app/build/reports/tests/
|
||||
retention-days: 7
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Release build smoke — exercises the release variant the android-v* tag
|
||||
# build runs (./gradlew bundleRelease assembleRelease, both flavors), so
|
||||
# release-only breakage (R8/minify, resource shrinking, bundletool OOM) is
|
||||
# caught BEFORE the tag instead of mid-release. Debug-signed — no secrets,
|
||||
# so it also runs on fork PRs. Runs on dev/main pushes (early signal after
|
||||
# each merge) and on the dev→main release PR (hard pre-tag gate); skipped on
|
||||
# dev-targeted feature PRs to avoid re-running a ~12-min build per iteration.
|
||||
# ──────────────────────────────────────────────
|
||||
release-smoke:
|
||||
name: Release build smoke (Android)
|
||||
if: ${{ github.ref == 'refs/heads/dev' || github.ref == 'refs/heads/main' || (github.event_name == 'pull_request' && github.base_ref == 'main') }}
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 35
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set up JDK 17
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: 17
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
with:
|
||||
cache-read-only: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
|
||||
|
||||
# Mirrors release-android.yml's build step. No keystore is provided here,
|
||||
# so app/build.gradle.kts falls back to debug signing — fine for a build
|
||||
# smoke; the goal is to exercise the build, not to produce a shippable AAB.
|
||||
- name: Build release bundles + APKs (both flavors, debug-signed)
|
||||
run: ./gradlew bundleRelease assembleRelease --console=plain
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
# Hermes-Relay — Vanilla-Upstream Route Contract (ADR 34)
|
||||
#
|
||||
# Proves the Android *standard path* (no-plugin) route surface exists on
|
||||
# UNMODIFIED NousResearch/hermes-agent — the invariant CLAUDE.md asserts but
|
||||
# that was never tested. Source-parses upstream's declared routes (no server
|
||||
# boot, no pip install, no model keys); see scripts/check-upstream-route-contract.py
|
||||
# for the design + tradeoff (catches renamed/removed routes; not runtime auth).
|
||||
#
|
||||
# PR/push runs check a pinned ref (non-flaky); the weekly schedule tracks
|
||||
# upstream `main` as a drift siren so a route rename surfaces on our clock.
|
||||
|
||||
name: CI — Upstream Contract
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main, dev]
|
||||
paths:
|
||||
- "scripts/check-upstream-route-contract.py"
|
||||
- ".github/workflows/ci-contract.yml"
|
||||
- "app/src/main/kotlin/com/hermesandroid/relay/network/upstream/**"
|
||||
pull_request:
|
||||
branches: [main, dev]
|
||||
paths:
|
||||
- "scripts/check-upstream-route-contract.py"
|
||||
- ".github/workflows/ci-contract.yml"
|
||||
- "app/src/main/kotlin/com/hermesandroid/relay/network/upstream/**"
|
||||
schedule:
|
||||
- cron: "0 6 * * 1" # Mondays 06:00 UTC — upstream-drift siren (tracks main)
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
upstream_ref:
|
||||
description: "NousResearch/hermes-agent ref to check (branch, tag, or SHA)"
|
||||
required: false
|
||||
default: ""
|
||||
|
||||
concurrency:
|
||||
group: ci-contract-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
|
||||
|
||||
jobs:
|
||||
route-contract:
|
||||
name: Vanilla-upstream route contract
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout hermes-relay
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Resolve upstream ref
|
||||
id: ref
|
||||
run: |
|
||||
# PR/push runs use a known-good NousResearch/hermes-agent commit so
|
||||
# normal CI is stable. The weekly schedule below intentionally tracks
|
||||
# main as the upstream-drift siren.
|
||||
DEFAULT_REF="ef4b897a1843cd32c4f141f55db60f0f0602cc98"
|
||||
if [ "${{ github.event_name }}" = "schedule" ]; then
|
||||
REF="main" # weekly drift siren
|
||||
elif [ -n "${{ github.event.inputs.upstream_ref }}" ]; then
|
||||
REF="${{ github.event.inputs.upstream_ref }}" # manual override
|
||||
else
|
||||
REF="$DEFAULT_REF"
|
||||
fi
|
||||
echo "ref=$REF" >> "$GITHUB_OUTPUT"
|
||||
echo "Checking standard-path route contract against upstream ref: $REF"
|
||||
|
||||
- name: Checkout vanilla upstream (no plugin, no bootstrap)
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
repository: NousResearch/hermes-agent
|
||||
ref: ${{ steps.ref.outputs.ref }}
|
||||
path: _upstream
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Set up Python 3.11
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Assert upstream checkout is vanilla (no relay bootstrap/plugin)
|
||||
run: |
|
||||
if [ -e "_upstream/hermes_relay_bootstrap" ] || \
|
||||
[ -e "_upstream/plugin/hermes_relay_bootstrap" ] || \
|
||||
find _upstream -name "hermes_relay_bootstrap.pth" 2>/dev/null | grep -q .; then
|
||||
echo "FAIL: upstream checkout contains a relay bootstrap — not vanilla."; exit 1
|
||||
fi
|
||||
echo "OK: upstream checkout carries no relay plugin/bootstrap."
|
||||
|
||||
- name: Run route-surface contract
|
||||
run: python scripts/check-upstream-route-contract.py "_upstream"
|
||||
@@ -5,13 +5,11 @@ on:
|
||||
branches: [main, dev]
|
||||
paths:
|
||||
- "plugin/dashboard/**"
|
||||
- "scripts/check-server-version-sync.py"
|
||||
- ".github/workflows/ci-dashboard.yml"
|
||||
pull_request:
|
||||
branches: [main, dev]
|
||||
paths:
|
||||
- "plugin/dashboard/**"
|
||||
- "scripts/check-server-version-sync.py"
|
||||
- ".github/workflows/ci-dashboard.yml"
|
||||
|
||||
permissions:
|
||||
@@ -49,11 +47,15 @@ jobs:
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Verify server-owned version metadata
|
||||
run: python scripts/check-server-version-sync.py
|
||||
- name: Verify plugin-owned version metadata
|
||||
run: python scripts/check-plugin-version-sync.py
|
||||
|
||||
- name: Install dashboard API test deps
|
||||
run: pip install -r relay_server/requirements.txt fastapi httpx pytest requests
|
||||
# The suite imports the `plugin` package transitively: __init__ loads
|
||||
# android_tool/desktop_tool (`import requests`), and one test imports
|
||||
# `plugin.relay`, whose server.py needs `aiohttp` (+ pyyaml) from
|
||||
# relay_server/requirements.txt. fastapi+httpx cover plugin_api itself.
|
||||
run: pip install -r relay_server/requirements.txt fastapi httpx requests
|
||||
|
||||
- name: Run dashboard API tests
|
||||
run: python -m unittest plugin.dashboard.test_plugin_api
|
||||
|
||||
@@ -14,6 +14,10 @@ on:
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ci-desktop-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
|
||||
|
||||
jobs:
|
||||
typecheck-and-build:
|
||||
name: Type-check + build
|
||||
@@ -25,7 +29,7 @@ jobs:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
@@ -47,16 +51,8 @@ jobs:
|
||||
# prebuilt dist/ that references a source file that moved.
|
||||
run: node bin/hermes-relay.js --version
|
||||
|
||||
- name: Upload dist/
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: desktop-dist
|
||||
path: desktop/dist
|
||||
retention-days: 7
|
||||
|
||||
smoke-help:
|
||||
name: Smoke — --help + --version work on every target OS
|
||||
needs: typecheck-and-build
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
@@ -69,7 +65,7 @@ jobs:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
# Hermes-Relay — Python Server CI Pipeline
|
||||
# Hermes-Relay — Plugin CI Pipeline
|
||||
#
|
||||
# Runs on pushes to main/dev and on PRs targeting main/dev, scoped to
|
||||
# server-affecting paths so Android-only changes don't spin up the
|
||||
# plugin-affecting paths so Android-only changes don't spin up the
|
||||
# Python toolchain.
|
||||
#
|
||||
# Pipeline: syntax-check -> focused server tests
|
||||
# Pipeline: syntax-check and focused plugin tests run concurrently.
|
||||
|
||||
name: CI — Server
|
||||
name: CI — Plugin
|
||||
|
||||
on:
|
||||
push:
|
||||
@@ -17,18 +17,17 @@ on:
|
||||
- "plugin/cli.py"
|
||||
- "plugin/pair.py"
|
||||
- "plugin/plugin.yaml"
|
||||
- "plugin/dashboard/manifest.json"
|
||||
- "plugin/dashboard/package.json"
|
||||
- "plugin/dashboard/package-lock.json"
|
||||
- "plugin/relay/**"
|
||||
- "plugin/tools/**"
|
||||
- "plugin/tests/**"
|
||||
- "relay_server/**"
|
||||
- "hermes_relay_bootstrap/**"
|
||||
- "pyproject.toml"
|
||||
- "scripts/check-plugin-version-sync.py"
|
||||
- "scripts/check-server-version-sync.py"
|
||||
- "scripts/bump-plugin-version.sh"
|
||||
- "scripts/bump-server-version.sh"
|
||||
- ".github/workflows/ci-server.yml"
|
||||
- ".github/workflows/ci-plugin.yml"
|
||||
pull_request:
|
||||
branches: [main, dev]
|
||||
paths:
|
||||
@@ -37,27 +36,26 @@ on:
|
||||
- "plugin/cli.py"
|
||||
- "plugin/pair.py"
|
||||
- "plugin/plugin.yaml"
|
||||
- "plugin/dashboard/manifest.json"
|
||||
- "plugin/dashboard/package.json"
|
||||
- "plugin/dashboard/package-lock.json"
|
||||
- "plugin/relay/**"
|
||||
- "plugin/tools/**"
|
||||
- "plugin/tests/**"
|
||||
- "relay_server/**"
|
||||
- "hermes_relay_bootstrap/**"
|
||||
- "pyproject.toml"
|
||||
- "scripts/check-plugin-version-sync.py"
|
||||
- "scripts/check-server-version-sync.py"
|
||||
- "scripts/bump-plugin-version.sh"
|
||||
- "scripts/bump-server-version.sh"
|
||||
- ".github/workflows/ci-server.yml"
|
||||
- ".github/workflows/ci-plugin.yml"
|
||||
|
||||
# Cancel in-progress runs for the same branch/PR, but let main and dev finish
|
||||
concurrency:
|
||||
group: ci-server-${{ github.ref }}
|
||||
group: ci-plugin-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.ref != 'refs/heads/main' && github.ref != 'refs/heads/dev' }}
|
||||
|
||||
jobs:
|
||||
# ──────────────────────────────────────────────
|
||||
# Python Server — py_compile syntax sanity
|
||||
# Python Plugin — py_compile syntax sanity
|
||||
# ──────────────────────────────────────────────
|
||||
syntax-check:
|
||||
name: Syntax check (Python)
|
||||
@@ -72,10 +70,7 @@ jobs:
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install -r relay_server/requirements.txt
|
||||
|
||||
- name: Syntax check (server/plugin.relay — canonical location)
|
||||
- name: Syntax check (plugin relay — canonical location)
|
||||
run: |
|
||||
python -m py_compile plugin/relay/server.py
|
||||
python -m py_compile plugin/relay/channels/terminal.py
|
||||
@@ -87,19 +82,18 @@ jobs:
|
||||
- name: Syntax check (relay_server shim)
|
||||
run: python -m py_compile relay_server/__init__.py relay_server/__main__.py
|
||||
|
||||
- name: Validate Server version metadata
|
||||
run: python scripts/check-server-version-sync.py
|
||||
- name: Validate Plugin version metadata
|
||||
run: python scripts/check-plugin-version-sync.py
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Python Server — focused route/auth/session tests
|
||||
# Python Plugin — focused route/auth/session tests
|
||||
#
|
||||
# Tests are ADVISORY on dev (push or PR) so WIP commits don't block the
|
||||
# merge queue. Strict on main — the dev → main release-merge PR surfaces
|
||||
# any real failures before release.
|
||||
# ──────────────────────────────────────────────
|
||||
unit-tests:
|
||||
name: Focused Server tests (Python)
|
||||
needs: syntax-check
|
||||
name: Focused Plugin tests (Python)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
# Advisory on dev, strict on main. Evaluates to false (= strict) for
|
||||
@@ -120,7 +114,7 @@ jobs:
|
||||
pip install -r relay_server/requirements.txt
|
||||
pip install pytest responses
|
||||
|
||||
- name: Run focused Server tests
|
||||
- name: Run focused Plugin tests
|
||||
run: |
|
||||
python -m pytest \
|
||||
plugin/tests/test_relay_security.py \
|
||||
@@ -2,7 +2,7 @@
|
||||
# branch protection on `main` has a check name it can rely on, regardless
|
||||
# of which paths the PR touches.
|
||||
#
|
||||
# Why this exists. The other CI workflows (`ci-android.yml`, `ci-server.yml`,
|
||||
# Why this exists. The other CI workflows (`ci-android.yml`, `ci-plugin.yml`,
|
||||
# `ci-desktop.yml`) are scoped via `paths:` filters so a docs-only or
|
||||
# desktop-only PR doesn't spin up the Android toolchain. Branch protection's
|
||||
# "required status checks" treat a check that doesn't run as failing — so
|
||||
|
||||
@@ -30,6 +30,11 @@ jobs:
|
||||
# alone — a title-format match (e.g. "release:") is fragile and silently
|
||||
# let a "Release v1.0.0 …"-titled PR run the full review and time out.
|
||||
IS_RELEASE_PR: ${{ github.event.pull_request.base.ref == 'main' && github.event.pull_request.head.ref == 'dev' }}
|
||||
# Bot-authored PRs such as Dependabot do not receive the same secret
|
||||
# surface as human-authored PRs, and Claude Code rejects bot actors unless
|
||||
# explicitly allow-listed. Keep the required check green with a no-op and
|
||||
# rely on the dependency CI/status checks for those PRs.
|
||||
IS_BOT_PR: ${{ github.event.pull_request.user.type == 'Bot' }}
|
||||
|
||||
steps:
|
||||
- name: Skip aggregate release PR review
|
||||
@@ -38,14 +43,40 @@ jobs:
|
||||
echo "Skipping Claude Code Review for aggregate dev -> main release PR."
|
||||
echo "Feature work is reviewed before it lands on dev; release PRs are gated by CI and release metadata checks."
|
||||
|
||||
- name: Skip bot-authored PR review
|
||||
if: env.IS_BOT_PR == 'true'
|
||||
run: |
|
||||
echo "Skipping Claude Code Review for bot-authored PR."
|
||||
echo "Bot PRs are gated by Required checks plus their path-specific CI jobs."
|
||||
|
||||
- name: Checkout repository
|
||||
if: env.IS_RELEASE_PR != 'true'
|
||||
if: env.IS_RELEASE_PR != 'true' && env.IS_BOT_PR != 'true'
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 1
|
||||
# Depth 2 includes the pull_request merge commit's first parent, which
|
||||
# lets the next step detect whether this PR changes the workflow file.
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Detect Claude review workflow changes
|
||||
if: env.IS_RELEASE_PR != 'true' && env.IS_BOT_PR != 'true'
|
||||
id: changed-workflow
|
||||
shell: bash
|
||||
run: |
|
||||
if git rev-parse --verify HEAD^1 >/dev/null 2>&1 &&
|
||||
git diff --name-only HEAD^1 HEAD | grep -Fxq ".github/workflows/claude-code-review.yml"; then
|
||||
echo "claude_review_workflow=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "claude_review_workflow=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Skip Claude review workflow self-change
|
||||
if: env.IS_RELEASE_PR != 'true' && env.IS_BOT_PR != 'true' && steps.changed-workflow.outputs.claude_review_workflow == 'true'
|
||||
run: |
|
||||
echo "Skipping Claude Code Review because this PR changes the review workflow itself."
|
||||
echo "The Claude action requires this workflow file to match the default branch before it can exchange the app token."
|
||||
|
||||
- name: Run Claude Code Review
|
||||
if: env.IS_RELEASE_PR != 'true'
|
||||
if: env.IS_RELEASE_PR != 'true' && env.IS_BOT_PR != 'true' && steps.changed-workflow.outputs.claude_review_workflow != 'true'
|
||||
timeout-minutes: 15
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@v1
|
||||
|
||||
@@ -36,14 +36,18 @@ jobs:
|
||||
fetch-depth: 0 # Full history for lastUpdated timestamps
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: 20
|
||||
# Node 24 ships npm 11, matching the npm that generates
|
||||
# user-docs/package-lock.json. On npm 10 (Node 20), `npm ci` rejects
|
||||
# the lock over the optional `search-insights` peer dep of bundled
|
||||
# docsearch. Keep this aligned with the npm used to write the lock.
|
||||
node-version: 24
|
||||
cache: npm
|
||||
cache-dependency-path: user-docs/package-lock.json
|
||||
|
||||
- name: Install dependencies
|
||||
run: npm install
|
||||
run: npm ci
|
||||
working-directory: user-docs
|
||||
|
||||
- name: Build VitePress site
|
||||
@@ -51,10 +55,10 @@ jobs:
|
||||
working-directory: user-docs
|
||||
|
||||
- name: Setup Pages
|
||||
uses: actions/configure-pages@v5
|
||||
uses: actions/configure-pages@v6
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
uses: actions/upload-pages-artifact@v5
|
||||
with:
|
||||
path: user-docs/.vitepress/dist
|
||||
|
||||
@@ -68,4 +72,4 @@ jobs:
|
||||
steps:
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deployment
|
||||
uses: actions/deploy-pages@v4
|
||||
uses: actions/deploy-pages@v5
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
name: Play Store Listing
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- "assets/screenshots/**"
|
||||
- "assets/play-store-icon-512.png"
|
||||
- "assets/play-store-feature-1024x500.png"
|
||||
- "docs/media/screenshots.json"
|
||||
- "app/src/googlePlay/play/default-language.txt"
|
||||
- "app/src/googlePlay/play/listings/**"
|
||||
- "scripts/screenshots.py"
|
||||
- ".github/workflows/play-listing.yml"
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
- dev
|
||||
paths:
|
||||
- "assets/screenshots/**"
|
||||
- "assets/play-store-icon-512.png"
|
||||
- "assets/play-store-feature-1024x500.png"
|
||||
- "docs/media/screenshots.json"
|
||||
- "app/src/googlePlay/play/default-language.txt"
|
||||
- "app/src/googlePlay/play/listings/**"
|
||||
- "scripts/screenshots.py"
|
||||
- ".github/workflows/play-listing.yml"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
publish_listing:
|
||||
description: "Publish Play Store listing metadata after validation"
|
||||
required: true
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
validate:
|
||||
name: Validate Listing Assets
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Install image tooling
|
||||
run: python -m pip install --upgrade rich Pillow
|
||||
|
||||
- name: Validate screenshots and listing metadata
|
||||
run: python scripts/screenshots.py validate
|
||||
|
||||
publish-listing:
|
||||
name: Publish Listing Metadata
|
||||
needs: validate
|
||||
# Auto-publish the listing when its assets change on `main` (the release
|
||||
# branch; the path filters above already scope this to screenshot/graphic/
|
||||
# text changes). `dev` pushes and PRs validate only. A manual dispatch with
|
||||
# `publish_listing` still works as an on-demand republish.
|
||||
if: >-
|
||||
${{ (github.event_name == 'workflow_dispatch' && inputs.publish_listing)
|
||||
|| (github.event_name == 'push' && github.ref == 'refs/heads/main') }}
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
|
||||
- name: Set up JDK 17
|
||||
uses: actions/setup-java@v5
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: 17
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
with:
|
||||
cache-read-only: false
|
||||
|
||||
- name: Write Play service account
|
||||
id: sa
|
||||
env:
|
||||
PLAY_SERVICE_ACCOUNT_JSON: ${{ secrets.PLAY_SERVICE_ACCOUNT_JSON }}
|
||||
run: |
|
||||
if [ -z "$PLAY_SERVICE_ACCOUNT_JSON" ]; then
|
||||
# Skip gracefully (no red CI) when the secret isn't configured — e.g.
|
||||
# an auto-publish push to main before the service account is set up.
|
||||
echo "::notice::PLAY_SERVICE_ACCOUNT_JSON not configured — skipping listing publish."
|
||||
echo "configured=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
printf '%s' "$PLAY_SERVICE_ACCOUNT_JSON" > play-service-account.json
|
||||
echo "configured=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Publish Play Store listing
|
||||
if: ${{ steps.sa.outputs.configured == 'true' }}
|
||||
run: ./gradlew publishGooglePlayReleaseListing
|
||||
|
||||
- name: Remove Play service account
|
||||
if: always()
|
||||
run: rm -f play-service-account.json
|
||||
@@ -3,7 +3,7 @@
|
||||
# Triggered when an Android release tag (android-v*) is pushed.
|
||||
# Validates the tag matches the app version in libs.versions.toml,
|
||||
# runs focused Android checks, builds release APK/AAB artifacts, and creates a
|
||||
# GitHub Release. Server/Python package releases use server-v* tags.
|
||||
# GitHub Release. Plugin/Python package releases use plugin-v* tags.
|
||||
|
||||
name: Release Android
|
||||
|
||||
@@ -60,9 +60,8 @@ jobs:
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
|
||||
- name: Build debug APK
|
||||
run: ./gradlew assembleDebug
|
||||
with:
|
||||
cache-read-only: false
|
||||
|
||||
# Keep the tag release gate aligned with CI — Android's broad Gradle
|
||||
# `test` aggregate currently hangs in deferred JVM suites tracked by
|
||||
@@ -90,6 +89,8 @@ jobs:
|
||||
|
||||
- name: Setup Gradle
|
||||
uses: gradle/actions/setup-gradle@v6
|
||||
with:
|
||||
cache-read-only: false
|
||||
|
||||
- name: Decode release keystore
|
||||
env:
|
||||
@@ -151,6 +152,38 @@ jobs:
|
||||
app/build/outputs/bundle/*Release/*.aab
|
||||
app/build/outputs/SHA256SUMS.txt
|
||||
|
||||
- name: Upload to Play Console (production draft)
|
||||
env:
|
||||
PLAY_SERVICE_ACCOUNT_JSON: ${{ secrets.PLAY_SERVICE_ACCOUNT_JSON }}
|
||||
HERMES_KEYSTORE_PASSWORD: ${{ secrets.HERMES_KEYSTORE_PASSWORD }}
|
||||
HERMES_KEY_ALIAS: ${{ secrets.HERMES_KEY_ALIAS }}
|
||||
HERMES_KEY_PASSWORD: ${{ secrets.HERMES_KEY_PASSWORD }}
|
||||
# Runs only when the Play service-account secret is configured AND this is
|
||||
# a stable tag (prereleases — versions containing a dash — are skipped so
|
||||
# an `-rc.N` build never lands on the production listing). HERMES_KEYSTORE_PATH
|
||||
# was exported into $GITHUB_ENV by the "Decode release keystore" step above
|
||||
# and persists across steps in this job, so the AAB is release-signed.
|
||||
#
|
||||
# `publishGooglePlayReleaseBundle` is the flavor-scoped task — only the
|
||||
# googlePlay AAB is uploaded (sideload is disabled via playConfigs in
|
||||
# app/build.gradle.kts). The play{} block pins releaseStatus = DRAFT, so the
|
||||
# build lands on the Production track as a DRAFT: CI does the upload, a human
|
||||
# clicks "Start rollout" in Play Console. A bad tag can never auto-go-live.
|
||||
if: ${{ env.PLAY_SERVICE_ACCOUNT_JSON != '' && !contains(needs.validate.outputs.version, '-') }}
|
||||
run: |
|
||||
printf '%s' "$PLAY_SERVICE_ACCOUNT_JSON" > play-service-account.json
|
||||
./gradlew publishGooglePlayReleaseBundle --track=production
|
||||
rm -f play-service-account.json
|
||||
|
||||
- name: Play upload skipped (no secret)
|
||||
env:
|
||||
PLAY_SERVICE_ACCOUNT_JSON: ${{ secrets.PLAY_SERVICE_ACCOUNT_JSON }}
|
||||
if: ${{ env.PLAY_SERVICE_ACCOUNT_JSON == '' }}
|
||||
run: |
|
||||
echo "ℹ️ PLAY_SERVICE_ACCOUNT_JSON not set — skipped Play Console upload." \
|
||||
"GitHub Release artifacts are still published; upload to Play manually" \
|
||||
"(see RELEASE.md §5)." >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Release summary
|
||||
env:
|
||||
HERMES_KEYSTORE_BASE64: ${{ secrets.HERMES_KEYSTORE_BASE64 }}
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
name: Release Desktop
|
||||
name: Release CLI
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ['desktop-v*']
|
||||
tags: ['cli-v*']
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
@@ -18,7 +18,7 @@ jobs:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup Node.js (for npm ci + tsc)
|
||||
uses: actions/setup-node@v4
|
||||
uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
@@ -89,7 +89,7 @@ jobs:
|
||||
- name: Upload CLI release assets
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: desktop-cli-release
|
||||
name: cli-binaries
|
||||
path: |
|
||||
desktop/dist/bin/hermes-relay-win-x64.exe
|
||||
desktop/dist/bin/hermes-relay-linux-x64
|
||||
@@ -147,10 +147,13 @@ jobs:
|
||||
- name: Smoke-test tray exe launch
|
||||
shell: pwsh
|
||||
run: |
|
||||
$home = Join-Path $env:RUNNER_TEMP 'hermes-tray-smoke-home'
|
||||
New-Item -ItemType Directory -Force -Path $home | Out-Null
|
||||
$env:USERPROFILE = $home
|
||||
$env:HOME = $home
|
||||
# $HOME is a read-only automatic variable in PowerShell (names are
|
||||
# case-insensitive), so use a distinct scratch name; only the
|
||||
# $env:HOME / $env:USERPROFILE environment vars are writable.
|
||||
$smokeHome = Join-Path $env:RUNNER_TEMP 'hermes-tray-smoke-home'
|
||||
New-Item -ItemType Directory -Force -Path $smokeHome | Out-Null
|
||||
$env:USERPROFILE = $smokeHome
|
||||
$env:HOME = $smokeHome
|
||||
$proc = Start-Process -FilePath tray/src-tauri/target/release/hermes-relay-desktop.exe -WindowStyle Hidden -PassThru
|
||||
Start-Sleep -Seconds 5
|
||||
if ($proc.HasExited) { throw "tray app exited early with code $($proc.ExitCode)" }
|
||||
@@ -160,7 +163,7 @@ jobs:
|
||||
- name: Upload Windows tray release asset
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: desktop-windows-tray-release
|
||||
name: cli-windows-tray-installer
|
||||
path: desktop/dist/tray/hermes-relay-desktop-windows-x64-setup.exe
|
||||
retention-days: 7
|
||||
|
||||
@@ -171,9 +174,13 @@ jobs:
|
||||
- build-cli-binaries
|
||||
- build-windows-tray-installer
|
||||
steps:
|
||||
- name: Extract desktop version
|
||||
# Needed so CLI_RELEASE_NOTES.md is available to render into the release body
|
||||
# (the other publish-release steps only consume downloaded build artifacts).
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Extract CLI version
|
||||
id: version
|
||||
run: echo "version=${GITHUB_REF_NAME#desktop-v}" >> "$GITHUB_OUTPUT"
|
||||
run: echo "version=${GITHUB_REF_NAME#cli-v}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- uses: actions/download-artifact@v4
|
||||
with:
|
||||
@@ -188,54 +195,31 @@ jobs:
|
||||
| sed -E 's#release-assets/[^/]+/##' > release-assets/SHA256SUMS.txt
|
||||
cat release-assets/SHA256SUMS.txt
|
||||
|
||||
# Render CLI_RELEASE_NOTES.md (hand-written per release) into the GitHub
|
||||
# Release body. __VERSION__ = bare version (0.3.0), __TAG__ = full tag
|
||||
# (cli-v0.3.0) so the install/pin commands stay accurate without manual edits.
|
||||
- name: Render release notes
|
||||
env:
|
||||
VERSION: ${{ steps.version.outputs.version }}
|
||||
TAG: ${{ github.ref_name }}
|
||||
run: |
|
||||
sed -e "s/__VERSION__/${VERSION}/g" -e "s/__TAG__/${TAG}/g" \
|
||||
CLI_RELEASE_NOTES.md > cli_release_notes_rendered.md
|
||||
echo "=== rendered release body ===" && cat cli_release_notes_rendered.md
|
||||
|
||||
- name: Publish GitHub Release
|
||||
uses: softprops/action-gh-release@v3
|
||||
with:
|
||||
name: Hermes-Relay-Desktop v${{ steps.version.outputs.version }}
|
||||
name: Hermes-Relay-CLI v${{ steps.version.outputs.version }}
|
||||
tag_name: ${{ github.ref_name }}
|
||||
draft: false
|
||||
prerelease: ${{ contains(steps.version.outputs.version, 'alpha') || contains(steps.version.outputs.version, 'beta') || contains(steps.version.outputs.version, 'rc') }}
|
||||
fail_on_unmatched_files: true
|
||||
body: |
|
||||
# Hermes-Relay-Desktop v${{ steps.version.outputs.version }}
|
||||
|
||||
**Experimental phase.** Assets are unsigned - Windows SmartScreen and macOS Gatekeeper will warn on first launch. Windows now ships a tray installer as the primary desktop surface; CLI binaries remain available for terminal/headless use and for macOS/Linux.
|
||||
|
||||
## Install
|
||||
|
||||
**Windows tray app (PowerShell):**
|
||||
```powershell
|
||||
irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
|
||||
```
|
||||
|
||||
**Windows CLI only:**
|
||||
```powershell
|
||||
$env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
|
||||
```
|
||||
|
||||
**macOS / Linux CLI:**
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.sh | sh
|
||||
```
|
||||
|
||||
Pin this specific release with `HERMES_RELAY_VERSION=${{ github.ref_name }}`.
|
||||
|
||||
## Verify
|
||||
|
||||
```text
|
||||
hermes-relay --version
|
||||
hermes-relay pair --remote ws://<host>:8767
|
||||
hermes-relay shell
|
||||
```
|
||||
|
||||
Open **Hermes Relay Desktop** from the Windows Start menu for tray pairing, devices, task log, settings, pause, and emergency stop.
|
||||
|
||||
See [Desktop docs](https://codename-11.github.io/hermes-relay/desktop/) for full usage.
|
||||
|
||||
body_path: cli_release_notes_rendered.md
|
||||
files: |
|
||||
release-assets/desktop-cli-release/hermes-relay-win-x64.exe
|
||||
release-assets/desktop-cli-release/hermes-relay-linux-x64
|
||||
release-assets/desktop-cli-release/hermes-relay-darwin-x64
|
||||
release-assets/desktop-cli-release/hermes-relay-darwin-arm64
|
||||
release-assets/desktop-windows-tray-release/hermes-relay-desktop-windows-x64-setup.exe
|
||||
release-assets/cli-binaries/hermes-relay-win-x64.exe
|
||||
release-assets/cli-binaries/hermes-relay-linux-x64
|
||||
release-assets/cli-binaries/hermes-relay-darwin-x64
|
||||
release-assets/cli-binaries/hermes-relay-darwin-arm64
|
||||
release-assets/cli-windows-tray-installer/hermes-relay-desktop-windows-x64-setup.exe
|
||||
release-assets/SHA256SUMS.txt
|
||||
@@ -1,16 +1,16 @@
|
||||
name: Release Server
|
||||
name: Release Plugin
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "server-v*"
|
||||
- "plugin-v*"
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
jobs:
|
||||
validate:
|
||||
name: Validate Server release
|
||||
name: Validate Plugin release
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
outputs:
|
||||
@@ -20,15 +20,15 @@ jobs:
|
||||
|
||||
- name: Extract version from tag
|
||||
id: version
|
||||
run: echo "version=${GITHUB_REF#refs/tags/server-v}" >> "$GITHUB_OUTPUT"
|
||||
run: echo "version=${GITHUB_REF#refs/tags/plugin-v}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Verify Server version sync
|
||||
run: python scripts/check-server-version-sync.py --expect "$TAG_VERSION"
|
||||
- name: Verify Plugin version sync
|
||||
run: python scripts/check-plugin-version-sync.py --expect "$TAG_VERSION"
|
||||
env:
|
||||
TAG_VERSION: ${{ steps.version.outputs.version }}
|
||||
|
||||
test:
|
||||
name: Test Server package
|
||||
name: Test Plugin package
|
||||
needs: validate
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
@@ -55,7 +55,7 @@ jobs:
|
||||
python -m py_compile plugin/tools/desktop_tool.py
|
||||
python -m py_compile relay_server/__init__.py relay_server/__main__.py
|
||||
|
||||
- name: Run focused Server tests
|
||||
- name: Run focused Plugin tests
|
||||
run: |
|
||||
python -m pytest \
|
||||
plugin/tests/test_relay_security.py \
|
||||
@@ -63,7 +63,7 @@ jobs:
|
||||
plugin/tests/test_session_grants.py
|
||||
|
||||
package:
|
||||
name: Build and publish Server package
|
||||
name: Build and publish Plugin package
|
||||
needs: [validate, test]
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
@@ -86,32 +86,25 @@ jobs:
|
||||
sha256sum * > SHA256SUMS.txt
|
||||
cat SHA256SUMS.txt
|
||||
|
||||
# Render PLUGIN_RELEASE_NOTES.md (hand-written per release) into the GitHub
|
||||
# Release body, substituting the version token so the Install command stays
|
||||
# accurate without a manual edit. The file is the single source of the notes;
|
||||
# see RELEASE.md "Plugin / Python package release".
|
||||
- name: Render release notes
|
||||
env:
|
||||
VERSION: ${{ needs.validate.outputs.version }}
|
||||
run: |
|
||||
sed "s/__VERSION__/${VERSION}/g" PLUGIN_RELEASE_NOTES.md > release_notes_rendered.md
|
||||
echo "=== rendered release body ===" && cat release_notes_rendered.md
|
||||
|
||||
- name: Publish GitHub Release
|
||||
uses: softprops/action-gh-release@v3
|
||||
with:
|
||||
name: Hermes-Relay-Server v${{ needs.validate.outputs.version }}
|
||||
tag_name: server-v${{ needs.validate.outputs.version }}
|
||||
name: Hermes-Relay-Plugin v${{ needs.validate.outputs.version }}
|
||||
tag_name: plugin-v${{ needs.validate.outputs.version }}
|
||||
prerelease: ${{ contains(needs.validate.outputs.version, '-') }}
|
||||
fail_on_unmatched_files: true
|
||||
body: |
|
||||
# Hermes-Relay-Server v${{ needs.validate.outputs.version }}
|
||||
|
||||
This release contains the server/Python plugin package.
|
||||
Android releases use `android-v*` tags. Desktop releases use
|
||||
`desktop-v*` tags. Historical server releases before this lane
|
||||
rename used `relay-v*` tags.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install hermes-relay==${{ needs.validate.outputs.version }}
|
||||
```
|
||||
|
||||
## Verify
|
||||
|
||||
```bash
|
||||
python -m relay_server --help
|
||||
```
|
||||
body_path: release_notes_rendered.md
|
||||
files: |
|
||||
dist/*.whl
|
||||
dist/*.tar.gz
|
||||
@@ -26,10 +26,15 @@ local.properties
|
||||
/app/build/
|
||||
/relay-core/build/
|
||||
/relay-ui/build/
|
||||
/ui-preview/build/
|
||||
/quest/build/
|
||||
/app/release/
|
||||
*.apk
|
||||
*.aab
|
||||
|
||||
# Scratch / working directory (local pet packs, generated test assets, etc.)
|
||||
/tmp/
|
||||
/build-*.log
|
||||
*.jks
|
||||
*.keystore
|
||||
/captures
|
||||
|
||||
@@ -13,11 +13,12 @@ then `docs/spec.md` and `docs/decisions.md`.
|
||||
- Release process → **[RELEASE.md](RELEASE.md)**
|
||||
- Contributor setup → **[CONTRIBUTING.md](CONTRIBUTING.md)**
|
||||
- `android_*` toolset + MCP → **[docs/mcp-tooling.md](docs/mcp-tooling.md)**
|
||||
- Follow-ups / deferred work / known gaps → **[TODO.md](TODO.md)** (the single home for "what's next" — never DEVLOG, never scattered code comments)
|
||||
|
||||
## Non-negotiables (the short list)
|
||||
|
||||
- **Standard path = vanilla upstream only.** The default (no-plugin) connection —
|
||||
chat via the API server, standard voice via the Hermes dashboard — must work
|
||||
- **Vanilla Hermes path = upstream-only.** The default (no-plugin) connection —
|
||||
chat via the API server, Vanilla Hermes voice via the Hermes dashboard — must work
|
||||
against unmodified upstream hermes-agent. Server-side needs go through upstream
|
||||
PRs or the optional relay plugin, never fork patches.
|
||||
- **Verify endpoints against upstream** (`gateway/platforms/api_server.py` /
|
||||
@@ -26,6 +27,10 @@ then `docs/spec.md` and `docs/decisions.md`.
|
||||
`--no-ff` merges, version bumps at release-prep on `dev`, tags cut from `main`.
|
||||
- **Android:** Jetpack Compose only (no XML), kotlinx.serialization (no Gson),
|
||||
OkHttp (no Ktor), `wss://` only. Run `./gradlew lint` before pushing Kotlin.
|
||||
- **Plugin (Python 3.11+):** aiohttp + asyncio (no threading), type hints
|
||||
everywhere, structured `logging` (no `print`). **Desktop CLI (Node ≥21):**
|
||||
zero runtime deps, strict TS + ES modules, ship compiled `dist/`. Full
|
||||
per-language style and the dev loop live in CLAUDE.md → "Code Style".
|
||||
|
||||
## Public-repo writing hygiene
|
||||
|
||||
|
||||
@@ -6,10 +6,116 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/), and this
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- **Desktop CLI: `hermes-relay audit`.** Shows what the remote agent has actually run on this machine through the desktop tools — tool, status, and a short detail per call — read from a local log, no network or auth. Answers "what did the agent just do?" at a glance.
|
||||
- **Desktop CLI: `hermes-relay relay`.** Inspect the relay server itself: `relay info` (version, uptime, sessions — on the relay host), `relay security` (runtime auth toggles), and `relay context` (audit the system-prompt context the relay injects into the agent, which works from a remote machine with your session).
|
||||
- **Desktop CLI: background daemon.** `hermes-relay daemon start` runs the headless tool router in the background (no console window, survives closing the terminal), with `daemon stop` and `daemon status` to manage it. `daemon status` reports state, uptime, relay, and advertised-tool count; bare `daemon` still runs in the foreground. Logs go to `~/.hermes/daemon.log`.
|
||||
- **Desktop CLI: per-command help.** Every subcommand now answers `--help`, and `devices`/`sessions`/`plugins`/`voice`/`relay` print their own usage (sub-commands, flags, examples) instead of a terse "unknown sub-verb".
|
||||
- **Desktop CLI: startup banner.** A slim "Hermes Relay" wordmark shows atop `--help`, the first-run welcome, and the chat REPL — and `hermes-relay logo` prints it on demand. Suppressed for piped/`--json`/`--no-color` output.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Desktop CLI: visual + ergonomics refresh.** A single color theme across the CLI, aligned tables for `devices`/`sessions`, status dots for on/off states, and progress spinners for slow operations (the multi-endpoint pairing probe and the gateway connect) so nothing looks hung. Errors now suggest the fix (e.g. re-pair on auth failure).
|
||||
- **Desktop CLI: smoother pairing.** The multi-endpoint probe shows per-endpoint progress and latency; a near-expiry session warns before it fails and prints the exact re-pair command; and a bare `ws://host` (no port) defaults to `:8767`.
|
||||
- **Desktop CLI: voice + consent transparency.** `voice` now surfaces enhanced-voice capabilities (Gemini tone tags / persona, xAI speech tags); the desktop-tool consent prompt is clear that it persists per relay and points at `hermes-relay audit`; and computer-use's observe → grant → act flow is documented in `--help`.
|
||||
- **Crash reports can be shared without GitHub.** The crash dialog now has a **Share** action alongside Copy and Report, handing the full report to the system share sheet (email, chat apps, notes, Drive). This covers users without a GitHub account and sideload installs that Play vitals never sees. Every outbound path stays user-initiated — nothing is sent automatically.
|
||||
|
||||
## [1.2.0] - 2026-06-20
|
||||
|
||||
### Added
|
||||
|
||||
- **Sensitive-media classification (relay).** The relay teaches the agent — server-side, via a removable system-prompt block — to mark private/NSFW media so the phone blurs it per your setting. **On by default for relay installs** (installing the relay is itself the opt-in); reversible from the "Agent context" toggle in the Relay dashboard, or `RELAY_AGENT_CONTEXT_ENABLED=0`. The exact injected instruction is visible in the chat "What the agent sees" sheet under "Relay context (server-side)". No on-device or relay-side classifier — sensitivity stays model-emitted. Vanilla upstream (no plugin) is unaffected. See `docs/plans/2026-06-20-relay-enhancement-layer.md`.
|
||||
- **Transport path is visible (chat).** The chat status strip now shows which streaming path is actually in use — ⚡ Gateway (live thinking), 📡 Sessions, Completions, or Runs — instead of a generic "api online", and Chat Settings adds a basic→best tier ladder explaining the active path and its fallback.
|
||||
- **Injected-context audit (chat).** Tap the context-usage meter in chat to open a "What the agent sees" sheet showing the exact extra context prepended to your next turn — persona/profile, phone status, and any per-turn (voice) hint. On the gateway path it notes the persona is applied server-side, so the audit is honest about what the phone does and doesn't send.
|
||||
- **Spoken-turn badges (chat).** Voice-mode replies now carry a "Voice" chip and realtime replies a "Realtime Agent" chip — both with a speaker glyph — so spoken turns are distinguishable from typed ones in the scrollback.
|
||||
- **App themes.** A new theme picker in Settings → Appearance ships eight looks: the signature Hermes Relay brand (with full light/dark) plus ports of the Nous Hermes baselines — Hermes Teal, Nous Blue (light), Midnight, Ember, Mono, Cyberpunk, and Rosé. The whole app — brand chrome, accents, and chat background — follows the chosen theme. Light/Dark/Auto applies to themes that ship both modes; fixed-mode themes show their own complete look.
|
||||
- **Hot-swappable agent sphere.** The orb is now a pluggable "skin": an Adaptive skin that recolors to match your theme, built-in Classic / Aurora / Solar / Mono looks, and support for **user-authored skins** loaded from a small JSON spec. Each skin declares which live signals it reacts to (voice, tool bursts, activity), shown as capability badges in the picker. See `docs/sphere-spec.md`.
|
||||
- **Connections separate features from routes (Android).** Connection settings now distinguish what a connection can *do* (a **Features** section) from how this phone *reaches* Hermes (a **Route** section), so you can enable Relay features over whichever transport you prefer. A plugin-provided **Secure proxy** route is surfaced alongside LAN, Tailscale, public, and custom routes. The standard direct-to-upstream path is unchanged and still needs no plugin. See `docs/plans/2026-06-18-native-secure-routes.md`.
|
||||
- **Enhanced voice control (Gemini & xAI).** When the relay uses a Gemini or xAI voice provider, Voice Settings can now steer it: pick a Gemini voice and model and turn on expressive tone tags (with optional natural-language voice direction), or set an xAI voice with expressive speech tags. Expressive tags also apply to xAI on the streaming voice-output renderer. Standard (no-plugin) voice stays configured server-side.
|
||||
- **Voice render-path visibility.** Voice Settings shows which path is rendering speech (streaming vs. basic), and Diagnostics records it each session, making voice issues easier to troubleshoot.
|
||||
- **Agent pets — a living, swappable avatar.** The orb can be replaced with an animated "pet" that reacts to what the agent is doing: idle / thinking / writing / speaking / listening states, a distinct **working** pose during tool calls, one-shot **greet** / **celebrate** reactions, and a loop that quickens as output streams. Add or remove pets right in Settings → Appearance (no `adb` needed), with a live state preview, a playback-speed slider, and optional frame auto-stabilization; capability badges (Voice · Tools · Activity) show honestly what each pet actually reacts to. Pets are pure data — an AI authoring kit and a JSON schema let you generate one from sprite art. See `docs/pet-spec.md` and the custom-avatars guide.
|
||||
- **Per-profile agent icon + single-image avatars.** Each agent profile can wear its own small icon beside its name (client-side, never sent to Hermes), shown in chat, the agent sheet, the top bar, and Settings. Importing an avatar now also accepts a single image (auto-wrapped as a one-frame pet) — no animated pack required.
|
||||
- **In-app crash reporting.** If the app ever force-closes, the next launch shows a clean dialog with the stack trace — **Copy** it, or **Report** to open a pre-filled GitHub issue from the bug template. The report persists until you acknowledge it, and the handler re-raises so the OS still records the crash in Play vitals.
|
||||
- **Clean text-flow mode (chat).** A distraction-free chat layout where your sent text slides up into a continuous flow, paired with the swappable-avatar/pet system.
|
||||
- **Permissions review screen.** A central page makes the permission model explicit — standard Chat and Manage need no phone-control permissions, while voice, camera, notifications, and sideload Device Control stay opt-in — reading the same live grants Bridge does.
|
||||
- **In-app attachment previews + richer capture.** Attachments preview inline before sending, sensitive media is blurred per your setting, and the capture flow is richer.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Much faster cold start.** The app was building several hardware-keystore-encrypted stores at launch, which serialize on a process-global lock and stalled the chat header (model, personality, approvals) for seconds. It now builds a single keyset and the dashboard cookies share it, cutting measured time-to-connected from ~2.9 s to ~1 s after first frame, with the keystore lock contention gone. Existing sign-ins are migrated automatically on first launch.
|
||||
- **Honest loading, never stale, never hidden.** Model, personality, and approvals now show a brief "checking…" state and fade in once the server confirms them, instead of popping in or showing a possibly-wrong value. Standard upstream controls (Model, YOLO, Fast, reasoning effort) are no longer hidden while loading or when unavailable — they always appear: a live control when ready, "checking…" while a value loads, or a cleanly disabled control with the reason (e.g. "available over the gateway transport") when this connection can't use them. The chat composer's reasoning-effort chip now shows alongside the model chip instead of lagging seconds behind the gateway check, and picker lists (models, personalities) show a brief, bounded "loading…" cue. The same fade-in is applied to the context meter, session drawer, and Manage panels.
|
||||
- **Tidier chat header.** The LAN/Tailscale chip was dropped from the top bar (the bottom status strip already shows the route, and is now tappable to open Connections), and a `none` personality is no longer shown — leaving more room for the model name.
|
||||
- **Connection toast reads like the cold-start screen.** The floating connection status toast now shows a live checklist — Route / API / Relay each with a spinner, ✓, or ✕ as the checks land — instead of flat text, matching the splash screen's stepper. Swiping it up now tracks your finger (slide + fade) rather than snapping, and connection problems get an explicit "Open Connections →" link at the bottom so the path to the detailed view is obvious.
|
||||
- **Tidier chat header.** The "approvals off" warning moved out of the agent subtitle into a single amber ⚡ icon in the top bar (tap for the full explanation in the agent sheet), and Share folded into a ⋮ overflow menu — so the personality · model subtitle no longer gets clipped by the trailing action icons.
|
||||
- **Voice replies are formatted for listening.** In voice mode the assistant is now guided to answer in short, conversational sentences without markdown, emoji, or raw URLs — without changing what is stored in chat history.
|
||||
- **Leaner terminal screen (Android).** The extra-keys bar scrolls horizontally with compact, fully-legible keys (no more clipped "CTRL"), the header is a single compact row showing one inline connection-status dot plus state, and the tab strip is hidden for single-tab sessions — the new-tab "+" moves into the header — reclaiming vertical space for the terminal.
|
||||
- **Relay terminals run on an isolated, TUI-tuned tmux.** Sessions now use a dedicated tmux server/socket with its own config — instant ESC (`escape-time 0`), truecolor `$TERM`, mouse and focus events on, and no status bar — so editors and full-screen tools behave correctly, without touching the user's personal tmux.
|
||||
- **"Standard" is now "Vanilla Hermes" throughout.** The user-facing name for the no-plugin upstream path is now **Vanilla Hermes**, so it's clear the default path runs on a plain Hermes agent.
|
||||
- **QR pairing degrades gracefully on unusual cameras.** On foldables and devices where the camera can't initialize, the scanner now shows a "camera unavailable — pair manually" card instead of force-closing.
|
||||
- **Image & attachment viewers rotate to landscape.** The full-screen image / attachment viewers can rotate to landscape even though the rest of the app stays portrait-locked.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Clearer error when a feature needs a newer relay.** Toggling a setting an older relay plugin doesn't recognize (e.g. xAI expressive speech tags) now shows "Relay update needed" instead of a generic HTTP 400 with a dead Retry button. Genuine input errors are unaffected.
|
||||
- **Connection status toast is no longer see-through.** The floating connection-lost/switching toast renders fully opaque so content behind it no longer bleeds through and hurts legibility.
|
||||
- **Provenance badges survive the post-turn history reload.** "Voice", "Realtime Agent", "Stopped", and "Error" chips are now preserved when the conversation reloads after a turn, instead of silently vanishing.
|
||||
- **Chat and Manage no longer stay dark in Light mode.** Brand-styled surfaces bypassed the theme and were effectively hardcoded dark; they now follow the selected theme and light/dark mode, and the glow/border flourishes key off the active theme rather than the system setting.
|
||||
- **Realtime voice no longer drops the conversation mid-session with some providers.** A normal end-of-turn signal was being rejected on certain voice providers, ending the session every turn.
|
||||
- **Relay voice synthesis no longer leaves temporary audio files behind** on the server.
|
||||
- **Clearer voice errors and an oversize-recording guard.** Standard voice now rejects an over-long recording before uploading it and shows a helpful message for audio the server can't read, instead of a generic HTTP error.
|
||||
- **Terminal paste no longer auto-runs multi-line text.** The key-bar PASTE now uses bracketed paste, so multi-line content lands intact in shells and editors instead of executing line by line.
|
||||
- **Terminal on-screen arrows behave inside TUIs.** Arrow/Home/End keys follow the running app's cursor-key mode (application vs. normal), so they work correctly in vim, less, and fzf.
|
||||
- **Terminal footer spacing.** A small gap now keeps the last terminal row clear of the key bar (it could previously look like the footer overlapped it), and a redundant navigation-bar inset that left empty space below the keys was removed.
|
||||
- **In-chat model picker now actually applies on a new chat.** Picking a model and provider in the chat composer (e.g. Grok 4.3 via your xAI subscription) is bound to the new conversation, so the agent runs on the picked model instead of silently falling back to the account's global default. Switching profiles retires an explicit pick so the profile's own model takes over, and the picker label updates immediately instead of lagging a round-trip.
|
||||
- **Server-generated images render in chat when paired to the relay.** An assistant image that points at a server-side file path is now fetched through the relay's media route and shown inline (tap to zoom), instead of degrading to an "image is on the server" notice. On the SSE chat path the agent is also told it can surface images and files by path when a relay route is configured (visible in the chat "What the agent sees" sheet). Standard (no-plugin) connections are unchanged.
|
||||
- **Smoother profile switching.** Switching profiles no longer blanks the conversation to an empty/"Loading…" state before the new history loads; the previous transcript is held and cross-fades to the new one.
|
||||
- **In-chat model switch now applies mid-conversation, not just on new chats.** Picking a model in an already-started chat switches the live session in place — the same path the desktop/TUI `/model` uses — instead of racing into a global-default write, so the turn runs the model you picked.
|
||||
- **Server-side turn errors always surface.** A failed turn (e.g. a provider rejecting the request) now stays on screen as an error bubble with the message, instead of appearing for a moment and then vanishing when the conversation reconciled after the turn.
|
||||
- **The model shown in chat matches the live session.** The chat header and the agent detail sheet now show the model the current session is actually running (reflecting a mid-session switch) rather than the profile/global default, and the agent sheet no longer pairs the global default model name with the session's provider — it now also names the host's "Server default" when the session runs something different.
|
||||
- **Server steering markers no longer appear as chat bubbles.** The "[System: the active model/personality changed]" notes the server injects into history for the agent's benefit are hidden from the transcript by default (matching the desktop/TUI); a new "Show system messages" debug toggle in Chat Settings can reveal them.
|
||||
- **Per-reply token counts (and other per-message details) survive the post-turn reload.** The input/output token subtext, provenance badges, tapped-card state, and voice/realtime sync traces are now preserved when the conversation reconciles against the server after a turn — previously a normal reply lost its token line once the turn finished (the error bubble kept it only because errored turns skip that reload). The reloader now preserves client-only message details by default instead of dropping any it doesn't re-derive from the server.
|
||||
- **PDF viewer no longer crashes when the document closes mid-render.** A PDF preview that was torn down during a layout pass could read a closed renderer and throw `IllegalStateException: Document already closed`; the renderer is now guarded so it returns nothing instead of crashing.
|
||||
- **No crash opening a chat with a server-local image.** Rendering a relay-fetched image could throw `ClassCastException: kotlin.Result cannot be cast to byte[]` because a `suspend` function returned `kotlin.Result` (which collides with the coroutine machinery's own wrapper); a purpose-built result type fixes it.
|
||||
- **Side-loaded avatars and sphere skins are reachable again.** Both loaders read internal storage while the docs (correctly) pointed `adb push` at external app-scoped storage, so a side-loaded pet or skin never appeared. Both now resolve through one external-preferred location, so the documented install path works.
|
||||
- **Reopened chats paint the session's real model** (not the profile/global default), the model-picker "Server default" caption shows the true default rather than the active override, and a chat's media badge shows only when paired — with the underlying server-image fetch-failure reason surfaced when a fetch fails.
|
||||
|
||||
## [1.1.0] - 2026-06-16
|
||||
|
||||
### Added
|
||||
|
||||
- **Automated Play Console upload on release.** When a `PLAY_SERVICE_ACCOUNT_JSON` secret is configured, pushing a stable `android-v*` tag uploads the `googlePlay` App Bundle to the Production track as a draft (a human still starts the rollout). Prereleases are skipped, and the `sideload` flavor is structurally blocked from ever publishing to Play. Without the secret, the release builds publish to GitHub Releases exactly as before.
|
||||
- **Desktop UI preview harness (`:ui-preview`).** A non-shipped Compose for Desktop module renders presentational composables in a window on the PC with Compose Hot Reload, for fast UI iteration without a device build/install loop. It reuses the shared sphere algorithm as its single source of truth.
|
||||
- **Plugin: guided env-key setup.** The relay plugin declares its optional voice-provider keys (`XAI_API_KEY`, `OPENAI_API_KEY`, `ELEVENLABS_API_KEY`) in its manifest, so `hermes plugins install` prompts for them (masked, with a "get yours" link) instead of hand-editing `.env`. The standard no-plugin path needs none.
|
||||
- **Plugin: native install path.** Tools-only setups can install via `hermes plugins install Codename-11/hermes-relay/plugin`; the full relay still uses the curl `install.sh`.
|
||||
- **`/relay` slash commands.** `relay status · devices · pair` usable mid-conversation from any platform (CLI / Discord / TUI).
|
||||
- **Dashboard relay-status widget.** A `Relay · connected / offline / unpaired` badge in the dashboard header, visible on every page.
|
||||
- **Session-start relay health check.** A minimal, fully-guarded `on_session_start` hook records relay reachability without slowing the gateway.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Release names normalized by surface.** Future GitHub Releases are named `Hermes-Relay-Android`, `Hermes-Relay-Plugin`, and `Hermes-Relay-CLI`, with future tags on `android-v*`, `plugin-v*`, and `cli-v*`. The CLI installer and updater still understand historical `desktop-v*` prereleases during the migration.
|
||||
- **Per-surface release notes.** Plugin and CLI GitHub Releases now use hand-written `PLUGIN_RELEASE_NOTES.md` / `CLI_RELEASE_NOTES.md` files (Summary + Added/Changed/Fixed + Install/Verify) — the same format as Android's `RELEASE_NOTES.md` — instead of static boilerplate baked into the workflow. The release workflows substitute the version into the install commands automatically.
|
||||
- **Settings screen overhaul (Android).** Status pills are now exception-only — they appear only when a surface needs attention and stay quiet when healthy. The Power tools section shows a single state-aware **Plugin active / required / offline** badge instead of an identical "Relay paired" chip on every card. Connections moved to the top (above the Hermes section), Diagnostics + Developer options moved into the App section, the status chips were restyled to match the app's translucent-bordered language, and the brand blue was deepened.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Force-close on connect when the stored credential keyset was corrupt.** A corrupt encrypted token store (which can happen after an app upgrade or device restore) threw during construction and crashed the app right after a successful pair, on both standard and relay connections. The token store now heals a corrupt keyset on the spot, and credential storage degrades to a re-pair instead of crashing if the device keystore is unusable.
|
||||
- **Dashboard plugin: unreadable button labels.** Solid buttons in the relay dashboard panel inherited the container text colour, which matched their background. Solid button variants now keep their proper contrast colour.
|
||||
- **Installer failed on uv-managed Hermes hosts.** `install.sh` assumed `pip` lived in the hermes-agent virtualenv, but environments created by `uv` (the upstream default) ship no `pip` module, so the editable install aborted at step 2. The installer now bootstraps `pip` via `ensurepip`, or falls back to `uv pip`, so the plugin installs cleanly on uv-managed cores.
|
||||
- **Chat settings (Android).** The streaming-endpoint picker no longer wraps "Gateway"/"Sessions" onto a second line, and the system-prompt preview now reflects the enabled context toggles (foreground app, battery, safety rails) with representative placeholder values instead of looking inert.
|
||||
- **Dashboard plugin: buttons rendered as blank boxes.** The host dashboard's Nous design-system `Button`/`Badge` use boolean variant flags (`outlined`/`ghost`/`invert`) and a `tone` prop — not the shadcn-style `variant` prop the plugin passed — so every button collapsed to a solid near-white fill with an invisible label. The plugin now translates its props to the design-system contract via an adapter, and drops a label-hiding CSS reset.
|
||||
|
||||
## [1.0.0] - 2026-06-14
|
||||
|
||||
### Added
|
||||
|
||||
- **Relay plugin diagnostics and install guidance.** `hermes relay doctor` now reports standard upstream API/dashboard reachability, Relay loopback state, dashboard plugin presence, plugin-manager layout, and whether the legacy bootstrap monkeypatch is installed. The plugin manifest now advertises its Android and desktop tools, and `after-install.md` gives the upstream plugin manager a first-run handoff.
|
||||
|
||||
- **Plugin-owned compatibility hook lifecycle.** `hermes relay compat status/install/remove` now owns the optional `hermes_relay_bootstrap.pth` startup hook, so the monkeypatch can be inspected, added, or removed without rerunning the legacy installer. The standard v1.0.0 path does not require this hook.
|
||||
|
||||
- **Legacy cleanup alignment.** The legacy installer now installs the optional `.pth` hook through the plugin compat lifecycle, and the uninstaller removes every shell shim it creates (`hermes-pair`, `hermes-status`, `hermes-relay`, `hermes-relay-update`, `hermes-relay-tailscale`) while delegating hook cleanup to `hermes relay compat remove` when available.
|
||||
|
||||
- **Gateway chat transport with live thinking.** Chat can ride the upstream dashboard `/api/ws` (the `tui_gateway` surface the official hermes-desktop client speaks) — the only vanilla-upstream path that streams reasoning *live*, so the Thinking block and sphere light up during generation. "Auto" prefers it when the dashboard is reachable and Manage is signed in, and falls back to the SSE endpoints per turn.
|
||||
|
||||
- **Gateway desktop parity.** Native image/PDF/file attachments (with an in-chat notice when a turn falls back to a transport that can't carry files), mid-turn **steering**, **edit & resend**, interactive **approval / clarify / sudo / secret** cards, live **subagent lanes**, a **context-window meter**, server **slash commands** in autocomplete, and **turn-complete notifications** when the app is backgrounded.
|
||||
@@ -30,6 +136,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/), and this
|
||||
|
||||
### Changed
|
||||
|
||||
- **Relay plugin/server version aligned to v1.0.0.** The Python package, plugin manifest, dashboard manifest, and relay runtime now use the same `1.0.0` line as the stable Android release so a retagged source checkout describes one product version.
|
||||
|
||||
- **The standard (no-plugin) path is first-class.** Chat, Manage, and voice all work against an unmodified upstream Hermes agent; standard voice rides the dashboard audio surface (`/api/audio/*`) with the Manage sign-in, and relay-paired voice is the profile-aware fallback. The relay plugin is now purely additive.
|
||||
|
||||
- **Seamless connection UX.** LAN↔Tailscale handoffs and reconnects no longer reload the chat; connection and update status are now in-theme slide-down toasts over the content instead of banners that pushed the UI around.
|
||||
@@ -82,7 +190,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/), and this
|
||||
|
||||
- **Google Play Bridge Core split.** The Google Play Android track keeps relay pairing, chat, profiles, voice, terminal/TUI, media, notification companion, relay sessions, diagnostics, and status while removing AccessibilityService-backed Device Control declarations and permissions. Sideload remains the track for screen reading, gestures, screenshots, SMS/calls, contacts/location, overlays, wake locks, and unattended control.
|
||||
|
||||
- **Release lanes now use explicit product tags and names.** Future Android releases use `android-v*`, server/Python releases use `server-v*`, and desktop continues on `desktop-v*`. GitHub Release names now publish as `Hermes-Relay-Android vX.Y.Z`, `Hermes-Relay-Server vX.Y.Z`, and `Hermes-Relay-Desktop vX.Y.Z`; the old relay-named server scripts remain compatibility shims.
|
||||
- **Release lanes now use explicit product tags and names.** Future Android releases use `android-v*`, plugin/Python releases use `server-v*`, and CLI releases continue on `desktop-v*`. GitHub Release names now publish as `Hermes-Relay-Android vX.Y.Z`, `Hermes-Relay-Plugin vX.Y.Z`, and `Hermes-Relay-CLI vX.Y.Z`; the old relay-named server scripts remain compatibility shims.
|
||||
|
||||
- **Realtime voice instructions are provider-neutral.** Realtime providers receive active interface context, local date/time, provider/model/voice/profile metadata, and guidance to ask Hermes for current facts, research, device/desktop state, project context, precise/versioned data, and any requested checks instead of guessing from model knowledge.
|
||||
|
||||
@@ -208,9 +316,9 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/), and this
|
||||
|
||||
### Fixed
|
||||
|
||||
- **desktop CLI binary was a no-op on alpha.3** — installed cleanly, exited 0, produced zero stdout/stderr, wasn't "recognized" as a CLI. Root cause: cli.ts guarded its entry-point invocation with `fileURLToPath(import.meta.url) === process.argv[1]`, which is a valid Node idiom but fails in Bun-compiled binaries because the entry module has a synthetic URL that doesn't match the `.exe` path — the check evaluated false, `main()` was never called, binary exited 0 silently. Replaced with `import.meta.main` (cross-runtime: Bun, Node 20.11+, tsx) which is true in the entry module regardless of compile mode. All four invocation paths stay correct (Bun --compile binary, `bin/hermes-relay.js` shim, `tsx src/cli.ts`, test imports). Caught by adding a local `npm run smoke` target that runs the compiled Windows binary against `--version` / `--help` / `doctor` and verifies each produces output. Same smoke added to `release-desktop.yml` on the Linux target so future regressions of this class are caught pre-publish. Affects `desktop-v0.3.0-alpha.3`; fix ships as `desktop-v0.3.0-alpha.4`.
|
||||
- **desktop CLI binary was a no-op on alpha.3** — installed cleanly, exited 0, produced zero stdout/stderr, wasn't "recognized" as a CLI. Root cause: cli.ts guarded its entry-point invocation with `fileURLToPath(import.meta.url) === process.argv[1]`, which is a valid Node idiom but fails in Bun-compiled binaries because the entry module has a synthetic URL that doesn't match the `.exe` path — the check evaluated false, `main()` was never called, binary exited 0 silently. Replaced with `import.meta.main` (cross-runtime: Bun, Node 20.11+, tsx) which is true in the entry module regardless of compile mode. All four invocation paths stay correct (Bun --compile binary, `bin/hermes-relay.js` shim, `tsx src/cli.ts`, test imports). Caught by adding a local `npm run smoke` target that runs the compiled Windows binary against `--version` / `--help` / `doctor` and verifies each produces output. Same smoke runs in `release-cli.yml` on the Linux target so future regressions of this class are caught pre-publish. Affects `desktop-v0.3.0-alpha.3`; fix ships as `desktop-v0.3.0-alpha.4`.
|
||||
- **`hermes-relay --version` printed `0.0.0` in compiled binaries.** `readVersion()` tried to read `package.json` via `__dirname + '../package.json'`, which doesn't resolve in a Bun `--compile` binary (no real filesystem layout). Replaced with a build-time-generated `src/version.ts` module (`npm run gen:version` writes the version from package.json before every build and every `build:bin:*`). `readVersion()` now just returns the embedded constant. Works identically in tsx / Node / Bun.
|
||||
- **desktop CLI binary segfaulted at startup on Bun 1.3.13 Windows x64** (`panic(main thread): Segmentation fault at address 0x100000D9C`). Root cause identified as Bun's experimental `--bytecode` flag; attempted fix in alpha.2 only edited `desktop/package.json`'s build scripts while the release workflow's inline `bun build` commands silently kept `--bytecode`, so alpha.2 shipped with the same crash. alpha.3 fixes the workflow two ways: (1) dropped `--bytecode` from release-desktop.yml, and (2) refactored the four build steps to delegate to `npm run build:bin:*` so the package.json scripts are the single source of truth for compile flags. Added a `bun --version` diagnostic step to the workflow for future triage. Versions affected: `desktop-v0.3.0-alpha.1` and `desktop-v0.3.0-alpha.2`. Fix ships as `desktop-v0.3.0-alpha.3`.
|
||||
- **desktop CLI binary segfaulted at startup on Bun 1.3.13 Windows x64** (`panic(main thread): Segmentation fault at address 0x100000D9C`). Root cause identified as Bun's experimental `--bytecode` flag; attempted fix in alpha.2 only edited `desktop/package.json`'s build scripts while the release workflow's inline `bun build` commands silently kept `--bytecode`, so alpha.2 shipped with the same crash. alpha.3 fixes the workflow two ways: (1) dropped `--bytecode` from the CLI release workflow, and (2) refactored the four build steps to delegate to `npm run build:bin:*` so the package.json scripts are the single source of truth for compile flags. Added a `bun --version` diagnostic step to the workflow for future triage. Versions affected: `desktop-v0.3.0-alpha.1` and `desktop-v0.3.0-alpha.2`. Fix ships as `desktop-v0.3.0-alpha.3`.
|
||||
- **Installer couldn't find alpha-only releases.** GitHub's `/releases/latest/download/` URL deliberately skips prereleases, so the default `curl | sh` / `irm | iex` one-liner failed against alpha.1 with "maybe no Windows release for this version yet?" Both `install.sh` and `install.ps1` now query the Releases API directly (`GET /repos/.../releases`, filter to `desktop-v*` tags, take first) when `HERMES_RELAY_VERSION=latest`. Pinned versions unchanged.
|
||||
|
||||
### Added
|
||||
@@ -1225,7 +1333,9 @@ MVP release — native Android companion app for Hermes agent with direct API ch
|
||||
- **Dev scripts** — build, install, run, test, relay via scripts/dev.bat
|
||||
- **ProGuard rules** — okhttp-sse, markdown renderer, intellij-markdown parser
|
||||
|
||||
[Unreleased]: https://github.com/Codename-11/hermes-relay/compare/android-v0.8.0...HEAD
|
||||
[Unreleased]: https://github.com/Codename-11/hermes-relay/compare/android-v1.0.0...HEAD
|
||||
[1.0.0]: https://github.com/Codename-11/hermes-relay/compare/android-v0.8.0...android-v1.0.0
|
||||
[0.8.1]: https://github.com/Codename-11/hermes-relay/compare/android-v0.8.0...android-v0.8.1
|
||||
[0.8.0]: https://github.com/Codename-11/hermes-relay/compare/v0.7.0...android-v0.8.0
|
||||
[0.7.0]: https://github.com/Codename-11/hermes-relay/compare/v0.6.1...v0.7.0
|
||||
[0.1.0]: https://github.com/Codename-11/hermes-relay/compare/v0.1.0-beta...v0.1.0
|
||||
|
||||
@@ -4,24 +4,26 @@
|
||||
|
||||
## What This Is
|
||||
|
||||
A native Android app (Kotlin + Jetpack Compose) paired with a Python relay server (aiohttp) for the Hermes agent platform. Chat connects directly to the Hermes API Server via HTTP/SSE; bridge and terminal use a relay over WSS.
|
||||
A native Android app (Kotlin + Jetpack Compose) paired with an optional Python relay plugin/server (aiohttp) for the Hermes agent platform. Vanilla Hermes chat, Manage, and dashboard voice work against unmodified upstream Hermes. Relay adds phone control, terminal, remote desktop tooling, extra voice engines, and dashboard Relay management.
|
||||
|
||||
**Current state:** v0.8.0 (release-prep on `dev`) — Phase 0–3 complete. Direct API chat, session management, pairing + security (now multi-endpoint, ADR 24), inbound media, voice mode (stable Hermes Chat + Voice Output plus opt-in provider-native Realtime Agent with reliable low-latency playback and a text/mic Voice Lab), bridge/accessibility control, notification companion, safety rails, multi-Connection, agent profiles + inspector, connection diagnostics, and first-class Tailscale (ADR 25). Two product flavors: `googlePlay` (conservative, Bridge Core without Device Control) and `sideload` (full-capability).
|
||||
**Current state:** v1.0.0 stable. The default no-plugin path supports chat, Manage, and voice on vanilla upstream Hermes. Chat auto-prefers the dashboard `/api/ws` gateway transport when Manage auth is ready, then falls back to API-server SSE routes. Vanilla Hermes voice uses dashboard `/api/audio/*` with the Manage session. Relay remains an additive power path for terminal, bridge/device control, notification companion, extra/provider-native voice, remote access, and desktop tooling. Two Android product flavors ship: `googlePlay` (conservative, no unattended Device Control surface) and `sideload` (full-capability).
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Phone (HTTP/SSE) → Hermes API Server (:8642) [chat — direct]
|
||||
Phone (WSS) → Relay Server (:8767) [bridge, terminal]
|
||||
Phone (WS) -> Hermes dashboard (:9119) [vanilla Hermes gateway chat, live thinking]
|
||||
Phone (HTTP/SSE) -> Hermes API Server (:8642) [vanilla Hermes chat fallback, sessions, runs]
|
||||
Phone (HTTP) -> Hermes dashboard (:9119) [vanilla Hermes Manage + voice]
|
||||
Phone (WSS/HTTP) -> Relay plugin/server (:8767) [optional bridge, terminal, relay voice, remote tools]
|
||||
```
|
||||
|
||||
Chat goes directly to the API server via HTTP/SSE. The API key (Bearer token) is optional — most local setups run without one. Terminal will go through tmux via the relay. Bridge wraps existing relay protocol. See docs/decisions.md for why.
|
||||
The Vanilla Hermes path must stay upstream-only. API-server bearer auth and dashboard cookie auth are separate. Terminal and bridge require Relay pairing; Vanilla Hermes chat, Manage, and dashboard voice must not.
|
||||
|
||||
### Upstream Hermes API Reference
|
||||
|
||||
**IMPORTANT:** Always verify endpoints against the actual hermes-agent source (`gateway/platforms/api_server.py`). The upstream repo is the source of truth — not our docs, not our memory, not assumptions from other frontends.
|
||||
|
||||
**Standard endpoints (confirmed in hermes-agent source):**
|
||||
**Vanilla Hermes endpoints (confirmed in hermes-agent source):**
|
||||
|
||||
| Endpoint | Purpose | Tool Call Format |
|
||||
|----------|---------|-----------------|
|
||||
@@ -42,7 +44,7 @@ Chat goes directly to the API server via HTTP/SSE. The API key (Bearer token) is
|
||||
Upstream main now contains the focused session-control API (`#33134`) and read-only skills/toolsets (`#33016`). The original broad PR [#8556](https://github.com/NousResearch/hermes-agent/pull/8556) was closed as superseded. Keep these distinctions straight:
|
||||
|
||||
1. **Native upstream** — `/api/sessions`, `/api/sessions/{id}/messages`, `/api/sessions/{id}/chat`, `/api/sessions/{id}/chat/stream`, `/v1/capabilities`, `/v1/skills`, and `/v1/toolsets` exist in current `gateway/platforms/api_server.py`.
|
||||
2. **Bootstrap compatibility** (`hermes_relay_bootstrap/`) — monkey-patches aiohttp on startup via `.pth` file for older or partial core builds. It skips native routes per method/path and should be retired per surface, not treated as the preferred path.
|
||||
2. **Bootstrap compatibility** (`plugin/hermes_relay_bootstrap/`) — monkey-patches aiohttp on startup via `.pth` file for older or partial core builds. It skips native routes per method/path and should be retired per surface, not treated as the preferred path. The repo-root `hermes_relay_bootstrap/` package is a legacy import shim.
|
||||
3. **Legacy fork branches** — useful as lineage only. Do not cite `feat/session-api` / `#8556` as the current upstream contract.
|
||||
|
||||
| Endpoint | Purpose | Provided by |
|
||||
@@ -63,9 +65,9 @@ The Android client probes per-endpoint capability via `HermesApiClient.probeCapa
|
||||
|
||||
**Dashboard web server (separate surface — standard Manage / Desktop remote gateway):**
|
||||
|
||||
hermes-agent ships a second web server at `hermes_cli/web_server.py` that hosts the React admin dashboard at `hermes_cli/web_dist/`. It has its **own** `/api/*` routes that **do not live on `api_server.py`** — notably: `GET/PUT /api/config` (full tree), `GET /api/config/schema`, `GET /api/config/defaults`, `GET/PUT /api/config/raw` (YAML text), `GET/PUT/DELETE /api/env` + `POST /api/env/reveal`, `PUT /api/skills/toggle`, `/api/cron/jobs/*` (different shape from `/api/jobs/*`), `/api/providers/oauth/*`, `/api/dashboard/themes`, `/api/dashboard/plugins`, `/api/model/info` + `/api/model/options` + `POST /api/model/set`, `/api/profiles/*` (CRUD, `POST /api/profiles/active`, per-profile soul/description/model), `/api/mcp/*`, `/api/logs`, `/api/analytics/usage`, and **`POST /api/audio/transcribe` + `POST /api/audio/speak`** (base64 data-url contract, built for hermes-desktop voice). The API server has **no audio routes** — its `/v1/capabilities` advertises `audio_api: false`; PR #8199 (`/v1/audio/*`) is the canonical future surface but is unmerged. Android's **standard (no-plugin) voice** therefore rides this dashboard surface via `StandardHermesVoiceClient` with the per-connection dashboard cookie session (Manage sign-in unlocks voice); `AutoVoiceAudioClient` prefers Relay when paired and falls back to standard.
|
||||
hermes-agent ships a second web server at `hermes_cli/web_server.py` that hosts the React admin dashboard at `hermes_cli/web_dist/`. It has its **own** `/api/*` routes that **do not live on `api_server.py`** — notably: `GET/PUT /api/config` (full tree), `GET /api/config/schema`, `GET /api/config/defaults`, `GET/PUT /api/config/raw` (YAML text), `GET/PUT/DELETE /api/env` + `POST /api/env/reveal`, `PUT /api/skills/toggle`, `/api/cron/jobs/*` (different shape from `/api/jobs/*`), `/api/providers/oauth/*`, `/api/dashboard/themes`, `/api/dashboard/plugins`, `/api/model/info` + `/api/model/options` + `POST /api/model/set`, `/api/profiles/*` (CRUD, `POST /api/profiles/active`, per-profile soul/description/model), `/api/mcp/*`, `/api/logs`, `/api/analytics/usage`, and **`POST /api/audio/transcribe` + `POST /api/audio/speak`** (base64 data-url contract, built for hermes-desktop voice). The API server has **no audio routes** — its `/v1/capabilities` advertises `audio_api: false`; PR #8199 (`/v1/audio/*`) is the canonical future surface but is unmerged. Android's **Vanilla Hermes (no-plugin) voice** therefore rides this dashboard surface via `StandardHermesVoiceClient` with the per-connection dashboard cookie session (Manage sign-in unlocks voice); `AutoVoiceAudioClient` prefers Relay when paired and falls back to standard.
|
||||
|
||||
Current upstream supports two auth modes on this surface. Loopback dashboards still use the injected `window.__HERMES_SESSION_TOKEN__` path. Remote/non-loopback dashboards use the Desktop-style dashboard auth gate: `/api/status` advertises `auth_required` and providers, `/auth/password-login` handles password providers, `/auth/login?provider=...` handles Nous/OIDC redirects, `/api/auth/me` returns the verified session, and `/api/auth/ws-ticket` mints a short-lived ticket for `/api/ws` / `/api/pty`. This dashboard session is **not** an `API_SERVER_KEY`; Android Chat still uses the API-server bearer path until a dashboard `/api/ws` chat adapter is wired. Note the event-richness gap: `/api/ws` is backed by `tui_gateway/server.py` (what hermes-desktop + the Ink TUI speak) and is the only upstream surface with **live** `reasoning.delta`/`thinking.delta` streaming; the api_server SSE paths emit reasoning only post-hoc (`reasoning.available` → `tool.progress` with `tool_name:"_thinking"`, ≤500 chars; full text in `run.completed.messages[].reasoning`). Android Manage may consume this dashboard surface directly, but relay-only capabilities remain behind Relay pairing. **Do not proxy dashboard auth or dashboard admin APIs over the relay.**
|
||||
Current upstream supports two auth modes on this surface. Loopback dashboards still use the injected `window.__HERMES_SESSION_TOKEN__` path. Remote/non-loopback dashboards use the Desktop-style dashboard auth gate: `/api/status` advertises `auth_required` and providers, `/auth/password-login` handles password providers, `/auth/login?provider=...` handles Nous/OIDC redirects, `/api/auth/me` returns the verified session, and `/api/auth/ws-ticket` mints a short-lived ticket for `/api/ws` / `/api/pty`. This dashboard session is **not** an `API_SERVER_KEY`. Android uses it for Manage, Vanilla Hermes voice, and the gateway chat transport. `/api/ws` is backed by `tui_gateway/server.py` (what hermes-desktop + the Ink TUI speak) and is the only upstream surface with **live** `reasoning.delta`/`thinking.delta` streaming; the api_server SSE paths remain the SSE fallback. Relay-only capabilities remain behind Relay pairing. **Do not proxy dashboard auth or dashboard admin APIs over the relay.**
|
||||
|
||||
**Tool call rendering paths:**
|
||||
1. **Runs API** — Emits `tool.started`/`tool.completed` as real SSE events → `ToolProgressCard` in real-time.
|
||||
@@ -73,10 +75,10 @@ Current upstream supports two auth modes on this surface. Loopback dashboards st
|
||||
3. **Annotation parser** — Fallback for servers emitting inline markdown annotations (`` `💻 terminal` ``).
|
||||
|
||||
## Key Instructions
|
||||
- **Standard path = vanilla upstream only.** The default (no-plugin) connection path — chat via the API server, standard voice via the dashboard surface — must work against **unmodified upstream hermes-agent**: no fork patches, no bespoke server config as a dependency. The app ships on Google Play to users whose servers we don't control. Features that need server-side changes go through upstream PRs (with graceful degradation until merged) or live behind the opt-in relay plugin.
|
||||
- **Vanilla Hermes path = upstream-only.** The default (no-plugin) connection path — gateway/API chat, Manage, and Vanilla Hermes voice via the dashboard surface — must work against **unmodified upstream hermes-agent**: no fork patches, no bespoke server config as a dependency. The app ships on Google Play to users whose servers we don't control. Features that need server-side changes go through upstream PRs (with graceful degradation until merged) or live behind the opt-in relay plugin.
|
||||
- **Always verify upstream before assuming an endpoint exists.** Check `gateway/platforms/api_server.py` in hermes-agent. If an endpoint isn't there, document whether bootstrap injects it or it requires the fork.
|
||||
- If we use a non-standard endpoint, ensure `probeCapabilities()` covers it and the auto-resolver degrades gracefully.
|
||||
- **Bootstrap maintenance:** Retire `hermes_relay_bootstrap/` per surface. Sessions and read-only skills/toolsets now have native upstream replacements; config, memory, legacy skill detail/toggle, available-models, and slash middleware still need explicit replacement decisions before full removal.
|
||||
- **Bootstrap maintenance:** Retire `plugin/hermes_relay_bootstrap/` per surface. Sessions and read-only skills/toolsets now have native upstream replacements; config, memory, legacy skill detail/toggle, available-models, and slash middleware still need explicit replacement decisions before full removal.
|
||||
|
||||
## Repository Layout
|
||||
|
||||
@@ -93,6 +95,10 @@ hermes-android/
|
||||
│ ├── accessibility/ # HermesAccessibilityService, ScreenReader, ActionExecutor
|
||||
│ ├── bridge/ # BridgeSafetyManager, BridgeForegroundService, BridgeStatusOverlay
|
||||
│ └── notifications/ # HermesNotificationCompanion
|
||||
├── relay-core/ ← [EXPERIMENTAL] Quest/XR shared core lib (com.axiomlabs.hermesrelay.core) — pairing, transport, terminal, voice, wire
|
||||
├── relay-ui/ ← [EXPERIMENTAL] Quest/XR shared Compose UI lib — sphere, terminal WebView, QR scanner
|
||||
├── quest/ ← [EXPERIMENTAL] Meta Spatial SDK Quest/XR app (gradle includeBuild; in development, not shipped)
|
||||
├── ui-preview/ ← Desktop Compose Hot Reload harness for PC UI iteration (NOT shipped; shares MorphingSphereCore)
|
||||
├── desktop/ ← Node thin-client CLI (`@hermes-relay/cli`)
|
||||
│ ├── bin/hermes-relay.js # #!/usr/bin/env node shim → dist/cli.js
|
||||
│ ├── src/
|
||||
@@ -116,7 +122,7 @@ hermes-android/
|
||||
│ ├── tools/ # android_navigate.py, android_notifications.py
|
||||
│ └── dashboard/ # hermes-agent dashboard plugin — manifest, React UI, FastAPI proxy
|
||||
├── relay_server/ ← Thin compat shim → plugin.relay (legacy entrypoint)
|
||||
├── hermes_relay_bootstrap/ ← Runtime compatibility patch; retire per surface as upstream replaces it
|
||||
├── hermes_relay_bootstrap/ ← Legacy import shim for older startup hooks
|
||||
├── skills/devops/hermes-relay-pair/ ← /hermes-relay-pair slash command
|
||||
├── scripts/ ← dev.bat, bridge-smoke.sh, bump-version.sh
|
||||
└── docs/ ← spec, decisions, security, relay-server, mcp-tooling
|
||||
@@ -125,9 +131,10 @@ hermes-android/
|
||||
## Project Conventions
|
||||
|
||||
### File Structure
|
||||
- **Root-level:** README.md, CLAUDE.md, AGENTS.md, DEVLOG.md, .gitignore
|
||||
- **Root-level:** README.md, CLAUDE.md, AGENTS.md, DEVLOG.md, TODO.md, .gitignore
|
||||
- **docs/** — spec, decisions, security, and any other long-form documentation
|
||||
- **DEVLOG.md** — update at end of each work session with what was done, what's next, blockers
|
||||
- **DEVLOG.md** — update at end of each work session with what was done + verification (the factual record of *what happened*). It churns; do NOT park forward work here.
|
||||
- **TODO.md** — the single home for follow-ups / deferred work / known gaps ("what's next"). Record them here — never buried in DEVLOG or scattered through code/doc comments where they get lost.
|
||||
- **CLAUDE.md hygiene:** Key Files entries must stay one line — implementation detail belongs in the file or `docs/`. Run `/revise-claude-md` after feature-heavy sessions to trim drift.
|
||||
|
||||
### Public-repo writing hygiene
|
||||
@@ -167,14 +174,14 @@ This is a **public, distributed repo** — every committed file (CHANGELOG, DEVL
|
||||
- **Branching model (as of 2026-04-19):** `main` + `dev`. Feature branches target `dev`, not `main`. `main` receives only release merges (and tags). No straight-to-main exemption — even single-file typos go through `dev`.
|
||||
- **Merge style:** `git merge --no-ff` — no squash. Preserves per-commit trail for agent-team branches on every merge in the chain (feature → dev → main).
|
||||
- **Merging ≠ releasing.** Feature branches land on `dev` continuously as CI goes green; each PR appends to `[Unreleased]` in `CHANGELOG.md` on `dev`. Releases are a separate act — cut when accumulated state is worth shipping, not per-feature. See `RELEASE.md` "When to cut a release."
|
||||
- **Version bumps happen on `dev`, then release-merge to `main`.** Bump only the surface being released: `scripts/bump-android-version.sh` for `android-vX.Y.Z`, `scripts/bump-server-version.sh` for `server-vX.Y.Z`, and `desktop/package.json` for `desktop-vX.Y.Z`. The release commit lives on `dev`, then a release PR merges `dev` → `main` with `--no-ff`, then the surface tag is cut from `main`.
|
||||
- **Version bumps happen on `dev`, then release-merge to `main`.** Bump only the surface being released: `scripts/bump-android-version.sh` for `android-vX.Y.Z`, `scripts/bump-plugin-version.sh` for `plugin-vX.Y.Z`, and `desktop/package.json` for `cli-vX.Y.Z`. The release commit lives on `dev`, then a release PR merges `dev` → `main` with `--no-ff`, then the surface tag is cut from `main`.
|
||||
- **Server tracks `dev` for staging.** The hermes-host deployment pulls `dev` so merged features are exercised before they reach a tag. Released state lives on tags cut from `main`.
|
||||
- **Branch protection** on `main` — direct push blocked; only release-merge PRs from `dev` land here. `dev` also requires CI to pass on PRs but accepts feature-branch merges freely.
|
||||
|
||||
### Testing
|
||||
- **Android:** JUnit + Compose testing for UI, MockK for mocks
|
||||
- **Python:** `python -m unittest plugin.tests.test_<name>` — avoid bare `pytest` (conftest imports `responses` which may not be installed in the venv)
|
||||
- **CI is split by path:** `.github/workflows/ci-android.yml` runs on app/Gradle changes; `.github/workflows/ci-server.yml` runs on plugin/Python changes. Both trigger on pushes to `main` and `dev` and on PRs targeting either. Build + tests must pass before merge to `dev`; release-merge to `main` requires the same.
|
||||
- **CI is split by path:** `.github/workflows/ci-android.yml` runs on app/Gradle changes; `.github/workflows/ci-plugin.yml` runs on plugin/Python changes. Both trigger on pushes to `main` and `dev` and on PRs targeting either. Build + tests must pass before merge to `dev`; release-merge to `main` requires the same.
|
||||
|
||||
## Key Files
|
||||
|
||||
@@ -260,9 +267,12 @@ This is a **public, distributed repo** — every committed file (CHANGELOG, DEVL
|
||||
| `plugin/tools/android_tool.py` | 18 `android_*` tool handlers (14 baseline + send_sms, call, search_contacts, return_to_hermes); `android_screenshot` first consumer of `register_media()` |
|
||||
| `plugin/tools/android_navigate.py` | Vision-driven navigation loop; up to 20 iterations; `llm_gap` error until vision client wired |
|
||||
| `plugin/pair.py` | QR payload builder + CLI; `build_payload(sign=True)`; `--register-code` fallback |
|
||||
| `plugin/doctor.py` | `hermes relay doctor`; checks standard upstream API/dashboard reachability, Relay loopback state, plugin layout, and compat hook state |
|
||||
| `plugin/compat.py` | `hermes relay compat status/install/remove`; owns the optional `hermes_relay_bootstrap.pth` lifecycle |
|
||||
| `plugin/hermes_relay_bootstrap/` | Plugin-owned runtime compatibility patch; skips native routes per method/path; retire only after remaining config/memory/legacy skill/slash gaps are handled |
|
||||
| `install.sh` | Canonical installer — 6 steps; idempotent; drops `hermes-relay-update` shim |
|
||||
| `uninstall.sh` | Canonical uninstaller; reverses install.sh; never touches `.env` or `state.db` |
|
||||
| `hermes_relay_bootstrap/` | Runtime compatibility patch; skips native routes per method/path; retire only after remaining config/memory/legacy skill/slash gaps are handled |
|
||||
| `hermes_relay_bootstrap/` | Legacy import shim for old `.pth` files and editable installs |
|
||||
| **Plugin — Dashboard** | |
|
||||
| `plugin/dashboard/manifest.json` | Declares tab, entry bundle, and FastAPI module for hermes-agent discovery |
|
||||
| `plugin/dashboard/plugin_api.py` | FastAPI router proxying 5 routes to relay over loopback; `/pairing` body = API-server overrides (host/port/tls/api_key), relay URL auto-derived |
|
||||
@@ -272,7 +282,17 @@ This is a **public, distributed repo** — every committed file (CHANGELOG, DEVL
|
||||
| `desktop/package.json` | `@hermes-relay/cli` package manifest — Node ≥21, one `hermes-relay` bin, pre-built dist |
|
||||
| `desktop/bin/hermes-relay.js` | Tiny shim: `import('../dist/cli.js').then(m => m.main())` + error surfacing |
|
||||
| `desktop/src/chatAttach.ts` | captureClipboardImage / captureScreenshot / readImageFile; ships base64 to server via `image.attach.bytes` RPC before next prompt.submit |
|
||||
| `desktop/src/cli.ts` | argv parser + subcommand dispatcher — bare → `shell` (PTY), positional-only → `chat` |
|
||||
| `desktop/src/cli.ts` | argv parser + subcommand dispatcher — bare → `shell` (PTY), positional-only → `chat`; command-scoped `--help` falls through to each command |
|
||||
| `desktop/src/lib/theme.ts` | Shared ANSI palette + `colorEnabled()` + `Theme` (semantic helpers, `statusDot`) — single visual language; `--no-color`/`NO_COLOR`/TTY aware |
|
||||
| `desktop/src/lib/table.ts` | Zero-dep column-aligned table renderer (ANSI-width aware, last column flexes to terminal width) — used by devices/sessions/audit |
|
||||
| `desktop/src/lib/spinner.ts` | Stderr braille spinner for slow ops (pair probe, gateway connect); no-op when piped/quiet/json |
|
||||
| `desktop/src/lib/usage.ts` | `UsageSpec` + `renderUsage`/`printUsage`/`unknownSubcommand` — per-subcommand `--help` + self-documenting sub-verb fallback |
|
||||
| `desktop/src/lib/hints.ts` | `suggestedFix(err, ctx)` → next-step command (re-pair on auth fail, etc.); `formatError` renders error + hint |
|
||||
| `desktop/src/lib/logo.ts` | Slim box-drawing "Hermes Relay" wordmark; shown atop `--help`, first-run welcome, REPL header, and `hermes-relay logo`; theme/no-color aware |
|
||||
| `desktop/src/lib/auditLog.ts` | Local desktop-tool audit JSONL (`~/.hermes/desktop-audit.jsonl`); router appends per dispatch; backs `audit` command (relay's ring is loopback-only) |
|
||||
| `desktop/src/lib/daemonStatus.ts` | Daemon heartbeat file (`~/.hermes/daemon-status.json`) + `isPidAlive` liveness; backs `daemon --status` |
|
||||
| `desktop/src/commands/audit.ts` | `hermes-relay audit` — tails the local audit log into a table (WHEN/TOOL/STATUS/DETAIL); `--limit`, `--json` |
|
||||
| `desktop/src/commands/relay.ts` | `hermes-relay relay info/security/context` — relay-server management surface; info/security loopback-only, context works remote with bearer |
|
||||
| `desktop/src/commands/chat.ts` | REPL + one-shot + piped-stdin; `runOneTurn` returns `{promise, cancel}` for safe SIGINT; auto-wires `DesktopToolRouter` when consented |
|
||||
| `desktop/src/commands/shell.ts` | Pipes the `terminal` relay channel to raw-mode stdin/stdout; post-attach `exec hermes` 350ms after tmux settles; `Ctrl+A .` detach / `Ctrl+A k` kill / `Ctrl+A Ctrl+A` literal |
|
||||
| `desktop/src/commands/pair.ts` | Either 6-char code + `--remote`, or full v3 QR via `--pair-qr` — probes + picks endpoint, records role; `--grant-tools` (TTY prompt) / `--auto-grant-tools` (silent) stamp `toolsConsented` so `daemon` works without a `shell` round-trip |
|
||||
@@ -308,10 +328,17 @@ This is a **public, distributed repo** — every committed file (CHANGELOG, DEVL
|
||||
| **Desktop CLI — dev iteration** | |
|
||||
| `npm run smoke` (in `desktop/`) | Builds Windows binary + runs `--version` / `--help` / `doctor`, fails loud on zero-output. Local pre-flight before cutting any tag. |
|
||||
| `npm run gen:version` | Regenerates `src/version.ts` from `package.json`. Runs automatically before every `build` / `build:bin:*`. |
|
||||
| `release-desktop.yml → Smoke-test Linux binary` step | CI-side equivalent: runs compiled Linux binary through the same 3-command check before uploading assets. Catches silent-exit-0 + segfault classes. |
|
||||
| `release-cli.yml → Smoke-test Linux binary` step | CI-side equivalent: runs compiled Linux binary through the same 3-command check before uploading assets. Catches silent-exit-0 + segfault classes. |
|
||||
| **Server — Desktop tool routing (Phase B)** | |
|
||||
| `plugin/relay/channels/desktop.py` | Mirrors `bridge.py` — `desktop.command`/`desktop.response`/`desktop.status`, UUID-correlated futures, 30s timeout, single-client MVP, per-session advertised-tools set |
|
||||
| `plugin/tools/desktop_tool.py` | 24 `desktop_*` tools (fs/shell/powershell/process/jobs/transfer/health) — registers with `tools.registry` under `desktop` toolset; per-tool `check_fn` pings `/desktop/_ping?tool=<name>`; `desktop_health` is `_RELAY_ONLY` and pings `/desktop/health` so it works even when the client is wedged |
|
||||
| **Gradle modules — experimental Quest/XR (in development)** | |
|
||||
| `relay-core/` | [EXPERIMENTAL] Android library (`com.axiomlabs.hermesrelay.core`) — shared pairing/transport/terminal/voice/wire for the Quest port; not yet wired into the shipped `:app` |
|
||||
| `relay-ui/` | [EXPERIMENTAL] Android library (`com.axiomlabs.hermesrelay.ui`) — shared Compose UI (sphere, terminal WebView, QR scanner) for the Quest port; carries its own sphere copy |
|
||||
| `quest/` | [EXPERIMENTAL] Meta Spatial SDK Quest/XR app — gradle `includeBuild("quest")`; needs further development, not shipped |
|
||||
| **Tooling — dev iteration (not shipped)** | |
|
||||
| `ui-preview/` | Desktop Compose Hot Reload harness — JVM Compose for Desktop; source-shares `MorphingSphereCore` from `:relay-ui`; `Main.kt` gallery; see `ui-preview/README.md` |
|
||||
| `app/src/test/.../screenshots/StoreScreenshotTest.kt` | Roborazzi host-side store/docs screenshot renderer — deterministic, no device, exact 1080×2160; reuses real components+chrome with mock data; `capture(name, themeId){…}` renders any view; see `docs/screenshot-automation.md` §Deterministic rendering (JDK-21 + no-plugin gotchas) |
|
||||
|
||||
## What NOT to Do
|
||||
|
||||
@@ -320,7 +347,8 @@ This is a **public, distributed repo** — every committed file (CHANGELOG, DEVL
|
||||
- **Don't use Ktor for networking** — OkHttp for WebSocket
|
||||
- **Don't use plaintext WebSocket** — `wss://` only, even in development
|
||||
- **Don't put documentation in root** — long-form docs go in `docs/`
|
||||
- **Don't forget DEVLOG.md** — update it
|
||||
- **Don't forget DEVLOG.md** — update it (record *what happened*)
|
||||
- **Don't bury follow-ups** — deferred work / known gaps go in `TODO.md`, never in DEVLOG or one-off code/doc comments
|
||||
|
||||
## MCP Tooling
|
||||
|
||||
@@ -359,7 +387,7 @@ Curls every bridge HTTP route via `localhost:8767`. Catches the silent-drop regr
|
||||
1. **Edit locally** — Windows checkout. Both plugin (`plugin/`) and app (`app/`) live here.
|
||||
2. **Python syntax check** — `python -m py_compile plugin/<file>.py`. Full tests run on the server.
|
||||
3. **Kotlin changes** — do NOT run `gradle build`. Bailey builds via Android Studio's ▶ button. Never `adb install` from Claude.
|
||||
4. **Before pushing Kotlin changes** — run `./gradlew lint` locally. It's the exact task CI runs (see `.github/workflows/ci.yml` → `gradlew lint` fallback) and catches errors Android Studio's live inspections miss — e.g. `UnsafeOptInUsageError` with `kotlin.OptIn` vs `androidx.annotation.OptIn`, `FlowOperatorInvokedInComposition` (mapped flows inside Composables), Media3 `@UnstableApi` propagation. Lint is a hard blocker in CI: Build + Test show "skipping" until lint passes, and lint prints only the **first failure** before aborting — so CI iterations reveal errors one at a time while a single local lint run surfaces all of them.
|
||||
4. **Before pushing Kotlin changes** — run `./gradlew lint` locally. It's the exact task CI runs and catches errors Android Studio's live inspections miss — e.g. `UnsafeOptInUsageError` with `kotlin.OptIn` vs `androidx.annotation.OptIn`, `FlowOperatorInvokedInComposition` (mapped flows inside Composables), Media3 `@UnstableApi` propagation. Android CI runs lint alongside build/test for faster feedback, but a local lint run still surfaces issues before the workflow spends runner time compiling and packaging.
|
||||
5. **Commit + push** — feature branch off `dev`, merged back to `dev` via PR. `main` is reserved for release merges.
|
||||
6. **Pull + restart on server** — see Server Deployment below.
|
||||
7. **Test on phone** — Bailey builds from Studio, installs to Samsung device, pairs via `/hermes-relay-pair`.
|
||||
@@ -378,6 +406,12 @@ Server is a Linux box running hermes-agent with hermes-relay editable-installed
|
||||
|
||||
**Update:** `hermes-relay-update` (idempotent, re-fetches install.sh). Or manually: `git pull --ff-only && systemctl --user restart hermes-relay`.
|
||||
|
||||
**Compat hook:** `hermes relay compat status/install/remove` manages only the
|
||||
optional `hermes_relay_bootstrap.pth` startup hook. New installs load the
|
||||
plugin-owned bootstrap from `plugin/hermes_relay_bootstrap/`; the repo-root
|
||||
package is only a legacy import shim. Vanilla Hermes chat, Manage, and dashboard voice
|
||||
must not depend on this hook.
|
||||
|
||||
**Key conventions:**
|
||||
- Phone re-pairs after each relay restart (SessionManager is in-memory; wiped on restart)
|
||||
- Use `python -m unittest` not `pytest` — conftest imports `responses` which may not be installed
|
||||
@@ -396,20 +430,25 @@ Server is a Linux box running hermes-agent with hermes-relay editable-installed
|
||||
|
||||
See [RELEASE.md](RELEASE.md) for the full recipe.
|
||||
|
||||
- **Version source:** `gradle/libs.versions.toml` (`appVersionName`, `appVersionCode`)
|
||||
- **Bump atomically:** `bash scripts/bump-version.sh <new-version>` — updates all three sources
|
||||
- **`appVersionCode` is monotonic** — always increment across prereleases
|
||||
- **Cut a release:** bump → commit → `git tag vMAJOR.MINOR.PATCH` → push tag → CI builds + GitHub Release
|
||||
- **Android version source:** `gradle/libs.versions.toml` (`appVersionName`, `appVersionCode`); bump with `scripts/bump-android-version.sh`
|
||||
- **Relay plugin version source:** `pyproject.toml`; keep plugin/dashboard metadata synced with `scripts/check-plugin-version-sync.py`; bump with `scripts/bump-plugin-version.sh`
|
||||
- **Desktop CLI version source:** `desktop/package.json`; regenerate `desktop/src/version.ts` with `npm run gen:version`
|
||||
- **Track audit:** `python scripts/check-version-tracks.py` reports Android, plugin, and CLI versions without forcing them to match
|
||||
- **`appVersionCode` is monotonic** — always increment across Android prereleases
|
||||
- **Cut a release:** bump the target surface → commit → merge `dev` to `main` → tag with `android-v*`, `plugin-v*`, or `cli-v*` → push tag → CI builds + GitHub Release
|
||||
- **Required secrets:** `HERMES_KEYSTORE_BASE64`, `HERMES_KEYSTORE_PASSWORD`, `HERMES_KEY_ALIAS`, `HERMES_KEY_PASSWORD`
|
||||
|
||||
## Integration Points
|
||||
|
||||
| Surface | Endpoint | Notes |
|
||||
|---------|----------|-------|
|
||||
| Chat (gateway) | Dashboard `POST /api/auth/ws-ticket` -> WS `/api/ws` | Vanilla Hermes dashboard/tui_gateway path; live thinking/reasoning; requires dashboard auth |
|
||||
| Chat streaming | `POST /v1/runs` → `GET /v1/runs/{id}/events` | Structured tool events; async run-control path |
|
||||
| Chat (sessions) | `POST /api/sessions/{id}/chat/stream` | Native upstream session-persisted SSE; preferred when capability probe finds it |
|
||||
| Chat (compat) | `POST /v1/chat/completions` (stream=true) | Inline tool annotations only |
|
||||
| Session CRUD | `GET/POST/PATCH/DELETE /api/sessions` | Native upstream (#33134); bootstrap fallback only for old builds |
|
||||
| Manage | Dashboard `/api/status`, `/api/auth/me`, `/api/config`, `/api/profiles/*`, `/api/env`, `/api/model/*`, `/api/mcp/*` | Vanilla Hermes dashboard surface; do not proxy through Relay |
|
||||
| Vanilla Hermes voice | Dashboard `POST /api/audio/transcribe`, `POST /api/audio/speak` | Vanilla Hermes no-plugin voice; uses dashboard session from Manage |
|
||||
| Pairing (QR) | `POST /pairing/register` (loopback only) | Via `/hermes-relay-pair` or `hermes-pair` shim; accepts optional `endpoints` for multi-endpoint QRs |
|
||||
| Pairing (multi-endpoint) | QR `endpoints` array (ADR 24) | `hermes: 3` schema; ordered `lan`/`tailscale`/`public`/... candidates; phone re-probes on network change |
|
||||
| Pairing auth | WSS `auth.ok` payload | Includes `expires_at`, `grants`, `transport_hint` |
|
||||
@@ -420,6 +459,8 @@ See [RELEASE.md](RELEASE.md) for the full recipe.
|
||||
| Voice transcribe | `POST /voice/transcribe` | multipart/form-data; bearer auth |
|
||||
| Voice synthesize | `POST /voice/synthesize` | JSON → audio/mpeg; max 5000 chars |
|
||||
| Voice config | `GET /voice/config` | Returns current tts/stt provider info |
|
||||
| Plugin diagnostics | `hermes relay doctor --json` | Reports upstream route reachability, Relay loopback state, plugin layout, and legacy bootstrap state |
|
||||
| Compat hook lifecycle | `hermes relay compat status/install/remove` | Optional legacy API compatibility hook; not required for the standard path |
|
||||
| Notifications | `GET /notifications/recent?limit=N` | Loopback callers skip bearer |
|
||||
| Relay health | `GET /health` on `:8767` | Used by `RelayHttpClient.probeHealth()` |
|
||||
| Capabilities | `GET /v1/capabilities` plus targeted `HEAD` probes | Prefer capabilities when present; HEAD probes keep mixed-version fallback working |
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
# Hermes-Relay-CLI v__VERSION__
|
||||
|
||||
**Release Date:** 2026-06-21
|
||||
**Since the previous CLI release:** a first-class command surface — activity audit, relay inspection, a background daemon, a polished visual layer, and v1.2.0 server parity.
|
||||
|
||||
This is a broad CLI uplift: new commands for seeing what the agent did and inspecting the relay, a daemon you can run in the background, and a consistent themed interface with per-command help. Everything is additive — existing commands, flags, and scripts keep working.
|
||||
|
||||
**Experimental phase.** Assets are unsigned — Windows SmartScreen and macOS Gatekeeper will warn on first launch. Windows ships a tray installer as the primary desktop surface; CLI binaries remain available for terminal/headless use and for macOS/Linux.
|
||||
|
||||
## What's changed
|
||||
|
||||
### Added
|
||||
- **`hermes-relay audit`** — see what the remote agent has run on this machine through the desktop tools (tool, status, detail), read from a local log. No network, no auth; works whether the relay is local or remote.
|
||||
- **`hermes-relay relay`** — inspect the relay server: `relay context` audits the system-prompt context the relay injects into the agent (works from any paired machine), and `relay info` / `relay security` report server state for operators on the relay host.
|
||||
- **Background daemon.** `hermes-relay daemon start` runs the headless tool router in the background — no console window, survives closing the terminal — with `daemon stop` and `daemon status` to manage it. Bare `daemon` still runs in the foreground. Logs go to `~/.hermes/daemon.log`.
|
||||
- **Per-command help.** Every subcommand answers `--help`, and `devices` / `sessions` / `plugins` / `voice` / `relay` print their own usage (sub-commands, flags, examples) instead of a terse "unknown sub-verb".
|
||||
- **Startup banner.** A slim "Hermes Relay" wordmark shows atop `--help`, the first-run welcome, and the chat REPL; `hermes-relay logo` prints it on demand. Suppressed for piped / `--json` / `--no-color` output.
|
||||
|
||||
### Changed
|
||||
- **Visual + ergonomics refresh.** One consistent color theme across the CLI, aligned tables for `devices` / `sessions`, on/off status dots, and progress spinners for slow operations (the multi-endpoint pairing probe and the gateway connect) so nothing looks hung. Errors now suggest the fix (e.g. re-pair on auth failure).
|
||||
- **Smoother pairing.** The multi-endpoint probe shows per-endpoint progress and latency; a near-expiry session warns before it fails and prints the exact re-pair command; and a bare `ws://host` (no port) defaults to `:8767`.
|
||||
- **Voice + consent transparency.** `voice` now surfaces enhanced-voice capabilities (Gemini tone tags / persona, xAI speech tags); the desktop-tool consent prompt is clear that it persists per relay and points at `hermes-relay audit`; and computer-use's observe → grant → act flow is documented in `--help`.
|
||||
|
||||
## Install
|
||||
|
||||
**Windows tray app (PowerShell):**
|
||||
```powershell
|
||||
irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
|
||||
```
|
||||
|
||||
**Windows CLI only:**
|
||||
```powershell
|
||||
$env:HERMES_RELAY_INSTALL_SURFACE='cli'; irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex
|
||||
```
|
||||
|
||||
**macOS / Linux CLI:**
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.sh | sh
|
||||
```
|
||||
|
||||
Pin this specific release with `HERMES_RELAY_VERSION=__TAG__`.
|
||||
|
||||
## Verify
|
||||
|
||||
```text
|
||||
hermes-relay --version
|
||||
hermes-relay pair --remote ws://<host>:8767
|
||||
hermes-relay shell
|
||||
```
|
||||
|
||||
Open **Hermes Relay Desktop** from the Windows Start menu for tray pairing, devices, task log, settings, pause, and emergency stop.
|
||||
|
||||
See [Desktop docs](https://codename-11.github.io/hermes-relay/desktop/) for full usage.
|
||||
@@ -1,5 +1,366 @@
|
||||
# Hermes-Relay — Dev Log
|
||||
|
||||
## 2026-06-21 — Profile lock + voice fixes (orchestration batch)
|
||||
|
||||
**Why.** User-requested batch (TODO User-Added) covering the profile-lock setting and the concrete voice TODOs. Investigated and implemented via a planning→implementation orchestration pass: four read-only investigators, then three disjoint file-ownership implementation lanes. All changes are client-side Kotlin; the server-side realtime-voice half is deferred to TODO. **Unbuilt at time of writing — pending Studio build + `./gradlew lint`.**
|
||||
|
||||
- **Profile lock (new).** Per-connection "lock to one profile": `data/ProfileLockStore.kt` (twin of `ProfileSelectionStore`, same `profile_selections` DataStore; `__server_default__` sentinel via `AgentDisplay`); `ProfileController` gains `lockedProfileName`/`isProfileLocked` + `lockProfile`/`unlockProfile`, with `selectProfile` no-op'd when locked and `resolvePendingProfileFrom` preferring (and holding on missing) the locked target; `ConnectionViewModel` delegations + lock-clear at reset/remove sites + a lock-flow observer; `ConnectionInfoSheet` collapses the picker to a static "Locked to <name>" row when locked; `SettingsScreen` adds the `ProfileLockCard` + dialog — the one surface that still lists all profiles, with a "not found on this server" banner.
|
||||
- **Voice override in 'auto' (fix).** `VoiceViewModel.shouldPreferRealtimeVoice()` gated on `.route` (configured) instead of `.effectiveRoute` (resolved), so 'auto'+relay-ready never engaged the override-capable relay path and fell back to the host-global Standard `/api/audio/speak` (no override slot) — hence only 'Relay' applied the chosen voice. Switched to `effectiveRoute`. Also wired `connectionId` for per-profile voice-prefs namespacing (`RelayApp` calls `setVoicePrefsConnection(activeConnectionId)` and passes `connectionId` to `VoiceSettingsScreen`, which now takes the param and feeds `setActiveScope`).
|
||||
- **Realtime voice (fix, client half).** Stall: `RelayVoiceClient.awaitRealtimeAgentCompletion` relaxes the 90s idle watchdog once a `hermes.run.promoted`/long run is seen, keeping the 5-min max-turn backstop. Over-chatty status: per-turn throttle in `VoiceViewModel.emitStatus` (≥22s gap, ≤3 spoken/turn). Waveform: realtime `outputAudioActive` now gates on real playback-start (`RealtimePcmPlayer` head-move/`playbackAmplitude`) instead of decoded-byte RMS, matching the basic-TTS path.
|
||||
- **Voice UI.** Profile icon now shows in the floating overlay header pill (`VoiceModeOverlay` reads `LocalAgentIconPath`; sphere/pet stays the fallback). Voice Settings: invalid engine/route combos made unreachable (RealtimeAgent disabled without relay, unavailable routes disabled, `coerceAudioRoute` auto-corrects on engine switch / relay loss); long dropdown/provider labels get `maxLines=1`+ellipsis.
|
||||
- **Method.** Disjoint file-ownership lanes (1: VoiceViewModel/RelayVoiceClient/RelayApp; 2: VoiceSettingsScreen/VoiceModeOverlay; 3: ProfileController/ConnectionViewModel/ConnectionInfoSheet/SettingsScreen/ProfileLockStore) so parallel implementers never touched the same file, and `ChatScreen.kt` was avoided (owned by a concurrent session). Pure helpers (`coerceAudioRoute`, `shouldSpeakStatusNow`, `shouldMarkRealtimeOutputActive`) extracted for unit-testing.
|
||||
- **Deferred.** See TODO.md "Orchestration batch (2026-06-21)": streaming-path override question, upstream per-profile Standard voice, ChatScreen lock glyph, export decision, CHANGELOG entries, on-device verification.
|
||||
- **Follow-up (same day).** Built + deployed to device as **1.2.1 / versionCode 15** (`:app:assembleSideloadDebug`, clean). Server-side realtime half implemented in `broker.py` (heartbeat-while-task-running + calmer spoken-status cadence) with `plugin/tests/test_realtime_heartbeat.py` (11) + promotion regression (5) green — **deployed**: committed `d1820fb` → pushed to `origin/dev` → server `~/.hermes/hermes-relay` fast-forwarded + `hermes-relay` restarted (active, clean startup on ws://…:8767). New Kotlin unit suite green: `ProfileLockStoreTest` (9, in-memory DataStore harness), `ProfileControllerLockTest` (8, Robolectric), `CoerceAudioRouteTest` (7), `VoiceStatusGatesTest` (12) — 36/36 via `:app:testSideloadDebugUnitTest`.
|
||||
|
||||
## 2026-06-21 — Desktop CLI first-class pass (audit-driven)
|
||||
|
||||
**Why.** The desktop CLI hadn't had feature work since 2026-05-19 while the relay plugin shipped a full v1.2.0 wave (relay-management surface, enhanced voice, context injection). A four-axis audit (command UX/visuals, pairing, desktop-tools, plugin parity) found the CLI surfaced ~⅓ of current plugin capability with an ad-hoc visual layer and weak discoverability. This pass closes those gaps; all changes are confined to `desktop/` (no Android, no Python).
|
||||
|
||||
- **Shared zero-dep UI foundation (`desktop/src/lib/`).** `theme.ts` (one ANSI palette + `colorEnabled` + `Theme` with `statusDot`/semantic helpers, extracted from `renderer.ts`'s pattern), `table.ts` (ANSI-width-aware column renderer, last column flexes to terminal width), `spinner.ts` (stderr braille spinner, no-op when piped/quiet/json), `hints.ts` (`suggestedFix(err)` → next-step command + `formatError`), `usage.ts` (`UsageSpec` → per-subcommand `--help` + self-documenting unknown-sub-verb), `logo.ts` (slim box-drawing wordmark).
|
||||
- **Discoverability.** Fixed the `cli.ts` dispatch so command-scoped `--help` reaches the command (was always short-circuiting to global help). Added `--help` + usage specs across `devices`/`sessions`/`status`/`tools`/`plugins`/`voice`/`relay`/`pair`/`daemon`/`doctor`/`workspace`/`paste`; ported list output (`devices`/`sessions`) to aligned tables + status dots; routed command failures through `formatError` (actionable hints); replaced `doctor`'s inconsistent `!!` warning markers with themed `⚠` lines.
|
||||
- **Pairing.** Threaded an `onProbe` callback into `probeCandidatesByPriority` so `pair` shows per-endpoint progress + latency during the multi-endpoint race; `credentials.ts` warns (TTY-only) when a stored token is near/at expiry with the exact re-pair command; `relayUrlPrompt.normalizeRelayUrl` defaults a bare `ws://host` to `:8767` (scoped to `ws://` so `wss://` proxy fronts on :443 aren't broken), surfaced not silent.
|
||||
- **Desktop tools first-class.** New `hermes-relay audit` backed by a local JSONL (`~/.hermes/desktop-audit.jsonl`) the `DesktopToolRouter` appends per dispatch — the relay's ring buffer is loopback-only, so the client (the executor) is the right source of truth and this works against a remote relay with no auth. Consent prompt rewritten to state persistence + point at `audit`; computer-use's observe→grant→act flow documented in `--help`.
|
||||
- **Daemon observability + background run.** `daemon` writes a heartbeat file (`~/.hermes/daemon-status.json`) on each lifecycle transition + a 30s tick; `daemon status` reads it, cross-checks pid liveness (`process.kill(pid,0)`), and exits non-zero when stale. Added `daemon start` (detached spawn — `detached:true` + `windowsHide:true` + stdio→`~/.hermes/daemon.log` + `unref`, no console window, survives terminal close) and `daemon stop` (kills the status-file pid + clears it); bare `daemon` still runs foreground. Validated start→status→stop on Windows against the live relay. A true OS service (reboot/login auto-start) remains the deferred follow-up.
|
||||
- **Dev loop.** Added `desktop/scripts/dev-install.mjs` + `npm run dev:install` — builds the bun binary for the current platform and drops it over the curl-installed `~/.hermes/bin/` binary (backs the old one up as `.bak`, surfaces EBUSY as "stop the daemon first"). Closes the gap where local changes could only be exercised via `npx tsx`, never as the real global binary.
|
||||
- **Plugin v1.2.0 parity.** `voice` now renders the `/voice/config` `enhanced` block (Gemini tone-tags/persona, xAI speech-tags). New `hermes-relay relay info|security|context` over the relay-management surface — `context` (the injected-system-prompt audit) works remote with a bearer; `info`/`security` are loopback-only and say so on a remote 403. Deliberately did **not** add a CLI-vs-server "version skew" warning — the two are on independent release tracks, so it would be a false alarm.
|
||||
- **Logo.** Slim box-drawing "Hermes Relay" wordmark atop `--help`, the first-run welcome, the chat REPL, and a `logo` command; theme/no-color aware, never on piped/`--json` stdout.
|
||||
- **Verification.** `npm run type-check` and `npm run build` (tsc) green. Runtime-smoked via `npx tsx src/cli.ts` (NO_COLOR): `--help`, `logo`, `devices --help`/`devices bogus` (usage fallback), `audit` (empty-state), `daemon --status` (no-daemon), `doctor`, `workspace`. Docs: CHANGELOG `[Unreleased]`, `desktop/README.md` (audit/relay/daemon-status sections), CLAUDE.md desktop Key Files refreshed. Version bump (`alpha.18`→`alpha.19`) left to the operator — not cutting a CLI release this cycle.
|
||||
|
||||
## 2026-06-20 — Release-prep: android-v1.2.0 + plugin-v1.2.0
|
||||
|
||||
**Why.** Cut a combined 1.2.0 across both lockstep surfaces (both were at 1.1.0). The accumulated `[Unreleased]` block had captured the major feature arcs but a second wave had landed undocumented — audited every commit since the `*-v1.1.0` tags and backfilled the changelog before promoting it.
|
||||
|
||||
- **Versions.** `bump-android-version.sh 1.2.0` (`appVersionName 1.1.0→1.2.0`, `appVersionCode 13→14`); `bump-plugin-version.sh 1.2.0` (pyproject + `plugin/relay/__init__.py` + plugin.yaml + dashboard manifest/package/lock, all in sync). `check-version-tracks.py` + `check-plugin-version-sync.py --expect 1.2.0` green.
|
||||
- **CHANGELOG backfill.** Promoted `[Unreleased]` → `[1.2.0] - 2026-06-20` with a fresh empty `[Unreleased]`. Added the missing shipped features the accumulator had skipped: **agent pets** (swappable animated avatar + reactivity + in-app add/remove + AI authoring kit), per-profile agent icons + single-image avatars, **in-app crash reporting**, clean text-flow mode, the permissions-review screen, attachment previews; **Changed**: "Standard"→"Vanilla Hermes" rename, QR camera hardening for foldables, viewer landscape rotation; **Fixed**: PDF mid-render crash, the `kotlin.Result`-in-suspend `ClassCastException` on server images, side-loaded avatar/skin storage path, reopened-session model + media-badge fixes.
|
||||
- **Release notes.** Rewrote `RELEASE_NOTES.md` (Android, "Make it yours" framing) and `PLUGIN_RELEASE_NOTES.md` (enhancement-layer + enhanced voice). Updated in-app `whats_new.txt`, Play `release-notes/en-US/default.txt` (438/500 chars), and the `docs/play-store-listing.md` What's-new block.
|
||||
- **Scrub.** Grep'd the `[1.2.0]` block for names / private infra / fork plumbing — clean (only pre-existing released blocks carry the LAN host IP from old desktop-alpha entries; out of scope for this cut, flagged separately).
|
||||
- **Verification.** `python -m unittest plugin.tests.test_enhancements plugin.tests.test_terminal_channel` (20 pass). Android AAB build + `keytool` cert verify is Studio-side (Bailey) per the dev loop. Prep committed on `dev` in two commits (`release(android)` / `release(plugin)`); merge-to-`main` + tags deferred to operator.
|
||||
|
||||
## 2026-06-20 — Static-image avatars + per-profile agent icon (Android)
|
||||
|
||||
**Why.** Two requests: a custom avatar shouldn't require authoring an animated pack (a single image should work), and each agent profile should be able to wear its own small icon beside its name — client-side, mirroring the existing local-name override.
|
||||
|
||||
- **Static-image import (`PetImporter`).** "Add a pet" now accepts a single image (`.png`/`.jpg`/`.gif`/`.webp`), not just a `.zip` — detected by **magic bytes**, not the file name. An image is auto-wrapped as a one-frame static pet (written as `idle.png` with a synthesized minimal `pet.json`), so a static avatar needs no manifest authoring. The renderer already supported a one-frame `idle`; this is purely import ergonomics. `importZip` → `importUri`; `ConnectionViewModel.importPetFromZip` → `importPet`. Test covers the image-wrap path.
|
||||
- **Per-profile agent icon (client-side).** A direct twin of `ProfileDisplayAliasStore`: new `ProfileIconStore` (own DataStore `profile_icons`, keyed per `(connection, profile)`, **never sent to Hermes**). It stores a **path** to an image copied into `files/profile-icons/` (not a SAF URI, so it survives without persistable permission). Wired through `ProfileController` next to `profileDisplayAlias` (`profileIcon` StateFlow + `setProfileIcon`/`clearProfileIcon` + the copy), exposed on `ConnectionViewModel`, provided at the app root as `LocalAgentIconPath`, and rendered beside the agent name in `MessageBubble` **and** as the **header avatar** in the agent sheet, the chat top bar, and Settings — via a shared `AgentAvatarFace` that shows the icon (Coil from the file path) or falls back to the name's initial. The picker (`AgentIconRow`) sits right under the local-name row in `ConnectionInfoSheet`. Scope: small name-adjacent icon only — the big empty-chat/voice avatar stays global. `ProfileIconStore` cleared alongside the alias on connection removal; test mirrors the alias store's.
|
||||
- **Verification.** `:app:assembleSideloadDebug` + `PetImporterTest`/`ProfileIconStoreTest` <pending>. Installed via `adb install -r`. On-device check (import an image as a pet; set a profile icon and see it by the name) in TODO.
|
||||
|
||||
## 2026-06-20 — Pet state preview in Appearance (Android)
|
||||
|
||||
**Why.** Testing a pet meant *inducing* each state by driving the agent (run a tool to see `working`, fail a turn for `error`, start voice for `speaking`/`listening`) — painful. An in-app preview turns Appearance into a pet test harness.
|
||||
|
||||
- **Live preview (`AppearanceSettingsScreen`).** Under the speed/stabilize controls (pet selected only): a ~140 dp canvas rendering the active pet, a `FilterChip` row for the seven sustained states (`Idle · Thinking · Working · Writing · Speaking · Listening · Error`), and `Greet`/`Done` buttons that replay the one-shots. Pure UI on the existing `AgentAvatar` seam — no new ViewModel/pref/renderer; it just calls `activeAvatar.Render(AvatarRenderState(state=…))` with a user-picked state, so it also reflects the live speed and stabilize settings.
|
||||
- **State→render mapping.** Base states feed `state=…`; **Working** feeds `state=Thinking, toolCallBurst=1f` (lights the overlay); **Greet** remounts the preview via a `key` (re-fires the on-appear reaction); **Done** drives a momentary `Speaking → Idle` transition on the live instance (fires the celebrate reaction), then returns to the selected chip.
|
||||
- **Verification.** `:app:assembleSideloadDebug` BUILD SUCCESSFUL; installed via `adb install -r`. On-device visual check recorded in TODO.
|
||||
|
||||
## 2026-06-20 — Pet frame auto-stabilization (Android)
|
||||
|
||||
**Why.** On-device audit of a 4×4 AI pet (Lucy) found the character's vertical center drifting 34 px across the 16 cells, with 8/16 frames touching the cell edge — the image model held *appearance* but not *position/scale*, so the pet floated upward and bled the next frame in. The renderer slices/centers exact cells faithfully, so the drift can't be cured per-frame there — but it can be neutralized by re-centering each frame on its own content.
|
||||
|
||||
- **Decode-time recenter (`PetAvatar`).** With stabilization on, `decodeClip` scans each frame's opaque pixels (alpha bbox) and stores a per-frame offset that moves the content's bbox center to the cell center; `drawPetFrame` applies it (source px → dest px, scaled). Works for sprite sheets (per-cell) and frame sequences (per-bitmap); empty/transparent frames get a zero offset. The scan is one-time per clip decode on `Dispatchers.IO` with a reused scratch buffer, so steady-state cost is nil.
|
||||
- **Global toggle, default on (`ConnectionViewModel`, `LocalPetStabilize`, `AppearanceSettingsScreen`).** A `pet_stabilize` pref → `LocalPetStabilize` provided at the app root → read in `PetAvatar.Render` (keys the decode `produceState`, so flipping re-decodes). A "Stabilize frames" Switch sits under the playback-speed slider when a pet is selected. Default on because AI sheets nearly always need it; a hand-authored pet with intentional motion can switch it off.
|
||||
- **Authoring (docs).** The prompt kit now also stresses *registration* (lock head/shoulders, same position + scale, only secondary motion) so the art improves at the source — stabilization is the safety net for what the model still gets wrong.
|
||||
- **Verification.** `:app:assembleSideloadDebug` BUILD SUCCESSFUL; installed via `adb install -r`. Fixes the *already-installed* Lucy at render time (no re-import). On-device visual confirmation recorded in TODO.
|
||||
|
||||
## 2026-06-20 — Pet playback-speed control + cell-resolution guidance (Android + docs)
|
||||
|
||||
**Why.** Two more on-device tuning gaps: a pet that still felt fast needed re-authoring/re-importing to slow down (slow loop), and a 128 px-celled pet looked pixelated blown up to the full-screen chat background (while crisp in the small voice overlay — same frames, different scale).
|
||||
|
||||
- **Playback-speed control (`ConnectionViewModel`, `LocalPetPlaybackSpeed`, `PetAvatar`, `AppearanceSettingsScreen`).** A global multiplier pref (`pet_speed`, 0.5×–1.5×, default 1.0) surfaced as a **Slider in Appearance** when a pet is selected. Provided at the app root via a new `LocalPetPlaybackSpeed` composition local and read **live** in `PetAvatar.Render` (`rememberUpdatedState`), so dragging the slider re-times the pet instantly with no restart. Applies to every clip (including one-shots) and composes with intensity (`baseFps × speed × intensityFactor`, clamped 1–60). The sphere ignores it.
|
||||
- **Cell-resolution guidance (docs).** Pixelation is a *resolution* axis (cell px) distinct from smoothness (frame count): one frame set is contain-fit into every surface, so author for the **largest** (the chat background). Bumped the kit default to **256 px cells** (a 1024×1024 sheet for a 4×4 grid), noted 512 px is fine for a sprite sheet (decodes as one bitmap), and that the old "≲256 px" note applied to frame-*sequences*. Updated `custom-avatars.md`, `pet-prompt-kit.txt`, `pet-spec.md`.
|
||||
- **Verification.** `:app:assembleSideloadDebug` BUILD SUCCESSFUL; installed to the sideload build via `adb install -r`. Slider behavior is on-device-visual (recorded in TODO).
|
||||
|
||||
## 2026-06-20 — Pet kit defaults to 4×4 (16-frame) sheets (docs)
|
||||
|
||||
**Why.** A 4-frame (2×2) sheet reads steppy no matter the fps — the on-device Lucy made that obvious. The renderer already slices any N×M grid (`decodeClip` derives `cols`/`rows` from sheet size ÷ cell size; `drawPetFrame` indexes `col = i%cols`, `row = i/cols`), so "support 4×4" is an authoring-default change, not a renderer one.
|
||||
|
||||
- **Kit + spec default to a 4×4 grid (16 frames).** The prompt template, the manifest example, and `pet-prompt-kit.txt` now use `frameCount: 16` with fps matched to the higher count (idle ~8 → a calm ~2 s loop); 2×2 / 4 frames stays documented as the easier-to-keep-consistent fallback. `docs/pet-spec.md` states any rectangular grid works (a 4×4 sheet holds 16 frames, decoded as one bitmap regardless of cell count). Added a `PetLoaderTest` case for a 16-frame sheet.
|
||||
- **Diagnosis note.** The "still fast" report was tracked to the *installed* `pet.json` still carrying `fps 6` (the tuned `fps 3` zip post-dated the import); `intensity` was ruled out by tracing `streamingIntensity` → `0f` at idle. Audited by `adb shell cat`-ing the on-device manifest, not the repo copy.
|
||||
|
||||
## 2026-06-20 — Pet frame-loop smoothness fix (Android)
|
||||
|
||||
**Why.** First on-device pet (Lucy) animated with a periodic hitch and felt a touch fast. Root cause: `PetAvatar.Render`'s frame loop awaited `withFrameNanos` (one vsync ≈16ms) **and** `delay(1000/fps)` each iteration, so every frame waited ~16ms longer than its `frameDurSec`; the surplus accumulated until the loop forced a 2-frame skip to catch up — a visible hitch, worst at low fps.
|
||||
|
||||
- **Vsync-paced loop (`PetAvatar.Render`).** Removed the per-frame `delay`. `withFrameNanos` already suspends until the next frame, so the loop is now purely vsync-paced (~60fps) and advances the sprite only when `frameDurSec` of real time has accumulated — no double-count, no periodic skips. Intensity modulation still recomputes fps each tick; the accumulator absorbs the variable rate without skipping.
|
||||
- **Authoring guidance (docs).** Clarified that smoothness comes from frame **count**, not fps: 4 frames (2×2) is the consistent-but-steppy minimum, 8–16 (3×3 / 4×4) for fluid motion; match fps to count (calm states 3–4, not 6+). Added to `docs/pet-spec.md`, the user-docs kit, and `pet-prompt-kit.txt`; lowered the example/kit `idle`+`listening` fps to 4.
|
||||
- **Verification.** `:app:assembleSideloadDebug` BUILD SUCCESSFUL. On-device re-check pending: the test device's wireless adb dropped mid-deploy; the rebuilt APK + a tuned `lucy.zip` (idle/listening fps lowered) are staged to install + re-import once it reconnects.
|
||||
|
||||
## 2026-06-20 — In-app pet avatar add/remove/refresh (Android)
|
||||
|
||||
**Why.** The Appearance screen could *select* avatars but offered no way to **add or remove** a pet from inside the app — the only path was `adb push` into app-scoped external storage, which scoped storage stalls on (confirmed hanging on a Samsung device: a push into `/sdcard/Android/data/<pkg>/files/pets/` wrote nothing, though `adb shell ls` of the dir worked). And the avatar list loaded once at startup, so even a successfully-pushed pet never appeared without a restart. Net effect: users saw only the Sphere.
|
||||
|
||||
- **In-app import (`PetImporter.kt`, new).** "Add a pet" launches a SAF document picker; the chosen `.zip` unpacks into `pets/`. Hardened: zip-slip guard (every entry confined to a staging dir under `cacheDir`), per-file / total-size / entry-count ceilings (zip-bomb), and post-extract validation through the same `PetSpec.toAvatar` the loader uses — an archive that wouldn't render is rejected up front. Accepts either shape (pet.json at the root, or one folder deep — shallowest wins); installs under the manifest `id` (sanitized), replacing a same-named pack.
|
||||
- **In-app remove (`PetLoader.deletePet`).** Resolves a pack by manifest `id` (not directory name) and deletes it, behind a confirm dialog. If the deleted pet was the selected avatar, the selection falls back to the Sphere.
|
||||
- **Live refresh (`ConnectionViewModel`, `RelayApp`).** A `avatarsRefreshTick` StateFlow keys the avatar `produceState`, so import/delete — and opening the Appearance screen — re-scan `pets/` and update every surface (chat, clean mode, voice, splash) without an app restart. This also resolves the standing "pet load is process-scoped" TODO. Add/remove results surface as snackbars via a one-shot `avatarEvents` flow.
|
||||
- **Appearance UI (`AppearanceSettingsScreen`).** Added an "Add a pet" button + "Rescan", an "Installed pets" management list with per-pet remove, and a remove-confirm dialog; replaced the static "drop a pack into pets/ via adb" hint with the in-app flow.
|
||||
- **Tests.** `PetImporterTest` (root + nested import, no-manifest, missing-idle, **zip-slip refused writes nothing outside staging**); `PetLoaderTest` delete cases (by id, id≠dirname, no-match). Both pass under `:app:testSideloadDebugUnitTest` (the 12 build failures are the pre-existing DataStore/`FileStorage.kt:114` JVM cases — `BargeIn`/`ProfileSelection`/`ProfileSession`Store).
|
||||
- **Verification.** `:app:assembleSideloadDebug` built `hermes-relay-1.1.0-sideload-debug.apk`; installed to the Samsung sideload build via `adb install -r` (Success). A `lucy` test pet (9 sprite-sheet states, 256×256 RGBA with real transparency, schema-validated) staged at `/sdcard/Download/lucy.zip` for an import smoke test (Add a pet → pick from Downloads). On-device import/delete smoke recorded in TODO.md.
|
||||
|
||||
## 2026-06-20 — Pet AI authoring kit + JSON schema (docs)
|
||||
|
||||
**Why.** Pets are pure data, so the only real barrier to making one is sourcing the art. Documented an AI-generation workflow and a machine-readable contract so both humans and AI agents can author and validate a pet without hand-drawing.
|
||||
|
||||
- **AI prompt kit (`user-docs/features/custom-avatars.md`).** A reference-image-first, character-agnostic prompt template (`{character}`/`{style}`/`{accent}`), a per-state motion table mapping image generation to our state vocabulary, a full 9-state manifest, transparency/consistency caveats, and a "fastest first pass" (one 3×3 sheet → nine stills). Mirrored as a one-click `user-docs/public/pet-prompt-kit.txt`. A vendor-neutral "let an AI agent build the pack" callout names Codex/Claude Code as examples and states the acceptance criteria + image-gen prerequisite.
|
||||
- **JSON Schema (`user-docs/public/pet.schema.json`).** Draft-07 schema mirroring the loader's structural rules (required `idle`, frames-XOR-sheet via `anyOf`, positive sheet dims, `schemaVersion` ≤ 1); `$schema` wired into the manifest examples and tolerated by the lenient loader (`ignoreUnknownKeys`). `docs/pet-spec.md` gained an "Editor validation" section, honest that file-existence/decodability remain load-time checks. Validated: legal draft-07 + accepts good / rejects no-idle, empty-clip, schemaVersion-2, sheet-missing-dims.
|
||||
|
||||
## 2026-06-20 — Relay enhancement layer + agent-context injection
|
||||
|
||||
**Why.** Teaching the agent to mark sensitive media (and, more generally, to know things only the relay can teach it) needs a way to inject context into the agent's system prompt — but hermes-agent exposes no plugin context hook (`system_prompt_block()` is memory-provider-only; lifecycle hooks are observers). The one transport-agnostic seam is `AIAgent._build_system_prompt`. Rather than a one-off patch, we built a reusable, removable **enhancement layer** so the relay can apply such patches cleanly and retire them per-surface as upstream catches up — the same pattern as the bootstrap route shims.
|
||||
|
||||
- **`plugin/enhancements/` (registry + contract).** Each enhancement declares `name · phase · enabled() · apply() · retirement note`. Config-gated, **default ON for relay installs** (the relay install is the opt-in; `RELAY_AGENT_CONTEXT_ENABLED=0` opts out), no-op on vanilla, removable with the plugin.
|
||||
- **`context_injection` enhancement (fail-open).** Wraps `AIAgent._build_system_prompt` at plugin-load; appends auditable fenced blocks (`<!-- hermes-relay:<name> -->`). Fail-open at every step — seam absent / block build throws / setattr fails ⇒ returns the base prompt unchanged. With `RELAY_AGENT_CONTEXT_ENABLED` off, the prompt is byte-for-byte unchanged. Works on BOTH gateway and SSE (agent core).
|
||||
- **First block: media-sensitivity.** Teaches the agent to mark private/NSFW media with the client's spoiler convention (`||||` / sentinel alt) — the bit the client already blurs. No soul/memory touched.
|
||||
- **`GET /context/injected` audit route + client audit.** The relay exposes exactly what it would inject; the chat "What the agent sees" sheet gained a "Relay context (server-side)" section. Server-side injection is never hidden.
|
||||
- **Sensitivity re-thread (client).** `ServerImageResult.Success` carries the fetched `sensitive` bit again; `RelayServerImage` blur ORs it with the markdown-parsed flag.
|
||||
- **Transport-path UI.** New `ChatTransportStatusBadge` + `RelayStatusStrip` surface the ACTUAL chat tier (⚡ Gateway / 📡 Sessions / Completions / Runs / offline) instead of a bare "api online"; Chat Settings gained a basic→best tier ladder + a gateway sign-in callout.
|
||||
- **Dashboard.** Relay management tab gained Agent-context master + per-block toggles (off by default; labeled experimental/server-side/removable).
|
||||
- **Verification.** Plugin: `python -m unittest plugin.tests.test_enhancements` (16 pass). Android: `:app:lintSideloadDebug :app:assembleSideloadDebug --no-daemon` green. Built by a 2-worker Orca orchestration (server + client slices), coordinator-integrated. Design: `docs/plans/2026-06-20-relay-enhancement-layer.md`.
|
||||
- **Follow-ups (TODO).** Confirm the `AIAgent` module on the live host when enabling; structured media channel (`docs/plans/2026-06-20-structured-media-channel.md`); incremental bootstrap migration into the enhancement layer; retire the wrap when upstream ships a context hook.
|
||||
|
||||
## 2026-06-20 — Pet intensity modulation (Android)
|
||||
|
||||
**Why.** The last continuous-reactivity gap: a pet's clip looped at a fixed rate regardless of how hard the agent was working. `intensity` (the activity ramp already fed to every avatar — ~0.7 while streaming) was plumbed to `PetAvatar.Render` but ignored. Wiring it completes the reactivity story (voice ✓ · tools ✓ · activity ✓) and un-clamps the last reserved badge flag — and unlike the deferred `attention`, the signal needed no host plumbing.
|
||||
|
||||
- **Live playback-rate modulation (`PetAvatar.Render`).** Opt-in via `reactive.intensity`. The active base/working loop's fps is scaled by `1 + intensity·PET_INTENSITY_RATE` (0.6 → ~1.4× at typical streaming, 1.6× peak, capped at `PET_MAX_FPS`), so it visibly "works harder" as output streams. Read **live** inside the frame loop via `rememberUpdatedState(state.intensity)` so the speed tracks the agent mid-clip without restarting the long-lived loop (re-keying on a continuously-animated float would thrash). One-shot reactions are excluded (`!playOnce`) so `greet`/`done` play at their authored rate.
|
||||
- **Badge un-clamp (`PET_RENDERER_CAPABILITIES.intensity` → true).** The loader's existing `reactive.intensity && capability` formula now lets a declared `intensity:true` through, so the pet advertises **Activity** honestly. No loader change needed beyond the flag.
|
||||
- **Tests.** `PetLoaderTest`: a declared `intensity:true` is now honored (`Voice · Activity`); the prior clamp test was split — `tools` without a `working` clip still stays off the badge.
|
||||
- **Docs (`docs/pet-spec.md`).** Reactivity table's `intensity` row rewritten from "Reserved" to the speedup behavior; removed from "Forthcoming" (now only `attention` remains there). Reactivity is now Voice · Tools · Activity complete.
|
||||
- **Verification.** Code + loader tests authored to the established patterns; not run here (Studio-side). On-device check (a writing/working loop quickening while streaming) recorded in TODO.md.
|
||||
|
||||
## 2026-06-20 — Pet one-shot reaction layer (Android)
|
||||
|
||||
**Why.** The behavior model's event tier: transient "reactions" that play once over the base loop, then return — the touch that turns a status display into a character (cf. the Peon Pet's celebrate-on-finish). Distinct from the sustained per-state loops and the `working` overlay.
|
||||
|
||||
- **Pet-local triggers, zero host plumbing (`PetAvatar`).** One-shots are derived from the activity-state transitions the avatar already observes each frame — no new `AvatarRenderState` edge from the host. `PetOneShot.Greet` fires on first composition (the pet appears); `PetOneShot.Done` fires when a *productive* turn ends (a `Streaming`/`Speaking` → `Idle` transition; `Thinking → Idle` and `Error → Idle` don't celebrate). Both are opt-in (only if the pet ships the clip) and require ≥2 frames.
|
||||
- **Play-once-then-revert (`PetAvatar.Render`).** The frame loop gained a `playOnce` mode: a reaction clip plays 0→end (no modulo wrap), parks on its last frame, clears `activeOneShot`, and recomposition hands back to the base loop. A live reaction overlays everything (including `working`). Suppressed under reduced motion (`paused`). An `ONE_SHOT_MAX_MS` (4s) backstop guarantees a reaction never lingers on decode failure / single frame / pause.
|
||||
- **Friendly aliases (`PetLoader.toAvatar`).** Resolves `greet`/`wake` → `PetOneShot.Greet` and `done`/`celebrate` → `PetOneShot.Done` from explicit `states` keys only (no fallback); absent reactions just don't play. One-shots are reactions, **not** a reactivity signal, so they don't touch the picker badge.
|
||||
- **Tests.** `PetLoaderTest`: a pack with `greet`/`done` keys loads cleanly and the badge stays `Voice` (no accidental Tools/Activity coupling). Render-time playback (the actual one-shot animation) is on-device/Compose-test territory — flagged in TODO.
|
||||
- **Docs (`docs/pet-spec.md`).** New "One-shot reactions" section (Greet/Done table, opt-in, play-once, reduced-motion), an Expressive tier on the authoring ladder, and the Loop-vs-one-shot note updated. `attention`-on-notification stays in "Forthcoming" — it needs a host event the avatar doesn't receive yet.
|
||||
- **Verification.** Code + loader test authored to the established patterns; not run here (Studio-side). On-device checks (greet on appear, celebrate on turn-finish, overlay-over-working) recorded in TODO.md.
|
||||
|
||||
## 2026-06-20 — Pet `working`/tool-use behavior (Android)
|
||||
|
||||
**Why.** The behavior-model spec called for a distinct "agent is running a tool" pose — the strongest cross-system convention (Microsoft Agent splits `Think` from `Process`/`Search`; the `pi-animations` indicator splits Thinking · Working · Tool) is that *acting* should look different from *thinking*. Our six `SphereState`s folded tool-use into thinking/streaming.
|
||||
|
||||
- **Pet-local tool overlay (`PetAvatar`).** Implemented as a sub-state derived from the already-plumbed `toolCallBurst`, **not** a 7th `SphereState` — zero blast radius on the Sphere or the call sites. `Render` swaps to an optional `workingClip` when `toolCallBurst ≥ WORKING_BURST_THRESHOLD` (0.5) during a `Thinking`/`Streaming` turn, and returns to the base-state clip as the burst decays (the signal ramps to ~1 in 200ms and decays over 1200ms, so 0.5 activates fast and lingers ~600ms — smoothing back-to-back tool calls). Error keeps its own clip; `toolCallBurst` is ~0 outside tool activity, so it never fires spuriously.
|
||||
- **Opt-in, clip-driven capability (`PetLoader.toAvatar`).** `workingClip` resolves only from an explicit `working` key (no fallback) — a pet without one keeps its base-state clip during tool use, exactly as before. Shipping a usable `working` clip is *itself* the tool-reactivity capability: it drives both the swap and the **Tools** badge (`reactivity.tools = (workingClip != null) && PET_RENDERER_CAPABILITIES.tools`), so the declared `reactive.tools` flag is no longer needed and can't over-promise. Flipped `PET_RENDERER_CAPABILITIES.tools` to `true` (the renderer now consumes the signal).
|
||||
- **Tests.** `PetLoaderTest`: a `working` clip lights the Tools badge (`Voice · Tools`); a `working` clip with missing files does not; the existing declared-but-no-clip case still clamps to `Voice`.
|
||||
- **Docs (`docs/pet-spec.md`).** `working` moved from "Forthcoming" into the implemented model: a `Working` row in the state table, a "The `working` overlay" subsection (opt-in, tool-use vs. thinking), the authoring ladder's Rich tier now 7 clips, and the reactivity table's `tools` row now "driven by the `working` clip." Forthcoming trimmed to one-shot reactions + intensity modulation.
|
||||
- **Verification.** Code + tests authored to the established patterns; not run here (Studio-side). On-device check (clean mode, a `working` clip swapping in during a tool run) recorded in TODO.md.
|
||||
|
||||
## 2026-06-19 — Pet reactivity: honest badge + behavior-model spec (Android)
|
||||
|
||||
**Why.** Follow-up to the custom-avatar audit. The pet picker badge read `reactivity.summary()` straight from `pet.json`, so a pet declaring `reactive:{tools:true,intensity:true}` advertised "Voice · Tools · Activity" while `PetAvatar.Render` only ever consumed voice — the badge could lie. Separately, the goal was to let pets *associate behavior with agent activity* (show "thinking" vs "writing" vs "speaking"), which needed a documented behavior model rather than a fallback table buried in the clip docs.
|
||||
|
||||
- **Honest capability badge (`PetAvatar`, `PetLoader`).** Added `PET_RENDERER_CAPABILITIES` (the live signals `Render` actually consumes today: voice only) and clamp a pet's effective `reactivity` to `declared AND supported` in `toAvatar`. One forward-compat switch: flip a flag there the day the renderer learns a signal and every manifest that already declared it lights up. `PetLoaderTest` gained a case asserting declared tools/intensity are dropped from the badge.
|
||||
- **Friendly `writing` alias (`PetLoader.STATE_CLIP_CHAIN`).** The Streaming (output-producing) state now resolves `writing` → `streaming` → … so authors can target it with the intuitive key; tidied the Speaking/Error chains to fall back through related activity clips before idle. Backward compatible (existing `streaming`/`speaking`/`error` keys still resolve).
|
||||
- **Behavior-model spec (`docs/pet-spec.md`).** New "Agent states & pet behavior" section: what each of the six activity states means, the friendly clip-key vocabulary + fallback chains, a loop-vs-one-shot note, and a Minimal→Basic→Standard→Rich authoring ladder. A "Forthcoming behavior" subsection specifies the designed-not-yet-rendered tier — a distinct `working`/tool-use clip, one-shot reactions (greet/celebrate/attention), and continuous tool/intensity modulation — so authors can plan. Reactivity table updated to state the clamp.
|
||||
- **Prior-art grounding.** Researched the convention (no first-party "Codex pet" exists — the agent-state→mascot pattern is third-party only: `pi-animations`, Peon Pet; canonical spec lineage is Microsoft Agent's `.acs` animation set, with Live2D/VRM/VTuber lip-sync and game-dev FSMs converging on idle-base + thinking/working/output split + amplitude-driven talking + one-shot reactions). The thinking≠tool-use split is the strongest cross-system signal and drives the recommended `working` state. Sources cited in TODO follow-up context.
|
||||
- **Verification.** Code + test authored to the established patterns; not run here (Studio-side). Behavior roadmap (working state, one-shots, intensity modulation) recorded in TODO.md.
|
||||
|
||||
## 2026-06-19 — Custom-avatar audit: storage fix, loader unification, tests, docs (Android)
|
||||
|
||||
**Why.** An audit of the just-shipped swappable agent-avatar / "pet" feature found one blocking bug and a set of clarity/coverage gaps. The only documented way to install a pet (and a sphere skin) was `adb push … /sdcard/Android/data/<pkg>/files/{pets,spheres}/` — i.e. external app-scoped storage — but both loaders read from `context.filesDir` (internal, `/data/data/<pkg>/files/`), which is not `adb push`-able on a non-rooted device. So the documented side-load path could never work, on either flavor. Secondary gaps: no in-app hint that pets exist, no user-docs coverage of avatars/skins at all, and the pure loader logic was Context-coupled and therefore untested.
|
||||
|
||||
- **Shared storage layer (`UserContentDir.kt`, new).** Both `PetLoader.userDir` and `SphereSkinLoader.userDir` now resolve through one helper that prefers external app-scoped storage (`getExternalFilesDir(null)` = `/sdcard/Android/data/<pkg>/files/<name>/`, reachable by `adb push`, no runtime permission on API 19+) and falls back to internal `filesDir` only when external is unmounted. Single source of truth for "where side-loaded customization content lives" — fixes the bug once for pets and sphere skins together. The `adb push` commands the docs already showed are correct against this location.
|
||||
- **Testability refactor (`PetLoader`, `SphereSkinLoader`).** Added pure `loadPets(dir: File)` / `loadUserSkins(dir: File)` overloads (no Android Context); the Context overloads delegate. The validation/resolution/skip-invalid path is now unit-testable against a temp directory, mirroring `DashboardManageDiskCache`'s `dir: File` shape.
|
||||
- **Tests (`PetLoaderTest`, `SphereSkinLoaderTest`, new).** 17 + 6 JUnit cases covering parse, id/label fallbacks, schema-version + missing-`idle` + missing-file rejection, the `safeChild` **path-traversal guard** (a real `../escape.png` outside the pack is refused), fps clamping, one-bad-pack-doesn't-break-the-rest, sort order, and empty/absent dirs. No real bitmaps needed (the loader only checks `isFile`); `unitTests.isReturnDefaultValues=true` makes `Log.w` a no-op so no `mockkStatic`.
|
||||
- **In-app discoverability (`AppearanceSettingsScreen`).** Added an "Add your own pet" pointer line under the Agent-avatar chips (mirrors the existing sphere-skin pointer) so users learn the feature exists even with no pets installed.
|
||||
- **Docs (`docs/pet-spec.md`, `docs/sphere-spec.md`).** Corrected the storage prose (app-scoped external, external-preferred / internal-fallback, both flavor paths), cross-linked the two specs under one "customize the agent avatar" framing, fixed a "Agent sphere" → "Agent avatar → Sphere skin" naming drift, and added two authoring caveats: a present-but-undecodable image renders blank (not caught at load), and frame-sequence pets decode every frame at full resolution into RAM (keep frames small / prefer sprite sheets).
|
||||
- **User docs (`user-docs/features/custom-avatars.md`, new + nav).** New user-facing page covering the two-level avatar→skin model, the reactivity badges, how to add skins and pets, reduced-motion behavior, and troubleshooting; wired into the Features sidebar.
|
||||
- **Verification.** Tests authored to the established temp-dir pattern and validated against the actual `toAvatar`/`toSkin` contracts (read from source); not run here (Android build/test is Studio-side). Follow-ups (per-frame memory cap/downsample, decoded-clip cache to kill re-decode churn) recorded in TODO.md.
|
||||
|
||||
## 2026-06-18 — Chat: mid-session model switch, error surfacing, model-scope UI (Android)
|
||||
|
||||
**Why.** On-device testing surfaced four linked issues. (1) Switching the in-chat model in an *existing* conversation showed the pick but the turn still ran the old model. (2) A failed turn (grok-4.3 hitting xAI's 200-tool cap) flashed an error bubble then vanished. (3) The chat header and agent drawer showed the global/profile model, not the session's live model — the drawer even paired the global model *name* with the session *provider* (`gpt-5.5 · xAI Grok`). (4) An upstream-injected `[System: the active model changed …]` marker rendered as a chat bubble.
|
||||
|
||||
- **Session-scoped model switch (`GatewayChatClient.prewarmAwait`, `ChatViewModel.selectModel`).** `selectModel` called fire-and-forget `prewarm()` then immediately `setModel()`, so `config.set {key:"model"}` ran with `liveSessionId == null` and upstream applied it as a GLOBAL write — never touching the live session (verified: `_apply_model_switch`, `tui_gateway/server.py:2134`, is the session-scoped in-place swap the CLI/TUI `/model` uses). Added a suspending `prewarmAwait()` that resolves/resumes the live session before returning; `selectModel` awaits it then applies `setModel` session-scoped, or skips the global write and defers to the next `session.create` override when there's genuinely no session. Confirmed on-device: `sessionScoped=true` → `session.info` flips → turn runs the picked model.
|
||||
- **Errors never swallowed (`GatewayChatClient.dispatchOn`, `ChatHandler.loadMessageHistory`).** Root cause: `dispatchOn` (marshals turn callbacks to the main thread) omitted `onStatusUpdate`, so it fell back to the data class's default no-op — the server's `❌` terminal-error lifecycle line never reached `markError`, the turn wasn't badged `Error`, and `onComplete`'s post-turn history reload (which a non-errored turn runs) wiped the client-only error bubble. Wired `onStatusUpdate` through `dispatchOn` (also restores live status lines, previously dead on the gateway), and hardened `loadMessageHistory` to re-inject local `Error`-badged assistant messages the server transcript lacks, so no reload path can swallow a failure.
|
||||
- **Model display scoped to the session (`ChatScreen` header, `ConnectionInfoSheet.AgentSheetHeader`).** The header subtitle resolved `profile.model ?? serverModelName`; now mirrors the input chip (`selectedModelOverride ?? gatewayCurrentModel ?? profile ?? server`). The agent-sheet header was pairing the global model name with the session provider; it now takes a `sessionModelName` so model+provider come from one scope, and adds a quiet "Server default: …" caption only when the session diverges (the always-visible global-vs-session split; the redundant in-section split was removed).
|
||||
- **System steering markers hidden (`ChatHandler`, `ConnectionViewModel`, `RelayApp`, `ChatSettingsScreen`).** Upstream injects `[System: …]` model/personality-change markers into history for the LLM (`tui_gateway/server.py:1769`); we rendered them as bubbles. `loadMessageHistory` now drops `role:system` `[System:`-prefixed rows by default (desktop/TUI parity), gated by a new `ChatHandler.showSystemMarkers` flag wired from a default-off "Show system messages" debug toggle in Chat Settings (DataStore-backed, mirrors `parseToolAnnotations`).
|
||||
- **Verification.** `:app:assembleSideloadDebug` BUILD SUCCESSFUL; deployed to device; each fix confirmed on-device via filtered logcat traces. Separate known item (not an app fix): xAI/grok models exceed the provider's 200-tool cap with the full relay toolset (server-side).
|
||||
|
||||
## 2026-06-18 — Chat UX: model-picker apply, relay inbound images, smooth profile switch (Android)
|
||||
|
||||
**Why.** An audit of profile switching and the chat composer surfaced three issues: (1) the in-chat model picker showed the picked model but the agent ran on the account's global default; (2) an agent-returned server-local image showed a path/"on server" notice instead of rendering, even when paired to the relay; (3) switching profiles visibly tore down and rehydrated the conversation.
|
||||
|
||||
- **Model picker binds to `session.create` (`GatewayChatClient`, `ChatViewModel`, `GatewayModels`).** Verified against upstream `tui_gateway/server.py`: a model is a per-session override, applied via `config.set {session_id,…}` on a live session or `model`/`provider` params on `session.create` for a fresh one. The app only did the first; on a brand-new chat the `config.set` carried no `session_id` (upstream treats it as a no-op) and `session.create` omitted the model, so the agent built from the global default. Added a live `sessionModelProvider` (mirrors `sessionProfileProvider`) so the picker's model+provider bind onto each `session.create`; mid-session switches still go through `config.set`. A profile switch now retires an explicit pick (the profile defines its own model) and seeds the picker pill from the profile model so the header doesn't lag the round-trip. SSE paths already carried the model in the request body. Tests added for the new binding.
|
||||
- **Relay-backed inbound images (`ChatImageContent`, `ChatScreen`, `ChatViewModel`).** Markdown images (``) flowed through a renderer that only understood `http(s)` → Coil; a server-local path fell to a static "image is on the server" notice and never consulted the relay (the relay media route was only wired to the `MEDIA:` marker path). Added a `RelayServerImageResolver` CompositionLocal, provided by ChatScreen from `ChatViewModel.resolveServerImage`, which fetches an absolute server path through the relay's bearer-auth `/media/by-path` route (same route + sandbox the `MEDIA:` path uses), decodes, caches (bounded LRU keyed by path), and renders inline with tap-to-zoom. Null when unpaired → unchanged standard (no-plugin) behavior. Complementary nudge: `composeInjectedContext` appends a one-line media-capability hint to the SSE `system_message` when a relay route is configured (`RelayHttpClient.mediaUrlConfigured()`), surfaced in the "What the agent sees" audit sheet. SSE-only — the gateway has no per-turn system slot, so there the client render fallback (and upstream's own `MEDIA:` instruction) carry it.
|
||||
- **Profile-switch transition (`ChatViewModel.switchProfileContext`, `ChatScreen`).** Stopped clearing the message list synchronously before the async history fetch; the previous transcript is held and swapped atomically when the new history resolves, so the `LazyColumn`'s per-item `animateItem()` cross-fades old→new instead of blanking to an empty/"Loading…" state. The top loading row is suppressed while held content is on screen.
|
||||
- **Verification.** Rebased `Codename-11/fix-ui-ux-issues` onto `dev` first (its only unique change was already on `dev`). `:app:compileSideloadDebugKotlin` + `:app:compileSideloadDebugUnitTestKotlin` BUILD SUCCESSFUL (no new warnings in the changed files); `GatewayChatClientTest` extended with model-binding cases. On-device verification via Studio.
|
||||
|
||||
## 2026-06-18 — Cold-start keystore contention + honest loading states (Android)
|
||||
|
||||
**Why.** A cold-start logcat trace showed the chat header's identity/model/approvals lagging seconds behind first frame. The cause was on-device, not the network: `EncryptedSharedPreferences.create()` decrypts a Tink keyset via a KeyStore op (~0.6–1 s on StrongBox) and Tink serializes those process-globally, and the app was building **three** keysets at startup (a throwaway legacy-sentinel `AuthManager`, the active connection's token store, and the dashboard cookie store) — they thrashed the lock (`Long monitor contention … AndroidKeysetManager.build()`, `waiters` up to 4; a `by lazy` held for **2.369 s**). The relay auth round-trip itself was ~150 ms. Separately, a design constraint surfaced: never display unconfirmed server state (model/provider/approvals) as if confirmed — show an honest loading state and make the load fast, don't cache a maybe-wrong value.
|
||||
|
||||
- **Process-global store cache + raw-build factory (`SessionTokenStore.kt`).** `SecureStoreCache.getOrBuild(prefsName){…}` (a synchronous `ConcurrentHashMap.computeIfAbsent`) builds each prefs file's keyset once process-wide, and `buildRawTokenStore()` is the shared backend factory. Synchronous on purpose so the SAME instance serves both the suspend token path (wrapped in IO) and the synchronous OkHttp cookie-jar path.
|
||||
- **Sentinel deferral (`AuthManager.kt` + `ConnectionViewModel.kt`).** Added `eagerHydrate` (false for the legacy sentinel that `ConnectionViewModel` builds at field-init and replaces the moment the active connection hydrates), so it no longer decrypts a keyset just to be discarded. Re-gated the pre-StrongBox `hermes_companion_auth → _hw` migration on the **file name** (not the sentinel's connection id) so deferral stays correct, and marker-gated it (`legacy_migrated`) so the legacy file is read at most once ever.
|
||||
- **Cookie keyset unification (`DashboardApiClient.kt`, `UpstreamTransportController.kt`, `ConnectionViewModel.kt`, `DataManager.kt`).** The dashboard cookie store now rides the connection's **token** file (threaded `tokenStoreKey` via a `tokenStoreKeyProvider`) instead of its own `hermes_dashboard_<id>` keyset — eliminating the second build. One-shot, marker-gated migration (`dashboard_cookies_migrated`) copies existing cookies across on first access; failure just means a one-time Manage re-login (cookies are re-obtainable, unlike the relay token). Also fixed a double `store.load()` per cookie request.
|
||||
- **Honest loading + fade-ins (`LoadedFadeIn.kt` + 5 call sites).** New shared `LoadedFadeIn`/`RelaySkeletonLine` mirroring the header's skeleton→identity spec. Wired into the header subtitle (model fades in when confirmed), the agent sheet's model line, `ContextMeterBar` (fade+expand), the session drawer (loading→list crossfade), and Manage (Loading→Loaded crossfade). The agent sheet's YOLO switch now shows "Checking…" instead of rendering the unknown (null) state as a definitive "off". No model/provider/approvals value is ever cached and shown as confirmed.
|
||||
- **Also (header/chrome).** Dropped the redundant LAN/Tailscale endpoint chip from the chat top bar (the footer status strip already shows `<status> / <route>`) and made that footer strip tappable → Connections; subtitle no longer renders a `none` personality.
|
||||
- **Verification.** `:app:assembleSideloadDebug` BUILD SUCCESSFUL; deployed to device. Cold-start logcat before→after (warm launch, same device): **3 keyset builds → 1**; the sentinel's "no stored session_token → Unpaired" build is gone; first-frame→Paired **~2.9 s → ~0.95 s**; steady-state (post-migration) trace shows **zero** `Long monitor contention` events. On-device check still wanted: Manage/voice remain signed in after the one-time cookie migration.
|
||||
|
||||
## 2026-06-18 — Connection toast stepper + chat header de-clutter (Android)
|
||||
|
||||
**Why.** Two adjacent UI/UX gaps. The floating connection-status toast already slid in from the top and supported swipe-up dismiss, but it rendered its trace entries as flat `label: detail` text and the swipe silently accumulated to a threshold then snapped — while the cold-start sphere right next to it had a far nicer live stepper (`·`/spinner/`✓`/`✕`). The two were built separately and never unified. Separately, the chat header subtitle (`personality · model · ⚡ approvals off`) was being clipped because the trailing actions row (endpoint chip + Share + Terminal + Settings) won the width fight; the appended approvals text made it worse.
|
||||
|
||||
- **Stepper state model (`RelayUiState.kt`).** Added `ConnectionStepState { Pending, Active, Done, Failed }` and an optional `state` on `ConnectionHandoffTraceEntry` (defaulted null so every producer keeps compiling). Null means "infer from position + snapshot"; producers that know a surface's verdict stamp it explicitly.
|
||||
- **Probe entries stamp real states (`ConnectionViewModel.kt`).** `buildGlobalConnectionProbeEntries` now tags Route=Done, and API/Relay as Active (Probing) / Done (Reachable) / Failed (Unreachable), plus the relay-socket "Session" step as Active.
|
||||
- **Toast redesign (`ConnectionHandoffBanner.kt`).** `ConnectionStatusToast` renders entries as a live stepper (fixed-width monospace glyph + spinner via `LaunchedEffect`, mirroring the splash's `StartupCheckRow` vocabulary), reusing `ConnectionStatusBadge`'s green for Done and `colorScheme.error` for Failed. Swipe-dismiss now tracks the finger with an `Animatable` offset + fade, flinging off-screen past threshold (then firing `onDismiss`) or springing back; keyed on status identity (title+tone) so frequent `updatedAtMs` trace bumps don't reset an in-flight swipe. Warning/Error poses get a bottom divider + "Open <destination> →" link for discoverability. The legacy edge-variant `ConnectionStatusBanner` was left as-is (only referenced within its own file).
|
||||
- **Chat header (`ChatScreen.kt`).** Dropped the inline ` · ⚡ approvals off` annotation from the subtitle (now a plain single line). Added an amber ⚡ `Icons.Filled.Bolt` to the app-bar actions, shown only when approvals are effectively off, tapping into the agent sheet where the full explanation already lives. Share moved into a `⋮` `DropdownMenu` (rendered only when there's a conversation to share), leaving Terminal + Settings + the endpoint chip as the visible actions. `RelayChromeIconButton` gained optional `tint` / `borderColor` params for the amber treatment.
|
||||
- **Verification.** Pending — Kotlin-only UI changes; per the project dev loop these build via Android Studio's run button (not `gradle build` from here). `./gradlew lint` recommended before push.
|
||||
|
||||
## 2026-06-18 — Terminal: TUI input correctness + chrome cleanup (Android + relay)
|
||||
|
||||
**Why.** On-device terminal use surfaced input bugs and wasted chrome, benchmarked against Orca's mobile terminal. The extra-keys bar clipped labels ("CTRL" → "CTR") because it was weight-distributed across a fixed width; the on-screen arrows and PASTE bypassed the emulator and sent fixed/raw bytes (wrong inside TUIs and unsafe for multi-line paste); and the header + tab strip + a stray inset ate vertical space, especially in the common single-tab case. Separately, the relay wrapped each PTY in the user's default tmux, inheriting tmux's 500ms `escape-time` and `screen` `$TERM` — the classic source of laggy ESC, mangled Alt, and degraded color in vim/htop.
|
||||
|
||||
- **Extra-keys bar (`ExtraKeysToolbar.kt`).** Rewrote from a weight-divided `Row` to `horizontalScroll` with fixed-min-width keys, so labels never clip and the cluster scrolls when wider than the screen (Orca's strategy). Then compacted to Orca's proportions — 32dp height, 36dp min width, 12sp, tighter padding/spacing — and centralized the key haptic.
|
||||
- **Mode-aware special keys (`index.html` + `TerminalScreen.kt`).** Added `window.termSendKey(name)` that reads xterm's `applicationCursorKeysMode` and encodes arrows/Home/End as SS3 (`\eOA`) vs CSI (`\e[A`); the toolbar arrows now route through it. PASTE routes through `term.paste()` (bracketed paste) instead of a raw `sendInput`, so multi-line paste no longer auto-executes.
|
||||
- **Compact header (`TerminalScreen.kt`).** Replaced Material's fixed 64dp `TopAppBar` with a ~52dp custom `Row` + `statusBarsPadding`: status is shown once, inline with the title (a `ConnectionStatusBadge` dot + one concise word, ellipsized via `weight(1f, fill = false)` so it can't push the actions off-screen), the whole block opens the info sheet. The subtitle no longer renders the full `hermes-<deviceId>-tabN` wire id (it was wrapping to two lines).
|
||||
- **Less wasted vertical space.** The tab strip + its divider now render only for 2+ tabs; with one tab the new-tab "+" lives in the header instead. Dropped a redundant `navigationBarsPadding()` on the keys bar (the app Scaffold's bottomBar already owns the nav-bar inset, leaving a dead gap), and added an 8px bottom gap in `#terminal` so the last row clears the keys bar.
|
||||
- **Isolated, TUI-tuned tmux (`plugin/relay/channels/terminal.py`).** Sessions now spawn on a dedicated `-L hermes-relay` socket with a generated `~/.hermes/hermes-relay-tmux.conf` (written lazily, best-effort): `escape-time 0`, `default-terminal "tmux-256color"`, truecolor `terminal-features/overrides`, `mouse on`, `focus-events on`, `set-clipboard on`, `aggressive-resize on`, `status off`. A dedicated socket is the only safe way to set server-global `escape-time` without altering the user's own tmux; persistence is unchanged (the socket's server outlives the relay process).
|
||||
- **Verification.** `:app:assembleSideloadDebug` BUILD SUCCESSFUL and deployed to device; `python -m unittest plugin.tests.test_terminal_channel` (4 tests) + `py_compile` pass. Relay change hand-deployed to the staging box and verified live on the `hermes-relay` socket (`escape-time 0`, `default-terminal tmux-256color`, `status off`, `mouse on`, `focus-events on`). tmux 3.4 with `tmux-256color` terminfo present.
|
||||
|
||||
## 2026-06-18 — Native secure routes: split connection Features from Routes (Android)
|
||||
|
||||
**Why.** Connection setup conflated two separate questions — what a Hermes connection can *do* (features) and how this phone *reaches* it (route) — which coupled Relay features to a single transport. Modeling them separately lets a user enable Relay tools over any route (LAN, Tailscale, public HTTPS, VPN, or a plugin-provided secure proxy) and sets up a plugin-assisted native encrypted route that does not require Tailscale. The standard path stays direct-to-upstream and plugin-free. (Backfilled log entry — the work landed in PR #88; full design in `docs/plans/2026-06-18-native-secure-routes.md`.)
|
||||
|
||||
- **Split connection model.** `ConnectionsSettingsScreen` / `ActiveConnectionSections` now render distinct **Features** and **Route** sections; `Endpoint.kt` + `ConnectionData.kt` carry the route/role model and `QrPairingScanner` threads it through pairing.
|
||||
- **Plugin secure proxy route.** A `plugin_proxy` route role surfaces as a "Secure proxy" option with encrypted / pinned-TLS treatment (recommended, not forced) alongside the existing LAN / Tailscale / public / custom roles.
|
||||
- **Docs.** Added the `2026-06-18-native-secure-routes` plan and a connections split-model mockup; fixed the docs-site hero sphere to keep its canvas backing store synced to the CSS box (`HeroDemo.vue`).
|
||||
- **Verification.** CI green on PR #88 (Android Build + Lint + Test).
|
||||
|
||||
## 2026-06-18 — Android onboarding permissions review surface
|
||||
|
||||
**Why.** Android onboarding already kept the standard path clean, but permissions were scattered between feature-specific prompts, Bridge, and Android Settings. A central review page makes the model explicit: standard Chat and Manage do not need phone-control permissions, while voice, camera, notifications, and sideload Device Control remain opt-in.
|
||||
|
||||
- **Shared permission snapshot.** Added `AppPermissionStatusProbe` so Bridge and Settings read the same runtime grants and special-access switches: notifications, microphone, camera, notification listener, accessibility, screen capture, overlay, contacts, SMS, phone, and location.
|
||||
- **Permissions screen.** Added `PermissionsStatusScreen` with Standard Hermes, On demand, and flavor-aware Device Control sections. Rows show required/optional/session status and link to the relevant Android Settings surface or Bridge session grant.
|
||||
- **Onboarding and Settings entry points.** The Power Tools onboarding page now has a "Review permissions" action, and Settings -> App includes a Permissions row. Google Play builds show the sideload Device Control section as unavailable rather than implying hidden phone-control permissions.
|
||||
- **Verification.** `:app:compileGooglePlayDebugKotlin`, `:app:compileGooglePlayDebugAndroidTestKotlin`, and `:app:compileSideloadDebugKotlin` pass with `ANDROID_HOME` pointed at the local SDK.
|
||||
|
||||
## 2026-06-17 — Chat transparency + provenance polish: injected-context audit sheet, spoken-turn badges, version-skew error
|
||||
|
||||
**Why.** On-device voice testing surfaced two transparency gaps and two papercuts. The per-turn system context the phone injects (persona + phone status + voice hint) was invisible — no way to audit what the agent actually receives. Spoken voice-mode turns were indistinguishable from typed ones in the scrollback, while realtime turns were already badged. A field an older relay plugin doesn't accept produced a misleading "Network error · HTTP 400" with a dead Retry. And the floating connection toast's Warning tone was semi-transparent, letting content bleed through.
|
||||
|
||||
- **Injected-context audit sheet.** `ChatViewModel.composeInjectedContext()` extracts the per-turn system-message composition (persona/profile precedence + phone-status block + per-turn interface context) into one builder used by both `startStream` (which sends `combinedSystemMessage`) and the new `previewInjectedContext()` — so the audit can't drift from what is sent. `combinedSystemMessage` is built byte-for-byte as before (`listOfNotNull` over the raw blocks); per-block fields null out blanks only for display, and the resolved profile is passed in to preserve the no-skew invariant with `modelOverride`. Tapping the `ContextMeterBar` (now with an ⓘ affordance) opens `InjectedContextSheet` ("What the agent sees"); on the gateway path the persona block is labeled "added server-side — not sent from this device", since the server owns the soul + personality overlay there.
|
||||
- **Spoken-turn badges.** Voice-mode replies are tagged "Voice", realtime replies keep "Realtime Agent"; both share a speaker glyph as the modality marker (`MessagePathBadge` gained an optional leading icon). The Voice tag is set on the assistant placeholder from the per-turn interface-context signal and rides the id-swap + content updates.
|
||||
- **Badges survive history reload.** `ChatHandler.loadMessageHistory` now carries provenance badges forward by message id when it rebuilds the list from server data — the post-turn reload previously wiped them (this also fixes the pre-existing loss of "Stopped"/"Error").
|
||||
- **Version-skew error.** `RelayErrorClassifier` maps a 400 whose body names an unsupported *field* to "Relay update needed" (non-retryable), distinct from a bad *value* like "unsupported codec" (which keeps its normal classification).
|
||||
- **Connection toast opacity.** `ConnectionStatusToast` composites its container color over the theme `surface` so the floating overlay is always opaque while keeping each tone's tint; the in-flow `ConnectionStatusBanner` is intentionally left translucent (it blends with a known backdrop).
|
||||
- **Verification.** `:app:compileSideloadDebugKotlin` BUILD SUCCESSFUL; `./gradlew lint` clean. On-device confirm via Studio/sideload.
|
||||
|
||||
## 2026-06-17 — App theming: theme-aware brand tokens, app themes, hot-swappable sphere
|
||||
|
||||
**Why.** Chat (and other brand-styled surfaces) were effectively hardcoded dark: the `RelayRefresh` brand palette was a single dark-only `object` of `val Color(...)` constants that ~150 call sites referenced directly, bypassing the Material light scheme. The goal was three-fold: fix that alignment, add real app themes (light/dark plus the Nous Hermes baselines), and make the agent sphere a hot-swappable component with user-authored skins.
|
||||
|
||||
- **Theme-aware brand tokens (the fix).** New `ui/theme/BrandPalette.kt` defines a 21-token `BrandPalette`, the `LocalBrand` CompositionLocal, a `toColorScheme()` derivation (Material scheme is now derived from the palette, never authored separately), and the `AppTheme`/`AppThemes` registry. `RelayRefresh` was converted from constants into a **snapshot-backed façade** over an `activePalette`: every `RelayRefresh.X` is now a getter reading a `mutableStateOf`, so reads in composition and draw phases subscribe and repaint on theme change — the ~150 existing call sites became theme-reactive with no edits. `HermesRelayTheme(appThemeId, themePreference, fontScale)` resolves the theme + light/dark/auto mode into one palette, drives the Material scheme + `LocalBrand`, and mirrors it into the façade via `SideEffect`.
|
||||
- **App themes (8).** Hermes Relay (the brand, with full light + dark) plus ports of the canonical Nous dashboard baselines from upstream `web/src/themes/presets.ts` — Hermes Teal, Nous Blue (light), Midnight, Ember, Mono, Cyberpunk, Rosé. Per the hybrid model, the brand honors Light/Dark/Auto while the character themes are fixed-mode looks (matching how Nous ships themes; light mode lives as the Nous Blue theme). `ConnectionViewModel` gained an `appTheme` id pref (DataStore `app_theme`); the picker is a swatch gallery in `AppearanceSettingsScreen`, and the Light/Dark/Auto control disables + explains itself for fixed-mode themes.
|
||||
- **Flourish alignment.** The glow/gradient/markdown-highlight flourishes across 13 files keyed off `isSystemInDarkTheme()`; they now read `LocalBrand.current.isDark`, so a fixed dark theme keeps its flourishes on a light-mode phone (and vice-versa). `isSystemInDarkTheme()` now lives only in `Theme.kt` (resolving Auto).
|
||||
- **Hot-swappable sphere.** The core algorithm (`MorphingSphereCore.kt`, mirrored in `preview/web/sphere.js`) was left untouched — parity contract preserved. A new `SphereSkin` layer supplies per-state colors/params and declares its reactivity (voice/tools/intensity/gaze), gated inside `MorphingSphere` so reactive inputs are optional + detectable. `SphereRegistry` ships Adaptive (recolors to the active theme via `LocalBrand`), Classic (the original look), Aurora, Solar, and Mono. `MorphingSphere` gained a `skin` param defaulting to `LocalSphereSkin.current`, so all ~11 call sites are untouched; `RelayApp` provides the resolved skin + available set. User skins load from app-private `spheres/*.json` via `SphereSpec` (kotlinx.serialization, data-only, validated, invalid files skipped) → `SphereSkinLoader`; surfaced in the picker with capability badges and a "Custom" tag. Format documented in `docs/sphere-spec.md`.
|
||||
- **Verification.** Reviewed-but-not-compiled — no Android SDK in this worktree and Studio owns builds. Brace/paren balance checked on all new/edited files; import resolution and call-site compatibility reviewed by hand. `./gradlew lint` + on-device confirm pending (via Android Studio). Palettes are single `BrandPalette` literals, easy to tweak after an on-device pass.
|
||||
|
||||
## 2026-06-17 — ConnectionViewModel decomposition: extract transport/pairing/profile collaborators (ADR 34 follow-up)
|
||||
|
||||
**Why.** `ConnectionViewModel` was the one god-object the ADR 34 fence deferred — ~5,531 lines reaching across both sides of the upstream/relay package boundary (the `HermesApiClient`/`GatewayChatClient`/`DashboardApiClient` zoo *and* the relay `ConnectionManager` *and* pairing *and* profiles). Goal: move cohesive concerns into named, testable collaborators in a new `viewmodel/connection/` package, behind the ViewModel's existing public surface, so the standard-vs-relay wiring lives in explicit seams. Pure mechanical, behavior-preserving extraction — the whole-module compile (production + every test source unchanged) is the public-API-preservation guard. Continues the `ConnectionSwitchCoordinator` precedent.
|
||||
|
||||
- **`PairingController`** (229 lines). Owns the paired-devices list (`GET /sessions`) + management (`loadPairedDevices`/`revokeDevice`/`extendDevice`/`revokeChannelGrant`, incl. the optimistic local removal and the full-grants-rebuild encoding) and the insecure-ack DataStore flags (`insecureAckSeen`/`insecureReason`/`setInsecureAckComplete`). Self-contained — nothing else reads its state except `applyPairingPayload`'s `insecureReason.value`. The ViewModel delegates unchanged.
|
||||
- **`UpstreamTransportController`** (269 lines). Owns the per-connection encrypted `DashboardCookieStore` cache, a single consolidated `DashboardApiClient` factory (was 4+ scattered build sites — the plan's headline), the cached `GatewayChatClient` (lazy build + mid-turn LAN/Tailscale retarget) with its availability tier + sticky-Unsupported verdict, and the per-endpoint capability snapshot + `chatMode` + the `streamingEndpoint`-preference resolution. The `@Synchronized` gateway-cache lock moved *with* the state (now the controller instance) — same mutual exclusion, different monitor. `rebuildApiClient` pushes the probed snapshot via `setCapabilitiesAndMode`.
|
||||
- **`ProfileController`** (313 lines). Owns the merged `agentProfiles` list (relay `auth.ok` ∪ dashboard `/api/profiles`), the per-connection selected-profile state machine + its three persistence stores, `profileDisplayAlias`, `activeSessionTransport`, and the per-profile last-session restore. Because the state machine is co-driven by ViewModel-level lifecycle observers (connection switch / active-connection change / agent-profile arrival / gateway-availability settle), those observers stay in the ViewModel and call `profileController.*` lifecycle hooks **in their original order** — orchestration stays put; only state + logic moved, so the state machine is now unit-testable in isolation. The three stores are exposed as public vals so the connection-lifecycle orchestrators (`removeConnection`, duplicate-merge, `resetAppData`, `saveLastSessionId`) keep their clear/persist call sites byte-identical.
|
||||
- **Deferred: `RelayTransportController`** (the plan's Step 2). Left in place per the plan's explicit "too entangled — don't force it" rule. `ConnectionManager` is referenced at ~38 ViewModel sites, including eager `StateFlow` initializers (`relayConnectionState`, `activeEndpoint`, `effectiveApiServerUrl`/`RelayUrl`/`DashboardUrl`, `relayReady`, `insecureMode`), the `ConnectionSwitchCoordinator`, and the `relayHttpClient`/`ScreenCapture`/`tailscaleDetector` URL-provider lambdas. The relay/route methods are inseparable from the central `_relayUrl`/`_apiServerUrl` state (also written by non-relay orchestrators + the switch coordinator) and from shared connection-store helpers (`persistActiveConnectionUrls`, `mergedRouteCandidates`); the route-probe public nested types (`RouteProbeStatus`, `RelayReachable`) are part of the frozen public API. A faithful extraction needs ~18 injected callbacks/refs — relocating coupling into lambdas rather than removing it, against the plan's "named seams, not emergent shared state" goal and at a regression risk the compile-plus-focused-slice verification can't catch. The other three controllers stand on their own; the `ChatTransportProvider` capstone (optional) is likewise not attempted.
|
||||
- **Result.** `ConnectionViewModel` 5,531 → 5,089 lines (−442); cohesive transport/pairing/profile concerns now live in 811 lines of `viewmodel/connection/` collaborators with narrow, provider-injected interfaces. Public API unchanged.
|
||||
- **Verification.** `:app:compileSideloadDebugKotlin` + `:app:compileSideloadDebugUnitTestKotlin` BUILD SUCCESSFUL after each extraction (every caller + test source compiles unchanged → public surface byte-identical). Focused slice `*ArchitectureBoundaryTest` (Konsist fence — `viewmodel/connection/` is outside `network.*` so it imports both worlds freely, fence stays green) + `*RelayUrlDeriverTest` + `*ConnectionSwitchTest` passes. `./gradlew lint` clean. On-device confirm pending (Bailey, via Studio).
|
||||
|
||||
## 2026-06-17 — Voice mode audit: relay bug fixes, spoken-output hint, Google enhanced voice
|
||||
|
||||
**Why.** A post-refactor audit of the standard and relay voice paths surfaced one reachable correctness bug plus several enhancement opportunities: leverage the desktop-style non-persisted per-turn context to instruct spoken-output formatting, and expose Google/Gemini enhanced voice (tone tags, voice/model/persona) that upstream added in recent PRs.
|
||||
|
||||
- **Relay realtime-agent loop drift (correctness).** The non-native ("render-after-Hermes") websocket loop in `plugin/relay/realtime_agent/broker.py` lacked a `playback.drained` branch, so the end-of-turn ack every client sends fell through to the unsupported-message error; the client treats `voice.error` as fatal and tore the session down on every turn whenever a non-native provider was configured. Extracted the client→server messages identical across the native and non-native loops (`session.start`, `session.resume`, `client.ack`, `playback.drained`, `hermes.confirm`) into a shared `_handle_common_client_message` dispatcher used by both loops so they can no longer drift, and added the missing `input_audio.clear` handling to the non-native path. The native loop's per-message `provider_task.done()` break check is preserved exactly. (`hermes.confirm` echo is by-design in both loops — the realtime model answers confirmations via its `hermes_confirm` tool, which returns `forwarded_to_hermes_ui` — so it was left unchanged.)
|
||||
- **TTS file leak.** `plugin/relay/voice.py` synthesize streamed upstream-written `~/voice-memos/*.mp3` files and never deleted them. It now passes its own temp `output_path` into the TTS tool and deletes the artifact after streaming (reads bytes into memory and returns a `web.Response` so cleanup can run; audio is bounded by `MAX_TEXT_CHARS`).
|
||||
- **Realtime "lab" session binding.** `plugin/relay/realtime_voice.py` `handle_ws` authenticated but never checked the websocket caller created the session. Added `_auth_matches_session` (mirrors `voice_output.py`) binding sessions to the creating principal's kind + device-id / token-hash, reusing `voice_auth`'s shared `AuthPrincipal` / `_bearer_from_request`.
|
||||
- **Spoken-output formatting hint (standard + relay).** Enriched the per-turn voice interface context (`VoiceViewModel.STABLE_VOICE_INTERFACE_CONTEXT`) to tell the model its reply will be spoken — short conversational sentences, no markdown/emoji/URLs. This rides the existing non-persisted `system_message` slot (upstream `ephemeral_system_prompt`), so it never lands in history. Because the gateway `prompt.submit` RPC has no system-message slot, `ChatViewModel.startStream` now forces any turn carrying a per-turn interface context (voice) onto an SSE endpoint so the hint always reaches the model.
|
||||
- **Provider-aware enhanced voice (Gemini + xAI) — relay path.** `/voice/synthesize` accepts optional per-request overrides (`voice`, `model`, `audio_tags`, `persona_prompt`, `language`) mapped onto the active provider. Upstream's `text_to_speech_tool` has no per-call override surface, so the relay merges overrides into a config copy and invokes the provider generator directly — `_generate_gemini_tts` (voice/model/`audio_tags` tone-tag rewrite/inline persona via a temp file) or `_generate_xai_tts` (`voice`→`voice_id`, `audio_tags`→`auto_speech_tags`, `language`). Gemini's audio-tag rewrite fails soft when the auxiliary LLM is unavailable. `/voice/config` advertises a provider-aware `tts.enhanced` capability block. Android plumbs `EnhancedVoiceOverrides` from new persisted prefs through `RelayVoiceClient.synthesize` + the relay adapter, with an "Enhanced Voice (<provider>)" Voice Settings card rendered from the capability flags. OpenAI is excluded — upstream exposes only voice/speed for it (no `instructions` tone steering).
|
||||
- **Enhanced voice on the streaming renderer (`/voice/output`).** Because `voice_output_enabled` defaults true, the streaming renderer — not `/voice/synthesize` — is the normal relay playback path, so the per-request override was effectively fallback-only. Added xAI `auto_speech_tags` as a per-profile `voice_output:` setting threaded through every layer (config dataclass + env + YAML loader, `voice_output_settings`, profile override, `VoiceOutputSession`, `_provider_options`, `config_payload`, the `PATCH /voice/output/config` allow-list), mirroring `text_normalization`. The render applies `upstream_voice.apply_xai_speech_tags()` (fail-soft) to each chunk before the `voice_lab` `xai_tts` renderer — keeping `voice_lab` standalone and the upstream import in the patch-point. Android: `VoiceOutputConfig.auto_speech_tags` + `updateVoiceOutputConfig(autoSpeechTags=)` + an "Expressive speech tags" switch in the **Hermes Chat + Voice Output** card (xai_tts, persisted with the existing Save buttons). No Gemini streaming provider in `voice_lab`, so Gemini enhanced voice stays `/voice/synthesize`-only.
|
||||
- **Render-path visibility (troubleshooting).** The streaming-vs-synthesize decision was logcat-only. Added a per-session `DiagnosticsLog` entry naming the active path and a persistent "Render path" row in the Voice Settings card (derived from `voiceOutputConfig`).
|
||||
- **Docs.** `docs/upstream-surface-matrix.md` gained a "Voice Surfaces (standard vs. relay)" section with an explicit **route-ownership table** (every `/voice/*` route is relay-owned; only dashboard `/api/audio/*` is upstream — no upstream streaming/WS audio route) and an enhanced-voice matrix across both relay paths; `docs/spec.md` Phase V documents the override, the `tts.enhanced` block, and `voice_output` `auto_speech_tags`; `user-docs/features/voice.md` gained an "Enhanced Voice (Gemini & xAI)" section + the streaming speech-tags toggle and corrected the stale `~/voice-memos` note.
|
||||
- **Standard voice polish.** Pre-flight 25 MB transcribe guard (matches upstream `_MAX_TRANSCRIPTION_UPLOAD_BYTES`) + friendly 413/400 copy in `StandardHermesVoiceClient`; hardened the dashboard audio HEAD probe to also try `/api/audio/speak`.
|
||||
- **Verification.** Python: `py_compile` on all touched relay modules; relay voice suite green at 103 tests except one pre-existing xAI-OAuth env-dependent failure (identical on the unmodified tree). New tests: non-native `playback.drained` regression (red-on-bug/green-on-fix), Gemini + xAI synthesize-override integration, override-parser + capability-block units, `auto_speech_tags` PATCH round-trip, `apply_xai_speech_tags` call-through/fail-soft. Kotlin not built locally (Studio + `./gradlew lint` are the pre-push gate).
|
||||
|
||||
## 2026-06-17 — Upstream/Relay isolation: package fence + Konsist rule + vanilla contract test
|
||||
|
||||
**Why.** The load-bearing "standard path = vanilla upstream" invariant was enforced only by `CLAUDE.md` convention (network clients were cleanly named but co-located in one package, so nothing *stopped* a standard-path file importing a relay client), and the standard path had never been validated against *true* vanilla upstream (staging runs the fork with relay routes compiled in). ADR 34 records the decision; this lands all three parts. Net-additive, behavior-preserving.
|
||||
|
||||
- **Package fence.** Split `app/.../network/` into `network/{upstream,relay,shared}` — main + mirrored test sources (38 files moved, ~83 touched for import repointing). `VoiceAudioClient.kt` split three ways: the `VoiceAudioClient` interface + `AutoVoiceAudioClient` router → `shared`; `StandardHermesVoiceClient` → `upstream`; `RelayVoiceAudioClientAdapter` → `relay` (co-locating them would force one file to import both worlds). `ChatHandler` → `upstream` (per ADR 3 chat never flows through the relay multiplexer; the handler is fed only by upstream transports). `AndroidManifest` `GatewayKeepAliveService` FQCN and the `ci-android.yml` `RelayUrlDeriverTest` path updated for the move.
|
||||
- **Hidden coupling surfaced.** The move exposed the one real upstream→relay dependency the import grep couldn't see (it was a same-package bare reference): `ChatHandler` renders phone-action bubbles from the bridge's `LocalDispatchResult` DTO. Resolved by moving that passive DTO to `network.shared` — both sides now depend only on shared to speak it.
|
||||
- **Konsist boundary test.** `ArchitectureBoundaryTest` (`scopeFromProduction`) asserts `upstream` ⊥ `relay` and `shared` imports neither. Added to the `ci-android.yml` explicit `--tests` list (the broad aggregate hangs, issue #32, so a named test is the only way it runs in CI). Konsist 0.17.3 resolves clean on Kotlin 2.3.21.
|
||||
- **Vanilla-upstream contract.** `scripts/check-upstream-route-contract.py` source-parses upstream's declared routes (aiohttp `add_*` + FastAPI decorators) — no server boot, no pip, no model keys. Two tiers: REQUIRED standard-path routes fail the build if missing; mode-dependent routes (auth-gate, `/api/pty`, `/v1/models`) only warn. A fork-marker guard refuses to pass against our own fork. `ci-contract.yml` checks out vanilla upstream with no relay bootstrap, asserts the checkout is vanilla, runs the contract; weekly schedule tracks upstream `main` as a drift siren, PR/push use a pinned ref. Notable: in the checked upstream commit the dashboard exposes no `/api/auth/ws-ticket` REST route (it uses the injected session token + a ws `ticket` query param), so the Desktop-style auth-gate routes are advisory, not required.
|
||||
- **Deferred (tracked in ADR 34).** The `ConnectionViewModel` transport-strategy split — the one true god-object leak — is intentionally deferred as the riskiest change; the fence now contains the blast radius. Same-package redundant imports left behind by the move (e.g. a relay file importing its own `network.relay` sibling) are harmless and not cleaned. The contract job's PR-run `UPSTREAM_REF` defaults to `main` until pinned to a confirmed-public known-good SHA.
|
||||
- **Verification.** `:app:compileSideloadDebugKotlin` + `:app:compileSideloadDebugUnitTestKotlin` BUILD SUCCESSFUL; `ArchitectureBoundaryTest` passes; contract script PASS against the local upstream clone (12/12 REQUIRED routes). `./gradlew lint`: BUILD SUCCESSFUL (clean, all four variants).
|
||||
|
||||
## 2026-06-17 — Gateway parity: live session.info sync + YOLO/Fast + stale-state refreshes
|
||||
|
||||
**Why.** An audit (full tui_gateway surface vs. what the official desktop uses vs. what we used) found we were dropping most `session.info` fields and fetching several server lists once. Goal: augment upstream, never show stale state. Verified every contract against the up-to-date upstream clone; a parallel review confirmed the new RPCs match `config.set` exactly and caught three race-window bugs (fixed).
|
||||
|
||||
- **More of `session.info` consumed live.** The interceptor now also surfaces `reasoning_effort`, `credential_warning`, `yolo`, and `fast` (added to `serverReasoningEffort`/`serverCredentialWarning`/`serverYolo`/`serverFast` flows); `startGatewayStateSync` gained one guarded collector each. A `/reasoning` change made on the desktop/TUI now reflects instantly instead of only on turn-complete.
|
||||
- **Credential warnings no longer silent.** `session.info.credential_warning` (present only when the active provider key is missing/invalid) is surfaced once per distinct warning as a ⚠ system notice — dedup'd against the constant `session.info` echoes, cleared when the key is fixed. Previously such turns just failed silently.
|
||||
- **YOLO + Fast mode.** New session-scoped toggles in the agent sheet (`config.set yolo` value `1`/`0` scope `session`; `config.set fast` value `fast`/`normal`) with optimistic set + rollback, live state from `session.info.yolo`/`fast`, and reset across every session/profile/connection switch. YOLO (approval bypass) renders loud — `error` caption + an `errorContainer` "Approvals are OFF" banner — and stays ephemeral so a backgrounded app can't leave global auto-approve armed.
|
||||
- **No fetch-once staleness.** `refreshSkills()` + `refreshModels()` (SSE `/v1/models`) now fire on agent-sheet open alongside the personality/model refreshes, so server-side skill/model changes appear without an app reload.
|
||||
- **Review fixes.** `activateGatewayProfile` now nulls YOLO/Fast (the missing 5th clear site); the `setYolo`/`setFast` optimistic rollback guards against a session switch landing during a slow `prewarm` (re-check client identity + only roll back if we still own the value).
|
||||
- **Verification.** `:app:testSideloadDebugUnitTest` compiles clean; contract-fidelity review = all PASS. Touches only `GatewayChatClient`/`ChatViewModel`/`ConnectionInfoSheet`. The command-palette skills-refresh-on-open is the one optional follow-up (palette lives in the co-owned `ChatScreen.kt`). `./gradlew lint` + on-device confirm still pending.
|
||||
|
||||
## 2026-06-17 — Personality: server-owned on the gateway + picker-command handling
|
||||
|
||||
**Why.** Two reports against the personality flow. (1) Sending `/personality` (no arg) showed an agent reply bubble that appeared then vanished; (2) `/personality none` returned a confirmation but the app never reflected that the overlay was cleared. Verified the actual contract against the up-to-date upstream clone: `/personality` is a *picker command* (`hermes_cli/commands.py` `_PICKER_COMMANDS`) that the desktop/TUI never raw-forward — a bare command expands to an arg step, and a named/`none` value is applied via `config.set {key:"personality"}`, which persists `display.personality` + applies `ephemeral_system_prompt` live to the session and emits `session.info`. The app instead blindly forwarded every slash to `slash.exec`/`command.dispatch`, had no `none` concept, and never consumed `session.info` — so it kept injecting a stale per-turn personality prompt that fought the server.
|
||||
|
||||
- **Slash results stopped vanishing.** `ChatHandler.loadMessageHistory` did a wholesale reload preserving only `voice-intent-`/`steer-`/`ask-` ids; `system-notice-` (every `addSystemNotice` slash result) was wiped by the next turn's reconcile. Added `system-notice-` to the preserve allow-list — fixes the disappearing bubble for `/personality` and all other inline command output.
|
||||
- **Gateway client owns personality.** `GatewayChatClient` gained `serverPersonality: StateFlow<String?>`, `getPersonality()` (`config.get`), and `setPersonality()` (`config.set {key:"personality", value, session_id}`), plus a connection-level `session.info` interceptor that captures the `personality` field even with no turn in flight. `"none"`/`"default"`/`"neutral"` all clear the overlay (upstream `_validate_personality` conflates them).
|
||||
- **ViewModel mirrors server truth.** `selectPersonality` pushes via `config.set` on the gateway (optimistic, rolled back on a server reject with a now-durable notice) and only drives per-turn injection on the SSE fallbacks; `startStream` skips the persona-prompt injection entirely on the gateway so it can't double-apply. A `startPersonalitySync` collector + a ready-socket `config.get` seed keep `_selectedPersonality` reconciled to whatever the server/desktop/TUI set.
|
||||
- **Picker-command UX.** `/personality` is intercepted client-side like the desktop: bare `/personality` opens the agent sheet's Personality section (new `openPersonalityPicker` one-shot), `/personality <name|none>` routes to `selectPersonality`. The synthetic client **"Default" row was removed** — the picker is now **None** + the server-provided personalities (the configured default, if any, shows tagged `(default)` and highlights when active); upstream's active value is just `none` or a name, so the client shouldn't invent a third state. Added a `/personality none` palette entry. `AgentDisplay` treats `none`/`neutral` as cleared-overlay aliases (base identity, not the literal word) via `isClearedPersonality`.
|
||||
- **`/model` sibling fixed.** `/model` is the other picker command (`_PICKER_COMMANDS = {model, skin, personality}`; `skin` is `cli_only` and already excluded on mobile). A bare `/model` had the same raw-forward dead-end — now it opens the model picker (`openModelPicker` one-shot → `ModelPickerSheet`), while `/model <args>` stays a real gateway switch.
|
||||
- **Live model/provider sync.** The `session.info` interceptor now also surfaces `model` + `provider` (`serverModel`/`serverProvider` flows); the VM's `startGatewayStateSync` (renamed from `startPersonalitySync`) drives the model pill from them, so a `/model` switch on the desktop/TUI reflects live. Format-safe — the pill normalizes through `AgentDisplay.displayModelName`, and `session.info` keeps model/provider separate like `model.options`.
|
||||
- **No app reload for server-supplied data.** `refreshPersonalities()` (list + default + active `config.get`) now fires on agent-sheet open alongside `refreshModelOptions`, so a personality added/changed server-side appears without restarting the app; the active value also tracks live via `session.info`.
|
||||
- **Profile SOUL double-inject fixed.** On the gateway the session is bound to the selected profile (SOUL applied server-side) AND the personality rides `config.set` — so `startStream` now sends NO persona/profile prompt on the gateway (only the phone-status block), where it previously re-injected the profile's `systemMessage` on top of the server's own SOUL. SSE fallbacks keep the client-side precedence rules.
|
||||
- **Verification.** `:app:testSideloadDebugUnitTest` compiles the full module clean; new `AgentDisplayTest` cases for `isClearedPersonality` / `none`-as-cleared pass. `./gradlew lint` still the pre-push gate. On-device confirm of the live gateway round-trip pending.
|
||||
|
||||
## 2026-06-16 — Per-surface release notes (plugin + CLI parity with Android)
|
||||
|
||||
**Why.** Plugin and CLI GitHub Release bodies were static boilerplate baked into the workflow YAML (version-interpolated, but change-agnostic — a reader couldn't tell what a `plugin-v*`/`cli-v*` release actually changed). Only Android had real per-release notes (`RELEASE_NOTES.md` via `body_path`). Brought plugin and CLI up to the same Summary/Added/Changed/Fixed format.
|
||||
|
||||
- **New notes files.** `PLUGIN_RELEASE_NOTES.md` and `CLI_RELEASE_NOTES.md` at repo root — hand-written per release, same structure/scrub as `RELEASE_NOTES.md`. They keep the valuable Install/Verify sections but add a per-release "What's changed" block.
|
||||
- **Version stays auto-accurate.** Rather than hardcoding the version in install commands (three spots for CLI), the files use `__VERSION__` (and `__TAG__` for CLI) placeholders; each release workflow `sed`-renders them into a temp body before `softprops/action-gh-release` consumes it via `body_path`. Human writes prose, pipeline fills the version.
|
||||
- **Workflow wiring.** `release-plugin.yml` (package job, already checks out) and `release-cli.yml` got a "Render release notes" step + `body_path:` in place of inline `body:`. The CLI `publish-release` job had **no `actions/checkout`** (it only downloaded build artifacts) — added one so the notes file is present in that job.
|
||||
- **Docs.** RELEASE.md: §2 cross-references all three per-surface files; the plugin release recipe now updates + commits `PLUGIN_RELEASE_NOTES.md`; the CLI CI-behavior section documents `CLI_RELEASE_NOTES.md` + the placeholder substitution.
|
||||
- **Verification.** All three release workflow YAMLs parse (`yaml.safe_load`); plugin/CLI confirmed on `body_path` with no leftover inline body. Placeholder `sed` substitution validated by inspection. No release cut.
|
||||
|
||||
## 2026-06-16 — Dev-velocity tooling: Play auto-publish, worktree workflow, desktop UI preview
|
||||
|
||||
**Why.** Three workflow improvements to expedite shipping: automate the manual Play Console upload step, write down the worktree mental model, and cut the Compose UI edit→build→install loop.
|
||||
|
||||
- **Play Console auto-upload (CI).** `gradle-play-publisher` 4.0.0 was already configured in `app/build.gradle.kts` (`play { }`, DRAFT status) but `release-android.yml` never invoked it — Play upload was fully manual. Added a publish step to the `release` job, gated on a new optional `PLAY_SERVICE_ACCOUNT_JSON` secret and skipped for prerelease tags (dash in version). It runs `publishGooglePlayReleaseBundle --track=production`, landing the build as a Production **draft** so a human still clicks Start rollout. Closed the multi-flavor footgun structurally: a `playConfigs { register("sideload") { enabled.set(false) } }` block means only the `googlePlay` flavor can ever reach Play, even via the aggregate task. RELEASE.md updated (secrets table + §5 note). When the secret is unset CI prints a "skipped" summary line and behaves exactly as before.
|
||||
|
||||
- **Worktree workflow doc.** Added `docs/worktree-workflow.md` — a one-paragraph mental model (worktree = second folder on the same `.git`, warm caches per branch), four rules, the Orca-manages-worktrees note (use `orca-cli` worktree commands, not raw `git worktree`, here), raw-`git worktree` fallback + gotchas, and how it maps onto the existing `main`/`dev` no-ff release contract. Complements RELEASE.md "Branching policy" / decisions.md §23 without duplicating them.
|
||||
|
||||
- **Desktop UI preview module (`:ui-preview`).** New JVM-only Compose for Desktop module for hot-reload UI iteration on the PC. Compose Multiplatform 1.10.3 (bundles stable Compose Hot Reload, enabled by default for desktop targets) against the repo's Kotlin 2.3.21 / JVM 17. Follows the existing sphere pattern: shares the platform-agnostic `MorphingSphereCore.kt` algorithm from `:relay-ui` via a Gradle `srcDir` include (excluding the Android `MorphingSphere.kt` renderer, whose `@Preview`/`androidx.*.tooling` imports don't exist on desktop), and provides a thin `DesktopSphere.kt` renderer + a `Main.kt` gallery with a state selector and live sliders. Additive — `include(":ui-preview")` in `settings.gradle.kts` and a `.gitignore` build entry; no shipped artifact depends on it. **First-sync verification pending:** the module pins the one CMP version in the repo, to be confirmed on the next Studio sync (realign per the Compose compatibility matrix if Kotlin/CMP drift). Run via the IDE "Run with Compose Hot Reload" gutter or `./gradlew :ui-preview:run`.
|
||||
|
||||
- **Verification.** Kotlin/Gradle changes not built locally (per workflow: builds happen in Studio; `./gradlew lint` is the pre-push gate). YAML and Gradle edits are additive and reviewed by inspection; the `:ui-preview` module is isolated behind one settings include and verified-on-first-sync.
|
||||
|
||||
## 2026-06-16 — fix v1.0.0 connect force-close (corrupt keyset) + dashboard button contrast
|
||||
|
||||
**Why.** Two reports against v1.0.0 from the same reporter. (1) A force-close after a successful pair/connect on both Standard and Relay modes, surviving a cache clear. (2) Dashboard relay-plugin buttons rendered with text the same colour as their background. The crash was opaque from code review alone — every obvious connect-path was already guarded — until a Play Console stacktrace pinned it.
|
||||
|
||||
- **Crash — `AEADBadTagException` escaping the legacy token-store constructor.** The Play Console stack showed `EncryptedSharedPreferences.create` → `LegacyEncryptedPrefsTokenStore.buildPrefs` (`SessionTokenStore.kt`) → `AuthManager.store`. `EncryptedSharedPreferences` decrypts its Tink keyset eagerly on construction, so a corrupt legacy keyset (the classic post-upgrade / post-restore case: the encrypted blob persists but the hardware master key it was sealed against is gone) throws AES-GCM tag-mismatch straight out of the constructor. Every *accessor* on the store already healed via `resetPrefs()`, and `KeystoreTokenStore` hides construction behind `tryCreate`'s `try/return null` — but the legacy store is `new`-ed directly (the fallback when `tryCreate` returns null, and the migration source), so nothing caught a throwing constructor. It crashed in both modes because both call `AuthManager.store` to read the session token. A cache clear didn't help because the keyset lives in `data`, not `cache`.
|
||||
- **Fix — heal at construction + a non-persistent last resort.** `LegacyEncryptedPrefsTokenStore` now builds via `buildPrefsResilient()`: on any build failure it deletes the corrupt prefs file and rebuilds a fresh keyset against the current master key (the token in the unreadable file was lost regardless, so the user re-pairs). As defence-in-depth, `AuthManager.store()` wraps the legacy fallback in `runCatching` and degrades to a new in-memory `InMemoryTokenStore` if even the rebuild fails (a fundamentally broken keystore) — the app stays up and the user re-pairs each cold start instead of force-closing. Corrected the inaccurate `KeystoreTokenStore` comment that claimed its constructor "can't throw".
|
||||
- **Dashboard button contrast (#71).** `plugin/dashboard/src/styles.css` scopes a form-control reset `.hermes-relay-plugin button { color: inherit }`. Scoping bumps its specificity to `(0,1,1)`, which outranks the host shadcn Button's `text-*-foreground` utilities `(0,1,0)`, so solid-variant buttons painted their label in the inherited container foreground — which on the dashboard theme nearly matches the button background. Removing the rule (the reporter's local workaround) would break inputs/textareas (they need light text on the dark `--hr-bg`) and ghost/outline buttons (they rely on inheritance). Fix follows the file's existing `.bg-white { …; color }` idiom: solid variants `.bg-primary` / `.bg-secondary` / `.bg-destructive` re-assert their paired foreground colour at `(0,2,0)`, winning back over the reset without `!important`. `dist/style.css` re-synced via the package's `copyFileSync` build step.
|
||||
- **Verification.** Dashboard CSS verified by analysis against the variants actually used (`destructive`×6, `secondary`×1, bare-default `bg-primary`; `outline`×23 / `ghost`×4 correctly keep inheriting). Kotlin changes are localized to `SessionTokenStore.kt` + `AuthManager.kt`; on-device confirmation and `gradlew lint` pending a Studio build.
|
||||
|
||||
## 2026-06-15 — Claude review required check and Dependabot PR cleanup
|
||||
|
||||
**Why.** Open Dependabot PRs targeting `main` were blocked by the required `claude-review` check. Re-running a current Dependabot PR showed the Claude GitHub App token exchange succeeds, but `anthropics/claude-code-action` stops before review because the actor is `dependabot[bot]` and bot actors are not allow-listed. Dependabot-triggered runs also do not expose the same secret surface as human-authored PRs, so forcing Claude review on those PRs is the wrong gate.
|
||||
|
||||
- **Workflow fix.** `.github/workflows/claude-code-review.yml` now detects bot-authored PRs with `github.event.pull_request.user.type == 'Bot'` and emits a passing no-op `claude-review` job. Human-authored PRs still run the full Claude Code Review action; aggregate `dev` -> `main` release PRs still use the existing no-op skip.
|
||||
- **Workflow self-change guard.** PRs that edit `claude-code-review.yml` now also no-op after checkout when the changed-file list includes that workflow. The Claude action requires the workflow file to match the default branch before app-token exchange, so workflow maintenance PRs must not invoke the action they are changing.
|
||||
- **Dependency PR cleanup.** Batched overlapping Gradle version-catalog updates after the required-check fix: Kotlin `2.3.21`, Compose BOM `2026.05.01`, Navigation Compose `2.9.8`, Activity Compose `1.13.0`, DataStore `1.2.1`, Markdown Renderer `0.41.0`, Media3 `1.10.1`, and Foojay resolver `1.0.0`. The Android Gradle Plugin and `softprops/action-gh-release` bumps were already present on `main`, so their stale Dependabot PRs were superseded by current main state.
|
||||
- **Verification.** Confirmed `CLAUDE_CODE_OAUTH_TOKEN` was refreshed in repository secrets after Claude Code GitHub setup. Re-ran Claude Code Review on PR #46 and confirmed the current blocker was bot-actor policy, not GitHub App installation. `git diff --check`, `.\gradlew.bat lint`, `.\gradlew.bat assembleDebug`, and the focused `:app:testSideloadDebugUnitTest` CI slice pass with `ANDROID_HOME` pointed at the local SDK.
|
||||
|
||||
## 2026-06-14 — profile chat: turn-complete wipe + cross-transport session continuity
|
||||
|
||||
**Why.** v1.0.0 polish (on `dev`, into release PR #61). After the per-profile session work landed, on-device testing of a non-default agent showed: a turn streamed fine (thinking + reply), then on turn-complete the chat **switched to a new/empty session** untouched; the just-finished conversation only reappeared in the drawer a moment later. Watched `adb logcat` during a repro to confirm root cause.
|
||||
@@ -433,7 +794,7 @@ One smoke-artifact: `~/.hermes/remote-sessions.json` got emptied during agent te
|
||||
|
||||
### Cut `desktop-v0.3.0-alpha.1`
|
||||
|
||||
`desktop/package.json` bumped 0.1.0 → 0.3.0-alpha.1 to align the published package version with the release-track tag. Build clean; `node bin/hermes-relay.js --version` prints `0.3.0-alpha.1`. Once this lands on `main` and the tag pushes, `release-desktop.yml` cross-compiles four Bun binaries (win-x64, linux-x64, darwin-x64, darwin-arm64), uploads with `SHA256SUMS.txt`, and the `install.{sh,ps1}` one-liners start working for any user.
|
||||
`desktop/package.json` bumped 0.1.0 → 0.3.0-alpha.1 to align the published package version with the release-track tag. Build clean; `node bin/hermes-relay.js --version` prints `0.3.0-alpha.1`. Once this lands on `main` and the tag pushes, the CLI release workflow cross-compiles four Bun binaries (win-x64, linux-x64, darwin-x64, darwin-arm64), uploads with `SHA256SUMS.txt`, and the `install.{sh,ps1}` one-liners start working for any user.
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
# GEMINI.md
|
||||
|
||||
Agent instructions for **Hermes-Relay**. This file exists so Gemini CLI (which
|
||||
does not read `AGENTS.md` natively) picks up the project's guidance.
|
||||
|
||||
**Read [AGENTS.md](AGENTS.md) — it is the single source of truth** for every
|
||||
coding agent: the entry point, the non-negotiables (standard-path-is-vanilla-
|
||||
upstream, verify-endpoints, Conventional Commits + `main`/`dev` branching, the
|
||||
per-language stack rules), and the public-repo writing hygiene. It links on to
|
||||
`CLAUDE.md` for the deep reference (architecture, upstream Hermes API, repo
|
||||
layout, code style, the dev loop, and the Key Files map).
|
||||
|
||||
Do not restate rules here — keep them in `AGENTS.md` so they can't drift.
|
||||
@@ -0,0 +1,39 @@
|
||||
# Hermes-Relay-Plugin v__VERSION__
|
||||
|
||||
**Release Date:** June 20, 2026
|
||||
**Since the previous plugin release:** A new, removable **enhancement layer** that lets the relay teach the agent things only the relay knows — starting with sensitive-media classification — plus provider-aware enhanced voice and an isolated, TUI-tuned tmux for relay terminals.
|
||||
|
||||
This release adds a clean way for the relay to extend the agent without forking or touching the user's soul/memory. The first use is **sensitive-media classification**: the relay appends a small, auditable system-prompt block teaching the agent to mark private/NSFW media so the paired phone can blur it — with sensitivity staying model-emitted. It's on by default for relay installs (installing the relay is the opt-in), reversible from the dashboard or an env flag, fully visible over a new audit route, and a complete no-op on vanilla upstream. Voice gains provider-aware controls for Gemini and xAI, and relay terminals now run on a dedicated, correctly-configured tmux.
|
||||
|
||||
## What's changed
|
||||
|
||||
### Added
|
||||
- **Relay enhancement layer + agent-context injection.** A reusable, removable layer that injects auditable, fenced blocks into the agent's system prompt at plugin-load. Fail-open at every step (seam absent / block build throws ⇒ base prompt unchanged), config-gated, and a byte-for-byte no-op on vanilla upstream. Built to be retired per-surface as upstream adds a context hook — the same pattern as the bootstrap route shims. See `docs/plans/2026-06-20-relay-enhancement-layer.md`.
|
||||
- **Sensitive-media classification (first block).** Teaches the agent to mark private/NSFW media with the client's spoiler convention so the phone blurs it per the user's setting. **On by default for relay installs**; opt out with `RELAY_AGENT_CONTEXT_ENABLED=0` or the dashboard toggle. Sensitivity stays model-emitted — no relay-side or on-device classifier. No soul/memory is touched.
|
||||
- **`GET /context/injected` audit route.** The relay exposes exactly what it would inject (loopback-open, bearer-gated remotely), so the injection is never hidden — surfaced in the Android chat "What the agent sees" sheet as "Relay context (server-side)".
|
||||
- **Dashboard Agent-context controls.** The Relay management tab gained a master toggle and per-block toggles (labeled experimental / server-side / removable), shown on-by-default for relay installs.
|
||||
- **Provider-aware enhanced voice (Gemini + xAI).** `/voice/synthesize` accepts per-request overrides so a paired client can steer a Gemini voice/model with expressive tone tags, or an xAI voice with expressive speech tags, without changing the server's global voice config.
|
||||
|
||||
### Changed
|
||||
- **Relay terminals run on an isolated, TUI-tuned tmux.** Sessions spawn on a dedicated tmux server/socket with a generated config — `escape-time 0`, truecolor `tmux-256color`, `mouse`/`focus-events` on, `status off` — so editors and full-screen tools behave correctly without touching the user's personal tmux.
|
||||
|
||||
### Fixed
|
||||
- **Relay voice synthesis no longer leaves temporary audio files behind** on the server.
|
||||
- **Clearer voice errors.** Standard voice rejects an over-long recording before uploading and returns a helpful message for audio the server can't read, instead of a generic HTTP error.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install hermes-relay==__VERSION__
|
||||
```
|
||||
|
||||
## Verify
|
||||
|
||||
```bash
|
||||
python -m relay_server --help
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
Tag prefixes: Android releases use `android-v*`, CLI releases use `cli-v*`. Historical
|
||||
relay/plugin releases used `relay-v*` tags.
|
||||
@@ -38,6 +38,10 @@ Hermes-Relay puts your [Hermes agent](https://github.com/NousResearch/hermes-age
|
||||
|
||||
A vanilla [hermes-agent](https://github.com/NousResearch/hermes-agent) install is enough — chat, management, and voice need **no plugin**. Add the optional relay only when you want terminal, phone control, or the CLI's tools. **Pair once from either surface; both work.**
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/diagrams/architecture-homepage.png" alt="How Hermes-Relay connects — Vanilla Hermes (Chat, Manage, Voice) runs with no plugin; the optional Relay plugin adds Terminal, Bridge, relay voice and desktop tools to the app and CLI; Device Control needs the sideload build." width="900">
|
||||
</p>
|
||||
|
||||
## Quick Start (Android)
|
||||
|
||||
Install → connect → talk, in about two minutes.
|
||||
@@ -51,7 +55,7 @@ Sideload builds check GitHub for updates and show a one-tap banner when you're b
|
||||
|
||||
### 2 · Have Hermes running
|
||||
|
||||
The app needs your Hermes **API server enabled and reachable from your phone**, plus an **API key** — the token the app sends to authenticate Chat (pick any value you like). Installing Hermes and choosing a provider is standard Hermes setup; the [full walkthrough](https://codename-11.github.io/hermes-relay/guide/getting-started) covers Windows, the dashboard for **Manage**, LAN scan, and QR setup.
|
||||
The app needs your Hermes **API server enabled and reachable from your phone**, plus an **API key** — the token the app sends to authenticate Chat (pick any value you like). Installing Hermes and choosing a provider is vanilla Hermes setup; the [full walkthrough](https://codename-11.github.io/hermes-relay/guide/getting-started) covers Windows, the dashboard for **Manage**, LAN scan, and QR setup.
|
||||
|
||||
```bash
|
||||
hermes setup --portal # install / log in / pick a provider — skip if already done
|
||||
@@ -78,9 +82,9 @@ hermes gateway
|
||||
|
||||
Open the app and pick how to connect — any of:
|
||||
|
||||
- **Standard Hermes** → tap **Scan for Hermes on LAN** to auto-find the server, then enter your key.
|
||||
- **Standard Hermes** → type the address (`http://<host>:8642`) and key by hand.
|
||||
- **Scan setup QR** → ask your Hermes agent to generate a QR with your URL + key (e.g. `{"api_url":"http://<host>:8642","api_key":"<key>"}`) and scan it.
|
||||
- **Vanilla Hermes** → tap **Scan for Hermes on LAN** to auto-find the server, then enter your key.
|
||||
- **Vanilla Hermes** → type the address (`http://<host>:8642`) and key by hand.
|
||||
- **Scan setup QR** → ask your Hermes agent to generate a QR with your URL + key (e.g. `{"api_url":"http://<host>:8642","api_key":"<key>","dashboard_url":"http://<host>:9119"}`) and scan it. `dashboard_url` is optional when the dashboard uses the conventional same-host `:9119` URL.
|
||||
|
||||
The wizard probes everything and finishes with a capability card:
|
||||
|
||||
@@ -92,7 +96,7 @@ The wizard probes everything and finishes with a capability card:
|
||||
| **Remote** | Fallback route configured — keeps working away from home |
|
||||
| **Relay** | Optional power tools — fine to leave unpaired |
|
||||
|
||||
If your dashboard requires sign-in, do it once under the **Manage** tab — the same session unlocks voice. That's the whole standard setup.
|
||||
If your dashboard requires sign-in, do it once under the **Manage** tab — the same session unlocks voice. That's the whole Vanilla Hermes setup.
|
||||
|
||||
> **Going places?** Put your server's Tailscale URL in the setup form's *Remote access* field (or add a route any time under **Settings → Connections → Routes**). The app uses LAN at home and switches routes automatically when you leave. See [Remote access](https://codename-11.github.io/hermes-relay/guide/remote-access).
|
||||
|
||||
@@ -101,20 +105,34 @@ If your dashboard requires sign-in, do it once under the **Manage** tab — the
|
||||
Install the Relay plugin on the server only when you want Terminal, Bridge phone control, relay sessions, media routes, or the realtime voice engine:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/install.sh | bash
|
||||
hermes plugins install Codename-11/hermes-relay/plugin --enable
|
||||
hermes relay doctor
|
||||
hermes relay start --no-ssl
|
||||
hermes pair
|
||||
```
|
||||
|
||||
The installer clones to `~/.hermes/hermes-relay/`, registers the plugin/skill paths, and can install a systemd user service. Scan the QR from the phone's Connections screen — or use `hermes pair --register-code ABCD12` with the manual code from Android **Settings → Connections → Advanced**.
|
||||
Use the legacy installer instead if you also want the systemd user service,
|
||||
shell shims, and the full clone/update workflow:
|
||||
|
||||
- **Update:** `hermes-relay-update` (idempotent) — or re-run the install one-liner.
|
||||
- **Uninstall:** `bash ~/.hermes/hermes-relay/uninstall.sh` — reverses every step, never touches shared Hermes state. Flags: `--dry-run`, `--keep-clone`, `--remove-secret`.
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/install.sh | bash
|
||||
```
|
||||
|
||||
The plugin-manager install owns the plugin code, dashboard tab, CLI commands,
|
||||
and agent tools. `hermes relay compat status/install/remove` manages only the
|
||||
optional legacy API compatibility hook when an older Hermes build needs it. Scan
|
||||
the QR from the phone's Connections screen — or use
|
||||
`hermes pair --register-code ABCD12` with the manual code from Android
|
||||
**Settings → Connections → Advanced**.
|
||||
|
||||
- **Plugin-manager uninstall:** `hermes relay compat remove --all` if you installed the optional hook, then `hermes plugins remove hermes-relay`.
|
||||
- **Legacy installer update:** `hermes-relay-update` (idempotent) — or re-run the install one-liner.
|
||||
- **Legacy installer uninstall:** `bash ~/.hermes/hermes-relay/uninstall.sh` — removes the service, shims, clone, external skill path, editable package, and compat hook. It never touches shared Hermes state. Flags: `--dry-run`, `--keep-clone`, `--remove-secret`.
|
||||
- **Dashboard plugin:** installs with the same symlink — restart the gateway and a **Relay** tab (paired devices, bridge activity, media tokens) appears in the web UI.
|
||||
|
||||
Full server setup, TLS, and systemd details: [docs/relay-server.md](docs/relay-server.md).
|
||||
|
||||
**Requirements:** Android 8.0+ (SDK 26) · [hermes-agent](https://github.com/NousResearch/hermes-agent) v0.8.0+ with Python 3.11+ on the server.
|
||||
**Requirements:** Android 8.0+ (SDK 26) · current upstream [hermes-agent](https://github.com/NousResearch/hermes-agent) with the API server and dashboard enabled · Python 3.11+ on the server.
|
||||
|
||||
## Screenshots
|
||||
|
||||
@@ -126,10 +144,10 @@ Full server setup, TLS, and systemd details: [docs/relay-server.md](docs/relay-s
|
||||
<td align="center" width="25%"><img src="assets/screenshots/04_sessions.png" alt="Session history" width="100%"><br><sub><b>Session history</b></sub></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" width="25%"><img src="assets/screenshots/05_commands.png" alt="Command palette" width="100%"><br><sub><b>Command palette</b></sub></td>
|
||||
<td align="center" width="25%"><img src="assets/screenshots/05_themes.png" alt="App themes" width="100%"><br><sub><b>App themes</b></sub></td>
|
||||
<td align="center" width="25%"><img src="assets/screenshots/06_manage.png" alt="Manage your agent" width="100%"><br><sub><b>Manage your agent</b></sub></td>
|
||||
<td align="center" width="25%"><img src="assets/screenshots/07_connections.png" alt="Connections and routes" width="100%"><br><sub><b>Connections & routes</b></sub></td>
|
||||
<td align="center" width="25%"><img src="assets/screenshots/08_settings.png" alt="Settings" width="100%"><br><sub><b>Settings</b></sub></td>
|
||||
<td align="center" width="25%"><img src="assets/screenshots/08_appearance.png" alt="Agent avatar & skins" width="100%"><br><sub><b>Avatars & skins</b></sub></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
@@ -139,7 +157,7 @@ Full server setup, TLS, and systemd details: [docs/relay-server.md](docs/relay-s
|
||||
|
||||
### Android
|
||||
|
||||
- **Streaming chat** — direct SSE to the Hermes API server with live markdown, tool-call cards, session history, a searchable command palette, file attachments, quote-in-reply, conversation share, and send-while-streaming queuing.
|
||||
- **Streaming chat** — rides vanilla Hermes, preferring the dashboard gateway (`/api/ws`, live thinking) when signed in to Manage and falling back to API-server SSE otherwise, with live markdown, tool-call cards, session history, a searchable command palette, file attachments, quote-in-reply, conversation share, and send-while-streaming queuing.
|
||||
- **Manage your agent** — the full Hermes dashboard, native: switch models from your provider catalog, manage keys (write-only, masked, rate-limited reveal), create and edit profiles including `SOUL.md`, and browse/install/update skills. One dashboard sign-in covers it all.
|
||||
- **Hands-free voice** — talk on a vanilla install: speech rides your server's configured providers, unlocked by the same Manage sign-in. Relay-paired setups add per-profile voice and an opt-in provider-native Realtime Agent with background task handoff.
|
||||
- **Works away from home** — add a Tailscale or public URL and the app roams automatically (LAN at home, fallback elsewhere). An unreachable server gets a diagnosis, not just a red dot.
|
||||
@@ -167,7 +185,7 @@ hermes-relay daemon # headless tool router — agent
|
||||
hermes-relay update # self-update via GitHub Releases
|
||||
```
|
||||
|
||||
It pairs against the **same relay and credential store** as the Android app — pair once from either, both work. Tagged on a separate `desktop-v*` [release track](https://github.com/Codename-11/hermes-relay/releases?q=desktop).
|
||||
It pairs against the **same relay and credential store** as the Android app — pair once from either, both work. Tagged on a separate `cli-v*` [release track](https://github.com/Codename-11/hermes-relay/releases?q=cli), with old alpha prereleases still visible under `desktop-v*`.
|
||||
|
||||
- **Docs:** [CLI guide](https://codename-11.github.io/hermes-relay/desktop/) · [`desktop/README.md`](desktop/README.md)
|
||||
- **AI-agent setup recipe:** `/hermes-relay-desktop-setup`
|
||||
@@ -175,13 +193,19 @@ It pairs against the **same relay and credential store** as the Android app —
|
||||
## How It Works
|
||||
|
||||
```
|
||||
Phone (HTTP/SSE) --> Hermes API Server (:8642) [chat — direct]
|
||||
Phone (HTTP) --> Hermes Dashboard (:9119) [manage + standard voice — cookie sign-in]
|
||||
Phone (HTTP/WSS) --> Hermes Dashboard (:9119) [chat gateway, manage, vanilla voice]
|
||||
Phone (HTTP/SSE) --> Hermes API Server (:8642) [chat fallback, sessions, runs]
|
||||
Phone (WSS/HTTP) --> Relay (:8767) [terminal, bridge, media, relay voice, sessions]
|
||||
CLI (WSS) --> Relay (:8767) [machine tools, tui, terminal]
|
||||
```
|
||||
|
||||
Chat connects **directly** to the Hermes API server with the API key — the same pattern Open WebUI and other Hermes frontends use. Manage and standard voice ride the Hermes dashboard with its own one-time sign-in, so a vanilla install needs no plugin for either. The optional relay on `:8767` adds the power surfaces — terminal, bridge phone control, media handoff, machine tools, and relay-side voice (preferred automatically when paired). One QR can configure API, dashboard, and relay routes without merging their auth models.
|
||||
Chat prefers the Hermes dashboard gateway when Manage auth is ready, then falls
|
||||
back to the upstream API server SSE path with the API key. Manage and Vanilla Hermes
|
||||
voice ride the Hermes dashboard with its own one-time sign-in, so a vanilla
|
||||
install needs no plugin for either. The optional relay on `:8767` adds the power
|
||||
surfaces: terminal, bridge phone control, media handoff, machine tools, and
|
||||
relay-side voice, which is preferred automatically when paired. One QR can
|
||||
configure API, dashboard, and relay routes without merging their auth models.
|
||||
|
||||
## Documentation
|
||||
|
||||
@@ -194,7 +218,7 @@ Chat connects **directly** to the Hermes API server with the API key — the sam
|
||||
| [API Reference](https://codename-11.github.io/hermes-relay/reference/api.html) | Hermes API endpoints used by both surfaces |
|
||||
| [Specification](docs/spec.md) | Full spec — protocol, UI, phases, dependencies |
|
||||
| [Architecture Decisions](docs/decisions.md) | ADRs — framework, channels, auth, terminal |
|
||||
| [Changelog](CHANGELOG.md) | Release history (`android-v*`, `server-v*`, `desktop-v*`) |
|
||||
| [Changelog](CHANGELOG.md) | Release history (`android-v*`, `plugin-v*`, `cli-v*`) |
|
||||
|
||||
<details>
|
||||
<summary><b>Install with an AI agent</b> — paste-ready prompt for Claude / GPT</summary>
|
||||
@@ -212,7 +236,7 @@ Read the canonical setup recipe before acting:
|
||||
Then guide me through:
|
||||
- Verifying hermes-agent is already installed (it's a prerequisite — Hermes-Relay is a plugin, not standalone)
|
||||
- Running the server-plugin install one-liner: `curl -fsSL https://raw.githubusercontent.com/Codename-11/hermes-relay/main/install.sh | bash`
|
||||
- Connecting my phone by Standard Hermes API URL/key first, then optionally pairing Relay via `hermes pair` or `/hermes-relay-pair` for power tools; OR pairing my laptop via the Hermes-Relay CLI (`irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex` on Windows, then `hermes-relay pair --remote ws://<host>:8767`)
|
||||
- Connecting my phone by Vanilla Hermes API URL/key first, then optionally pairing Relay via `hermes pair` or `/hermes-relay-pair` for power tools; OR pairing my laptop via the Hermes-Relay CLI (`irm https://raw.githubusercontent.com/Codename-11/hermes-relay/main/desktop/scripts/install.ps1 | iex` on Windows, then `hermes-relay pair --remote ws://<host>:8767`)
|
||||
- Verifying with `hermes-status` (server) or `hermes-relay doctor` (CLI)
|
||||
|
||||
Always confirm before running shell commands. Never restart hermes-gateway without asking. If any step fails, consult the Troubleshooting section in the SKILL.md and ask me for the exact error.
|
||||
@@ -263,7 +287,7 @@ hermes-relay/
|
||||
├── user-docs/ # VitePress documentation site
|
||||
├── docs/ # Spec, decisions, security
|
||||
├── scripts/ # Dev helper scripts
|
||||
├── .github/workflows/ # CI + release pipelines (ci-android / ci-server / ci-desktop)
|
||||
├── .github/workflows/ # CI + release pipelines (ci-android / ci-plugin / ci-desktop)
|
||||
└── gradle/ # Wrapper (8.13) + version catalog
|
||||
```
|
||||
|
||||
|
||||
@@ -13,20 +13,23 @@ with optional prerelease identifiers.
|
||||
- `PATCH` — bug fixes, backwards compatible
|
||||
- Prerelease suffixes: `-alpha`, `-beta`, `-rc.N` (e.g. `0.2.0-beta.1`)
|
||||
|
||||
Hermes-Relay now ships three independently versioned surfaces:
|
||||
Hermes-Relay now ships three independently versioned surfaces. Public GitHub
|
||||
Release titles use product names (`Hermes-Relay-Android`,
|
||||
`Hermes-Relay-Plugin`, `Hermes-Relay-CLI`); tag prefixes stay short and stable
|
||||
for automation.
|
||||
|
||||
| Surface | Tag prefix | Version source | Bump script | Release workflow |
|
||||
|---|---|---|---|---|
|
||||
| Android app | `android-v*` | `gradle/libs.versions.toml` | `scripts/bump-android-version.sh` | `.github/workflows/release-android.yml` |
|
||||
| Server / Python package | `server-v*` | `pyproject.toml` plus checked plugin/dashboard metadata | `scripts/bump-server-version.sh` | `.github/workflows/release-server.yml` |
|
||||
| Desktop CLI | `desktop-v*` | `desktop/package.json` | `npm version` or manual package bump | `.github/workflows/release-desktop.yml` |
|
||||
| Hermes-Relay-Android | `android-v*` | `gradle/libs.versions.toml` | `scripts/bump-android-version.sh` | `.github/workflows/release-android.yml` |
|
||||
| Hermes-Relay-Plugin | `plugin-v*` | `pyproject.toml` plus checked plugin/dashboard metadata | `scripts/bump-plugin-version.sh` | `.github/workflows/release-plugin.yml` |
|
||||
| Hermes-Relay-CLI | `cli-v*` | `desktop/package.json` | `npm version` or manual package bump | `.github/workflows/release-cli.yml` |
|
||||
|
||||
This split is intentional. The server now carries features for both Android
|
||||
and desktop, so server fixes can ship without forcing an Android app
|
||||
`versionCode` bump, and desktop CLI alphas can continue on their own cadence.
|
||||
Historical Android releases before this naming split used bare `v*` tags, and
|
||||
historical server releases used `relay-v*` tags. New releases use the explicit
|
||||
surface prefixes above.
|
||||
This split is intentional. The plugin carries relay features for both Android
|
||||
and CLI clients, so plugin fixes can ship without forcing an Android app
|
||||
`versionCode` bump, and CLI alphas can continue on their own cadence. Historical
|
||||
Android releases before this naming split used bare `v*` tags. Historical
|
||||
plugin/server releases used `relay-v*` tags, and historical CLI prereleases used
|
||||
`desktop-v*` tags. New releases use the explicit tag prefixes above.
|
||||
|
||||
### Android app versioning
|
||||
|
||||
@@ -70,35 +73,47 @@ bash scripts/bump-android-version.sh 0.6.2
|
||||
`scripts/bump-version.sh` remains as a backward-compatible alias for the
|
||||
Android script.
|
||||
|
||||
### Server / Python package versioning
|
||||
### Plugin / Python package versioning
|
||||
|
||||
Server version metadata lives in these server-owned files and must stay in
|
||||
Plugin version metadata lives in these plugin-owned files and must stay in
|
||||
lockstep:
|
||||
|
||||
| File | Line | Purpose |
|
||||
|---|---|---|
|
||||
| `pyproject.toml` | `version = "..."` | Python package metadata |
|
||||
| `plugin/relay/__init__.py` | `__version__ = "..."` | runtime version reported by `/health` |
|
||||
| `plugin/relay/__init__.py` | `__version__ = "..."` | runtime version reported by `/health` and `/relay/info` |
|
||||
| `plugin/plugin.yaml` | `version: ...` | Hermes plugin metadata |
|
||||
| `plugin/dashboard/manifest.json` | `"version": "..."` | Hermes dashboard plugin metadata |
|
||||
| `plugin/dashboard/package.json` | `"version": "..."` | dashboard build/package metadata |
|
||||
| `plugin/dashboard/package-lock.json` | `"version": "..."` | locked dashboard package metadata |
|
||||
|
||||
Always bump Server releases via:
|
||||
Always bump Plugin releases via:
|
||||
|
||||
```bash
|
||||
bash scripts/bump-server-version.sh 0.6.2
|
||||
bash scripts/bump-plugin-version.sh 0.6.2
|
||||
```
|
||||
|
||||
Check the current metadata with:
|
||||
|
||||
```bash
|
||||
python scripts/check-server-version-sync.py
|
||||
python scripts/check-plugin-version-sync.py
|
||||
```
|
||||
|
||||
The `server-v*` release workflow validates the tag against the same metadata,
|
||||
runs server tests, builds a wheel and sdist, generates checksums, and publishes
|
||||
a GitHub Release with the package artifacts.
|
||||
Check all release tracks at once with:
|
||||
|
||||
```bash
|
||||
python scripts/check-version-tracks.py
|
||||
```
|
||||
|
||||
This aggregate check reports Android, plugin, and CLI versions
|
||||
side by side and validates that each track's own source files are internally
|
||||
consistent. It deliberately does not require all three tracks to share the same
|
||||
SemVer.
|
||||
|
||||
The `plugin-v*` release workflow validates the tag against the same metadata,
|
||||
runs plugin tests, builds a wheel and sdist, generates checksums, and
|
||||
publishes a `Hermes-Relay-Plugin vX.Y.Z` GitHub Release with the package
|
||||
artifacts.
|
||||
|
||||
## Branching policy
|
||||
|
||||
@@ -156,15 +171,15 @@ Squash merges lose that detail and are **not** the house style.
|
||||
### Version bumps happen at release-prep on `dev`, NOT on feature branches
|
||||
|
||||
Feature branches **never** touch `gradle/libs.versions.toml`,
|
||||
server-owned version metadata, or `desktop/package.json`.
|
||||
plugin-owned version metadata, or `desktop/package.json`.
|
||||
If two feature branches both bumped a release version, they'd collide on
|
||||
version files and, for Android, on `appVersionCode` (which must be
|
||||
monotonic).
|
||||
|
||||
Version-bump commits live on `dev` as the last commit of release-prep
|
||||
work. Android commits use `release(android): android-vX.Y.Z`; server commits
|
||||
use `release(server): server-vX.Y.Z`; desktop commits use the existing
|
||||
`release: desktop-vX.Y.Z` convention. A release PR then merges `dev` →
|
||||
work. Android commits use `release(android): android-vX.Y.Z`; plugin commits
|
||||
use `release(plugin): plugin-vX.Y.Z`; CLI commits use
|
||||
`release(cli): cli-vX.Y.Z`. A release PR then merges `dev` →
|
||||
`main` with `--no-ff`, and the matching tag is cut from the resulting
|
||||
`main` tip.
|
||||
|
||||
@@ -173,7 +188,7 @@ use `release(server): server-vX.Y.Z`; desktop commits use the existing
|
||||
Light branch protection is enabled:
|
||||
|
||||
- **`main`** — direct pushes blocked; only release PRs from `dev` merge
|
||||
here. PR must pass CI (Android + Server) before merge. Force push and
|
||||
here. PR must pass CI (Android + Plugin) before merge. Force push and
|
||||
branch deletion blocked.
|
||||
- **`dev`** — direct pushes blocked for non-trivial work; feature
|
||||
branches PR in. PR must pass CI. Force push and branch deletion
|
||||
@@ -275,22 +290,34 @@ for the full text.
|
||||
|
||||
### 3. Play Developer API service account (optional)
|
||||
|
||||
Required only if you want `gradlew publishReleaseBundle` to upload directly
|
||||
to Play Console. Manual UI uploads work without this.
|
||||
Required for automated upload (the `android-v*` workflow's Play step, or local
|
||||
`gradlew publishGooglePlayReleaseBundle`). Manual UI uploads work without this.
|
||||
|
||||
1. Open <https://console.cloud.google.com/> and select the project linked
|
||||
to your Play Console account (Play Console > Setup > API access shows
|
||||
which one).
|
||||
2. **IAM & Admin > Service Accounts > Create Service Account** (e.g.
|
||||
`hermes-relay-publisher`). No project roles needed.
|
||||
3. On the new service account, **Keys > Add key > Create new key > JSON**
|
||||
and download the file.
|
||||
4. In Play Console > **Setup > API access**, find the service account,
|
||||
click **Grant access**, and assign the **Release manager** role.
|
||||
5. Save the JSON as `play-service-account.json` in the repo root (already
|
||||
in `.gitignore`).
|
||||
6. Verify with `gradlew bootstrapReleasePlayResources` — should succeed
|
||||
without auth errors.
|
||||
The service account is **created in Google Cloud Console** and then **authorized
|
||||
in Play Console** — two separate consoles. (Play Console's older "Setup > API
|
||||
access" page has been reorganized; there is no longer a "Setup" group. Use the
|
||||
paths below.)
|
||||
|
||||
1. **Create the service account (Google Cloud Console).** Open
|
||||
<https://console.cloud.google.com/iam-admin/serviceaccounts>, pick the project
|
||||
(any project works; if Play Console's **API access** page already names a linked
|
||||
project, use that one). **Create service account** → name it e.g.
|
||||
`hermes-relay-publisher` → **Done**. No project roles needed.
|
||||
2. **Create a JSON key.** On the new service account → **Keys** tab → **Add key >
|
||||
Create new key > JSON** → download. This file's *contents* are the secret.
|
||||
3. **Authorize it in Play Console.** Open the Play Console account-level left
|
||||
sidebar → **Users and permissions** → **Invite new users** → paste the service
|
||||
account's email (`...@...iam.gserviceaccount.com`). Under **App permissions**
|
||||
(for `com.axiomlabs.hermesrelay`) or **Account permissions**, grant the
|
||||
**Release** permissions — "Release apps to testing tracks" and "Release to
|
||||
production, exclude devices, and use Play App Signing" — plus "View app
|
||||
information". (Granting **Admin (all permissions)** also works but is broader
|
||||
than needed.) **Invite user**.
|
||||
4. **Use it.** For CI, paste the JSON contents into the `PLAY_SERVICE_ACCOUNT_JSON`
|
||||
repo secret (step 4 / secrets table). For local publish, save the JSON as
|
||||
`play-service-account.json` in the repo root (already in `.gitignore`).
|
||||
5. Verify locally with `gradlew bootstrapGooglePlayReleaseResources` — succeeds
|
||||
without auth errors once permissions propagate (allow a few minutes).
|
||||
|
||||
### 4. GitHub Actions secrets
|
||||
|
||||
@@ -350,6 +377,12 @@ the new app version and a higher `appVersionCode`.
|
||||
|
||||
### 2. Update release notes and changelog
|
||||
|
||||
> Each surface has its own GitHub-Release-body file, all in the same format
|
||||
> (Summary + Added/Changed/Fixed + Install/Verify): `RELEASE_NOTES.md` (Android),
|
||||
> `PLUGIN_RELEASE_NOTES.md` (plugin), `CLI_RELEASE_NOTES.md` (CLI). This step covers
|
||||
> the Android artifacts; the plugin/CLI files are filled in their own release
|
||||
> sections below but follow the identical scrub and Keep-a-Changelog grouping.
|
||||
|
||||
- `CHANGELOG.md` — promote the accumulated `[Unreleased]` block to a
|
||||
versioned header. The block already exists: every feature PR has
|
||||
been appending to it. All you do here is:
|
||||
@@ -372,6 +405,13 @@ the new app version and a higher `appVersionCode`.
|
||||
shown in the settings/about screen. Update with the version number
|
||||
and a brief feature summary. Gets stale silently if forgotten
|
||||
(v0.4.0 shipped with 0.1.0 content until caught post-release).
|
||||
- `app/src/googlePlay/play/release-notes/en-US/default.txt` — the Play
|
||||
Console **"What's new"** text, which gradle-play-publisher reads at
|
||||
upload to fill the Production-draft release notes. This is **separate**
|
||||
from `RELEASE_NOTES.md` (that one is only the GitHub Release body) — if
|
||||
this file is missing or stale, the Play draft ships with empty/wrong
|
||||
notes (shipped empty in v1.1.0 until caught post-release). Keep it
|
||||
**≤500 chars per language**, user-facing, Android-only.
|
||||
- `docs/play-store-listing.md` — Play Store listing copy. Update
|
||||
the version reference and the "Release Notes" section that gets
|
||||
pasted into the Play Console "What's new" field. Keep the Play
|
||||
@@ -452,64 +492,100 @@ Pushing a tag matching `android-v*` triggers `.github/workflows/release-android.
|
||||
which builds, signs, checksums, and creates a GitHub Release. Watch the
|
||||
run under the **Actions** tab.
|
||||
|
||||
Server/Python version files are intentionally not part of an Android app
|
||||
release unless the server package itself is also being released.
|
||||
Plugin/Python version files are intentionally not part of an Android app
|
||||
release unless the plugin package itself is also being released.
|
||||
|
||||
### Server / Python package release
|
||||
### Plugin / Python package release
|
||||
|
||||
Use this when Server behavior changes independently of Android app
|
||||
delivery, for example desktop channel support, bridge routes, pairing
|
||||
server fixes, voice auth, or packaging changes.
|
||||
Use this when plugin or relay behavior changes independently of Android app
|
||||
delivery, for example CLI channel support, bridge routes, pairing server fixes,
|
||||
voice auth, dashboard plugin UI, or packaging changes.
|
||||
|
||||
First **rewrite `PLUGIN_RELEASE_NOTES.md`** — it is the GitHub Release body for
|
||||
`plugin-v*` tags (the same role `RELEASE_NOTES.md` plays for Android). Fill the
|
||||
Summary and the Added/Changed/Fixed groups from the plugin-relevant bullets in the
|
||||
promoted `CHANGELOG.md` block, keep the `__VERSION__` token in the Install command
|
||||
(the workflow substitutes it), and apply the same public-distribution scrub as §2.
|
||||
|
||||
```bash
|
||||
git checkout dev
|
||||
git pull --ff-only origin dev
|
||||
|
||||
bash scripts/bump-server-version.sh 0.6.2
|
||||
git add pyproject.toml plugin/relay/__init__.py plugin/plugin.yaml plugin/dashboard/manifest.json plugin/dashboard/package.json plugin/dashboard/package-lock.json CHANGELOG.md
|
||||
git commit -m "release(server): server-v0.6.2"
|
||||
bash scripts/bump-plugin-version.sh 0.6.2
|
||||
git add pyproject.toml plugin/relay/__init__.py plugin/plugin.yaml plugin/dashboard/manifest.json plugin/dashboard/package.json plugin/dashboard/package-lock.json CHANGELOG.md PLUGIN_RELEASE_NOTES.md
|
||||
git commit -m "release(plugin): plugin-v0.6.2"
|
||||
git push origin dev
|
||||
|
||||
# Open the release PR (dev -> main) and merge with --no-ff.
|
||||
# After merge, tag from the new main tip:
|
||||
git checkout main
|
||||
git pull --ff-only origin main
|
||||
git tag server-v0.6.2
|
||||
git push origin server-v0.6.2
|
||||
git tag plugin-v0.6.2
|
||||
git push origin plugin-v0.6.2
|
||||
```
|
||||
|
||||
Pushing `server-v*` triggers `.github/workflows/release-server.yml`, which
|
||||
validates all server-owned version metadata with
|
||||
`scripts/check-server-version-sync.py`, runs server tests, builds a wheel and
|
||||
sdist, generates `SHA256SUMS.txt`, and creates a GitHub Release for the server
|
||||
package.
|
||||
Pushing `plugin-v*` triggers `.github/workflows/release-plugin.yml`, which
|
||||
validates all plugin-owned version metadata with
|
||||
`scripts/check-plugin-version-sync.py`. Run
|
||||
`python scripts/check-version-tracks.py` locally before tagging when a change
|
||||
touches more than one release surface. The workflow also runs plugin tests,
|
||||
builds a wheel and sdist, generates `SHA256SUMS.txt`, and creates a GitHub
|
||||
Release named `Hermes-Relay-Plugin v<version>` for the plugin package.
|
||||
|
||||
### 5. Upload to Play Console
|
||||
|
||||
**Manual upload (default):**
|
||||
> **If `PLAY_SERVICE_ACCOUNT_JSON` is configured as a repo secret, this step is
|
||||
> automated for stable tags.** The release workflow runs
|
||||
> `publishGooglePlayReleaseBundle --track=production` and the build appears as a
|
||||
> Production **draft** — skip to the Play Console, confirm the draft, and click
|
||||
> **Start rollout**. The manual path below is the fallback when the secret is
|
||||
> unset (or for staging on a non-production track).
|
||||
>
|
||||
> This automated tag path is intentionally bundle-only. It uploads the
|
||||
> `googlePlayRelease` AAB and release-scoped "What's new" notes, but it does
|
||||
> not republish static listing assets such as screenshots, title, description,
|
||||
> icon, or feature graphic. Use the Play Store Listing workflow when those
|
||||
> assets change.
|
||||
|
||||
**Pick the track first.** The AAB is track-agnostic — the same
|
||||
`-googlePlay-release.aab` goes to whichever track you publish on. Choose by intent,
|
||||
not habit:
|
||||
|
||||
- **Production** — the default for a stable GA release (`android-vX.Y.Z`). The
|
||||
listing is live, so this is where real releases land. The org account is
|
||||
D-U-N-S-verified, so the 14-day / 12-tester closed-testing gate does **not**
|
||||
apply — you can publish straight to Production.
|
||||
- **Open / Closed testing** — only when you actually want a public/private beta
|
||||
channel for this build.
|
||||
- **Internal testing** — only for a throwaway pre-release smoke check (e.g. a
|
||||
prerelease tag), not for a GA. Don't default here.
|
||||
|
||||
**Manual upload:**
|
||||
|
||||
1. Download the file ending in `-googlePlay-release.aab` from the GitHub
|
||||
Release assets (for example, `hermes-relay-0.3.0-googlePlay-release.aab`),
|
||||
Release assets (for example, `hermes-relay-1.0.0-googlePlay-release.aab`),
|
||||
or use your local build at
|
||||
`app\build\outputs\bundle\googlePlayRelease\hermes-relay-<version>-googlePlay-release.aab`.
|
||||
2. In Play Console: **Release > Testing > Internal testing** (the 14-day
|
||||
closed-testing rule does NOT apply to this account — see "Google Play
|
||||
Console developer account" above).
|
||||
2. In Play Console, open the track you chose above — for a GA that's
|
||||
**Release > Production**.
|
||||
3. **Create new release** > upload the AAB.
|
||||
4. Paste `RELEASE_NOTES.md` into the release notes field.
|
||||
5. **Review release** > **Start rollout.**
|
||||
4. Paste the Play "What's new" from `docs/play-store-listing.md` (≤500 chars) into
|
||||
the release notes field. (`RELEASE_NOTES.md` is the GitHub-Release body, not the
|
||||
Play field — don't paste that; it's over the limit.)
|
||||
5. **Review release** > **Start rollout** (set the staged-rollout percentage if you
|
||||
want a gradual production ramp).
|
||||
|
||||
**Automated upload (if `play-service-account.json` is configured):**
|
||||
|
||||
```bat
|
||||
scripts\dev.bat bundle
|
||||
gradlew publishReleaseBundle
|
||||
gradlew publishReleaseBundle --track=production
|
||||
```
|
||||
|
||||
Defaults to the `internal` track with `DRAFT` status (configured in the
|
||||
`play { }` block in `app/build.gradle.kts`). Override per-invocation with
|
||||
`--track=alpha` (= Closed testing), `--track=beta` (= Open testing), or
|
||||
`--track=production`.
|
||||
The `play { }` block in `app/build.gradle.kts` defaults to the `internal` track
|
||||
with `DRAFT` status as a safety net for unattended runs, so pass `--track` explicitly
|
||||
for a real release: `--track=production` (GA), or `--track=alpha` (Closed) /
|
||||
`--track=beta` (Open) for a beta channel.
|
||||
|
||||
To promote an existing release between tracks without rebuilding:
|
||||
|
||||
@@ -517,18 +593,24 @@ To promote an existing release between tracks without rebuilding:
|
||||
gradlew promoteReleaseArtifact --from-track=internal --promote-track=alpha
|
||||
```
|
||||
|
||||
### 6. Promote through tracks
|
||||
### 6. Tracks (a menu, not a mandatory ladder)
|
||||
|
||||
Typical path:
|
||||
The org account is exempt from the 14-day / 12-tester closed-testing rule, so a
|
||||
stable GA publishes **straight to Production** — there is no required promotion
|
||||
chain. The other tracks are opt-in tools, not steps you must climb:
|
||||
|
||||
1. **Internal testing** — personal smoke test (no tester or time minimum)
|
||||
2. **Closed testing (alpha)** — optional for staged rollout; Axiom-Labs'
|
||||
org account is exempt from the 14-day / 12-tester rule, so you can skip
|
||||
straight from Internal to Production if the build is ready
|
||||
3. **Open testing (beta)** — optional public beta
|
||||
4. **Production** — live on the Play Store
|
||||
- **Production** — live on the Play Store. Where GA releases go.
|
||||
- **Open testing (beta)** — opt-in public beta channel.
|
||||
- **Closed testing (alpha)** — opt-in private beta (named tester lists).
|
||||
- **Internal testing** — throwaway smoke check (e.g. a prerelease tag), no tester
|
||||
or time minimum.
|
||||
|
||||
Promote via the Play Console UI or `gradlew promoteReleaseArtifact`.
|
||||
If you *do* stage through tracks, promote an existing release without rebuilding via
|
||||
the Play Console UI or:
|
||||
|
||||
```bat
|
||||
gradlew promoteReleaseArtifact --from-track=internal --promote-track=production
|
||||
```
|
||||
|
||||
### 7. After release
|
||||
|
||||
@@ -549,9 +631,9 @@ Promote via the Play Console UI or `gradlew promoteReleaseArtifact`.
|
||||
|
||||
## CI Behavior
|
||||
|
||||
Android, Server, dashboard, and desktop now have separate CI/release lanes.
|
||||
Android, Plugin, dashboard, and desktop now have separate CI/release lanes.
|
||||
This keeps a dashboard CSS fix from running the full server suite, and keeps
|
||||
server changes from forcing an Android app `versionCode` bump.
|
||||
plugin changes from forcing an Android app `versionCode` bump.
|
||||
|
||||
On every push of a tag matching `android-v*`, `.github/workflows/release-android.yml`:
|
||||
|
||||
@@ -571,20 +653,25 @@ On every push of a tag matching `android-v*`, `.github/workflows/release-android
|
||||
succeeded. If `HERMES_KEYSTORE_BASE64` is missing, the summary warns
|
||||
that the artifacts are debug-signed and unsuitable for Play Store.
|
||||
|
||||
On every push of a tag matching `server-v*`,
|
||||
`.github/workflows/release-server.yml`:
|
||||
On every push of a tag matching `plugin-v*`,
|
||||
`.github/workflows/release-plugin.yml`:
|
||||
|
||||
1. Validates the tag matches all server-owned version metadata checked by
|
||||
`scripts/check-server-version-sync.py`.
|
||||
2. Runs server syntax checks and the focused route/auth/session test slice.
|
||||
1. Validates the tag matches all plugin-owned version metadata checked by
|
||||
`scripts/check-plugin-version-sync.py`.
|
||||
2. Runs plugin syntax checks and the focused route/auth/session test slice.
|
||||
3. Builds the Python wheel and sdist with `python -m build`.
|
||||
4. Generates `dist/SHA256SUMS.txt`.
|
||||
5. Creates a GitHub Release named `Hermes-Relay-Server v<version>` with the wheel,
|
||||
5. Creates a GitHub Release named `Hermes-Relay-Plugin v<version>` with the wheel,
|
||||
sdist, and checksum file attached.
|
||||
|
||||
On every push of a tag matching `desktop-v*`,
|
||||
`.github/workflows/release-desktop.yml` builds and publishes the desktop
|
||||
CLI binaries. Dashboard-only changes are covered by
|
||||
On every push of a tag matching `cli-v*`,
|
||||
`.github/workflows/release-cli.yml` builds and publishes the CLI binaries and
|
||||
Windows tray installer. Its GitHub Release body comes from `CLI_RELEASE_NOTES.md`
|
||||
(rewritten per release — the CLI counterpart of `RELEASE_NOTES.md`); the workflow
|
||||
substitutes `__VERSION__` (bare, e.g. `0.3.0`) and `__TAG__` (full, e.g.
|
||||
`cli-v0.3.0`) so the install/pin commands stay accurate. Fill its Summary and
|
||||
Added/Changed/Fixed groups at CLI release-prep and apply the §2 public scrub.
|
||||
Dashboard-only changes are covered by
|
||||
`.github/workflows/ci-dashboard.yml`, which builds the dashboard plugin,
|
||||
runs the dashboard API tests, and verifies the modal CSS markers are present
|
||||
in the built bundle.
|
||||
@@ -597,6 +684,13 @@ in the built bundle.
|
||||
| `HERMES_KEYSTORE_PASSWORD` | Store password | Password set during `keytool -genkey` |
|
||||
| `HERMES_KEY_ALIAS` | Key alias | Alias set during `keytool -genkey` |
|
||||
| `HERMES_KEY_PASSWORD` | Key password | Usually the same as the store password |
|
||||
| `PLAY_SERVICE_ACCOUNT_JSON` | **Optional** — Play auto-upload | Paste the full Play Developer API service-account JSON (step 3) |
|
||||
|
||||
If `PLAY_SERVICE_ACCOUNT_JSON` is set, the `android-v*` release workflow uploads
|
||||
the `googlePlay` AAB to the **Production track as a DRAFT** automatically (stable
|
||||
tags only — prereleases are skipped). CI does the upload; you still click **Start
|
||||
rollout** in Play Console. If the secret is unset, the workflow skips the upload
|
||||
and you upload manually (§5) — nothing else changes.
|
||||
|
||||
## Hotfix Recipe
|
||||
|
||||
@@ -622,9 +716,9 @@ For an Android app hotfix:
|
||||
`dev`'s `appVersionCode` lags behind `main` and the next app release
|
||||
bump collides.
|
||||
|
||||
For a Server hotfix, branch from the affected `server-v*` tag, apply
|
||||
the fix, run `bash scripts/bump-server-version.sh <next-version>`, merge to
|
||||
`main`, and tag `server-v<next-version>`. Do not touch
|
||||
For a Plugin hotfix, branch from the affected `plugin-v*` tag, apply
|
||||
the fix, run `bash scripts/bump-plugin-version.sh <next-version>`, merge to
|
||||
`main`, and tag `plugin-v<next-version>`. Do not touch
|
||||
`gradle/libs.versions.toml` unless an Android app release is also shipping.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
@@ -1,22 +1,22 @@
|
||||
# Hermes-Relay-Android v1.0.0
|
||||
# Hermes-Relay-Android v1.2.0
|
||||
|
||||
**Release Date:** June 14, 2026
|
||||
**Since v0.8.1:** The 1.0 milestone — a rechromed app, a first-class standard (no-plugin) path, live-thinking gateway chat, and a broad polish pass.
|
||||
**Release Date:** June 20, 2026
|
||||
**Since v1.1.0:** A big personalization release — app themes, swappable sphere skins, and animated agent **pets** — paired with a transparency pass (see which transport you're on and exactly what the agent is told), a much faster cold start, in-app crash reporting, and a broad reliability sweep.
|
||||
|
||||
v1.0.0 is the first stable release. The headline is that a **plain, unmodified Hermes agent is now enough**: chat, Manage, and voice all work against vanilla upstream with no relay plugin. The relay plugin is now purely additive (phone control, terminal, notification companion, extra voice engines).
|
||||
v1.2.0 is about making Hermes-Relay feel like *yours* and making it honest about what it's doing. Dress the app in one of eight themes, swap the agent orb for a hand-picked or AI-generated **pet** that reacts to what the agent is doing, and give each profile its own icon. At the same time, the chat status strip now names the actual streaming path (⚡ Gateway, 📡 Sessions, …), a "What the agent sees" sheet shows the exact extra context prepended to your next turn, and cold start is roughly three times faster. If something does go wrong, the app now catches the crash and offers a one-tap, pre-filled bug report.
|
||||
|
||||
---
|
||||
|
||||
## Download
|
||||
|
||||
v1.0.0 ships in two Android build flavors. APK and AAB filenames are version-tagged:
|
||||
v1.2.0 ships in two Android build flavors. APK and AAB filenames are version-tagged:
|
||||
|
||||
| Flavor | File | Who it's for |
|
||||
|---|---|---|
|
||||
| Google Play | `hermes-relay-1.0.0-googlePlay-release.aab` | Upload this Android App Bundle to Play Console. It has no AccessibilityService, screen reading, screenshots, gestures, SMS/calls, contacts/location, overlays, or unattended phone control. |
|
||||
| sideload | `hermes-relay-1.0.0-sideload-release.apk` | Direct-install APK for full Device Control. Installs as `com.axiomlabs.hermesrelay.sideload`. |
|
||||
| googlePlay APK | `hermes-relay-1.0.0-googlePlay-release.apk` | Parity/testing artifact. |
|
||||
| sideload AAB | `hermes-relay-1.0.0-sideload-release.aab` | Parity/testing artifact. |
|
||||
| Google Play | `hermes-relay-1.2.0-googlePlay-release.aab` | Upload this Android App Bundle to Play Console. It has no AccessibilityService, screen reading, screenshots, gestures, SMS/calls, contacts/location, overlays, or unattended phone control. |
|
||||
| sideload | `hermes-relay-1.2.0-sideload-release.apk` | Direct-install APK for full Device Control. Installs as `com.axiomlabs.hermesrelay.sideload`. |
|
||||
| googlePlay APK | `hermes-relay-1.2.0-googlePlay-release.apk` | Parity/testing artifact. |
|
||||
| sideload AAB | `hermes-relay-1.2.0-sideload-release.aab` | Parity/testing artifact. |
|
||||
|
||||
Verify integrity with `SHA256SUMS.txt` from the same release. See the [Sideload guide](https://codename-11.github.io/hermes-relay/guide/getting-started.html#sideload-apk) for APK install steps.
|
||||
|
||||
@@ -24,43 +24,43 @@ Verify integrity with `SHA256SUMS.txt` from the same release. See the [Sideload
|
||||
|
||||
## Highlights
|
||||
|
||||
### Standard path is first-class — no plugin required
|
||||
### Make it yours
|
||||
|
||||
Chat, Manage, and voice now work against an unmodified upstream Hermes agent. Chat streams over the API server; Manage and voice use the Hermes dashboard with a single sign-in. The relay plugin stays optional and only adds power tools.
|
||||
- **App themes.** A theme picker in Settings → Appearance ships eight looks — the signature Hermes Relay brand (full light/dark) plus ports of the Nous Hermes baselines: Hermes Teal, Nous Blue, Midnight, Ember, Mono, Cyberpunk, and Rosé. The whole app follows your choice; Light/Dark/Auto applies to themes that ship both modes.
|
||||
- **Agent pets — a living avatar.** Replace the orb with an animated pet that reacts to the agent: idle / thinking / writing / speaking / listening, a distinct **working** pose during tool calls, one-shot **greet** and **celebrate** reactions, and a loop that speeds up as output streams. Add or remove pets right in Appearance (no `adb`), preview each state, tune playback speed, and toggle frame auto-stabilization. Pets are pure data — an AI authoring kit and JSON schema let you generate one from sprite art.
|
||||
- **Hot-swappable sphere skins + per-profile icons.** Keep the orb but reskin it (Adaptive, Classic, Aurora, Solar, Mono, or your own JSON skin), and give each agent profile its own small icon beside its name — all client-side, never sent to Hermes.
|
||||
|
||||
### Gateway chat transport with live thinking
|
||||
### See what's actually happening
|
||||
|
||||
Chat can now ride the upstream dashboard `/api/ws` gateway (the same surface the official hermes-desktop client speaks). It's the only vanilla-upstream path that streams reasoning **live**, so the Thinking block and sphere light up *during* generation instead of after. "Auto" prefers the gateway when the dashboard is reachable and Manage is signed in, and falls back to the SSE endpoints per turn on any failure.
|
||||
- **Transport path is visible.** The chat status strip now shows which streaming path is in use — ⚡ Gateway (live thinking), 📡 Sessions, Completions, or Runs — and Chat Settings adds a basic→best tier ladder explaining the active path and its fallback.
|
||||
- **"What the agent sees" sheet.** Tap the context meter to see the exact extra context prepended to your next turn — persona/profile, phone status, any per-turn voice hint, and (when paired) the relay's own server-side context. The audit is honest about what the phone sends versus what's applied on the server.
|
||||
- **Spoken-turn badges + voice render-path visibility.** Voice and Realtime Agent replies carry a chip in the scrollback, and Voice Settings shows whether speech is rendering over the streaming or basic path.
|
||||
|
||||
- **Warm-start + keep-alive.** The app pre-warms the gateway on foreground so the first token lands fast (the cold session-setup cost moves off the send path). An opt-in **Keep connected in background** toggle (both flavors) holds the connection open via a foreground service so a long-backgrounded conversation resumes instantly.
|
||||
- **Attachments at desktop parity.** Images, PDFs, and any other file upload natively over the gateway (`image.attach_bytes` / `pdf.attach` / `file.attach`). Turns that fall back to an endpoint that can't carry a file now post a visible notice instead of dropping it silently.
|
||||
- **Steering, edit & resend, subagent lanes.** Send mid-turn to inject guidance into the running turn; edit your own messages to rewind and regenerate; watch per-task subagent lanes stream under the bubble; a context-window meter warns as the window fills.
|
||||
- **Turn-complete notifications** when the app is backgrounded.
|
||||
### Privacy
|
||||
|
||||
### Manage parity with the desktop dashboard
|
||||
- **Sensitive-media blur.** When paired to the relay, the agent can mark private/NSFW media and the phone blurs it per your setting — sensitivity stays model-emitted (no on-device or relay-side classifier), and the exact instruction is visible in the "What the agent sees" sheet. Vanilla Hermes (no plugin) is unaffected.
|
||||
|
||||
The Manage tab now does what the desktop dashboard does: change models from the full provider catalog, manage provider keys (write-only, masked, reveal), create/edit profiles and SOUL.md, and browse/install/update skills. Manage data is cached to disk so a cold launch renders instantly.
|
||||
### Faster, calmer, more honest
|
||||
|
||||
### Per-conversation agent profiles
|
||||
- **~3× faster cold start.** The app was building several hardware-keystore-encrypted stores at launch, serializing on a process-global lock and stalling the chat header for seconds. It now builds a single keyset shared with the dashboard cookies, cutting measured time-to-connected from ~2.9 s to ~1 s. Existing sign-ins migrate automatically.
|
||||
- **Honest loading, never stale.** Model, personality, and approvals show a brief "checking…" state and fade in once the server confirms them; standard controls (Model, YOLO, Fast, reasoning effort) always appear — live when ready, "checking…" while loading, or cleanly disabled with the reason — instead of being hidden or showing a maybe-wrong value.
|
||||
|
||||
Switch the whole agent — model, persona (SOUL), and skills — from the chat header. The selection is **ephemeral and per-conversation** (bound to the session, like the official desktop): it never changes your server's default agent for other clients. The session drawer scopes to the active profile, opening one of its chats loads that profile's history, and the right agent is restored on cold start.
|
||||
### More reliable
|
||||
|
||||
### Redesigned chat input + seamless connection UX
|
||||
- **In-app crash reporting.** A force-close now surfaces a clean dialog on next launch with the stack trace — Copy it, or **Report** to open a pre-filled GitHub issue from the bug template. The handler re-raises so Play vitals still record the crash.
|
||||
- **QR pairing hardened for foldables.** On devices where the camera can't initialize, the scanner shows a "camera unavailable — pair manually" card instead of force-closing.
|
||||
- **Crash fixes.** No more crash opening a chat with a server-local image, and the PDF viewer no longer crashes when a document closes mid-render.
|
||||
- **Chat correctness.** In-chat model picks now actually apply — both on a new chat and mid-conversation — server-side turn errors stay on screen as an error bubble, per-reply token counts and provenance badges survive the post-turn reload, and server steering markers (`[System: …]`) no longer appear as chat bubbles.
|
||||
|
||||
A cleaner Telegram-style input bar (pill field, one morphing Send/Voice/Stop/Steer/Queue button, no slash button). Network route handoffs (LAN↔Tailscale) and reconnects no longer repaint or reload the chat, and connection/update status now slide down as in-theme toasts over the content instead of pushing the UI around.
|
||||
### Voice & terminal polish
|
||||
|
||||
### Voice
|
||||
|
||||
The provider-native Realtime Agent keeps one session open across turns (follow-ups retain context), and long Hermes runs are promoted to tracked background tasks so the conversation stays responsive and the answer is spoken when it's ready.
|
||||
|
||||
### Docs + branding
|
||||
|
||||
The documentation site was rechromed to the app's cockpit theme and repositioned around the two-path story (just connect → give it hands), with a reworked Android getting-started funnel and a Google Play badge.
|
||||
- **Enhanced voice control (Gemini & xAI).** When the relay uses a Gemini or xAI voice provider, Voice Settings can pick a voice/model and turn on expressive tone/speech tags. Vanilla Hermes voice stays configured server-side.
|
||||
- **Leaner terminal.** A scrollable, fully-legible key bar, TUI-correct arrows and bracketed paste, a compact single-row header, and relay sessions on an isolated, TUI-tuned tmux so editors and full-screen tools behave.
|
||||
|
||||
---
|
||||
|
||||
## Upgrade notes
|
||||
|
||||
- **Google Play submission:** the opt-in keep-alive feature adds a `FOREGROUND_SERVICE_SPECIAL_USE` service. Complete the Play Console **Foreground service permissions** declaration for `specialUse` at submission (see `docs/play-store-listing.md`).
|
||||
- **PDF attachments** over the gateway require `poppler-utils` (`pdftoppm`) on the Hermes host; without it, PDF attach reports an error and the message still sends as text.
|
||||
- `appVersionCode` is **12**.
|
||||
- Cold-start speedup migrates the encrypted credential and dashboard-cookie stores automatically on first launch; in rare cases Manage/voice may ask for a one-time re-login (cookies are re-obtainable).
|
||||
- App themes, sphere skins, and pets are available on **both** flavors — they're client-side and need no Device Control.
|
||||
- `appVersionCode` is **14**.
|
||||
|
||||
@@ -14,7 +14,7 @@ Native Android companion for the [Hermes agent platform](https://github.com/Nous
|
||||
|
||||
### Desktop track (parallel lane to Android) — **experimental**
|
||||
|
||||
Release tags: `desktop-v*` (separate cadence from Android `android-v*` and Server `server-v*`). Curl-installed prebuilt binaries (no Node required); Windows first, macOS / Linux same release. Workflows: [`ci-desktop.yml`](.github/workflows/ci-desktop.yml) + [`release-desktop.yml`](.github/workflows/release-desktop.yml).
|
||||
Release tags: `cli-v*` (separate cadence from Android `android-v*` and Plugin `plugin-v*`). Historical alpha prereleases used `desktop-v*`, and the installer/updater keep a migration fallback. Curl-installed prebuilt binaries (no Node required); Windows first, macOS / Linux same release. Workflows: [`ci-desktop.yml`](.github/workflows/ci-desktop.yml) + [`release-cli.yml`](.github/workflows/release-cli.yml).
|
||||
|
||||
**Shipped (2026-04-23 — first tagged release `desktop-v0.3.0-alpha.1`):**
|
||||
|
||||
@@ -47,11 +47,11 @@ Release tags: `desktop-v*` (separate cadence from Android `android-v*` and Serve
|
||||
|
||||
**Earlier alpha.2–alpha.5 workstreams (now in-flight / done — see DEVLOG 2026-04-23 entries for specifics):**
|
||||
|
||||
- **`hermes-relay update` subcommand + auto-update nudge.** The binary today does NOT self-update — users have to re-run the `curl | sh` / `irm | iex` one-liner to pick up a new release. Close the gap: `hermes-relay update` polls the GitHub Releases API, filters to `desktop-v*`, compares to `readVersion()`, and either shells out to the installer or downloads the binary directly + `rename` over the current one (Windows can rename while running; Linux/macOS atomic replace is fine for long-lived daemons because the running process keeps the old inode open). Add a once-per-day background check in `daemon` mode that emits `update_available` as a log event — opt-in via `--check-updates`, never auto-installs without user action. Signing prerequisite: SmartScreen/Gatekeeper would warn on every auto-downloaded binary until we sign, so this is behind code signing.
|
||||
- **`hermes-relay update` subcommand + auto-update nudge.** The binary self-update path polls the GitHub Releases API, prefers `cli-v*`, falls back to historical `desktop-v*` prereleases during migration, compares to `readVersion()`, and downloads the binary directly + `rename` over the current one (Windows can rename while running; Linux/macOS atomic replace is fine for long-lived daemons because the running process keeps the old inode open). Add a once-per-day background check in `daemon` mode that emits `update_available` as a log event — opt-in via `--check-updates`, never auto-installs without user action. Signing prerequisite: SmartScreen/Gatekeeper would warn on every auto-downloaded binary until we sign, so this is behind code signing.
|
||||
- **Workspace-awareness — desktop client sends cwd/git/hostname on connect.** Biggest lingering "is the agent working against the right tree?" problem. On WSS auth, the client advertises an ephemeral workspace descriptor — `cwd`, `git_root`, `git_branch`, `git_status_summary` (staged/modified counts), `repo_name`, `hostname`, `platform`, `active_shell`. Server-side `DesktopHandler` stashes it as live session metadata (NOT persistent state). New hermes-agent plugin hook injects a one-line ephemeral prompt prefix into the session context — *"Active desktop workspace: machine=Bailey-PC · repo=hermes-relay · branch=dev · staged=3"* — so the LLM reads it every turn without the operator having to explain. Also default `desktop_terminal` / `desktop_read_file` / `desktop_search_files` `cwd` to the repo root when unset. Expose the snapshot in `hermes-relay doctor` + `hermes-relay status` + a new `hermes-relay workspace` subcommand + a relay dashboard tab so both operator and agent have a common view. Pair with a `.hermes/workspace-context.json` file-based fallback for when the socket path can't be reached. Requires: new WSS envelope (`desktop.workspace` on connect), hermes-agent plugin hook for ephemeral context injection, schema coordination with the upstream `ContextVar` multi-client work.
|
||||
- **Service installers** — `scripts/install-service-{win,linux,mac}.{ps1,sh}` — Windows Service via `sc.exe create`, `systemd --user` unit with `loginctl enable-linger`, `launchctl load` plist for macOS. Auto-start on login so the daemon is always reachable.
|
||||
- **Multi-client routing on the `desktop` channel** — replace single-client MVP with per-token indexing + device-id reconnect handoff. Hermes session state carries `desktop_session_token` via a new `ContextVar` in `gateway/session_context.py` (hermes-agent PR candidate — won't affect Android). Natural pairing with the workspace-awareness envelope — the ContextVar scheme determines which client's workspace the active session sees.
|
||||
- **Harden `release-desktop.yml` retag semantics.** The `softprops/action-gh-release` step failed during the alpha.1 retag with `tag_name already_exists` after deleting + re-uploading all 5 assets; recovered by `gh api` cleanup (delete orphan draft + PATCH draft→false on the release with the real assets). Follow-up: pin the action version, add `make_latest: false` + explicit `release_id` lookup, or switch to `ncipollo/release-action` which handles retags without the duplicate-draft creation.
|
||||
- **Harden `release-cli.yml` retag semantics.** The `softprops/action-gh-release` step failed during the alpha.1 retag with `tag_name already_exists` after deleting + re-uploading all 5 assets; recovered by `gh api` cleanup (delete orphan draft + PATCH draft→false on the release with the real assets). Follow-up: pin the action version, add `make_latest: false` + explicit `release_id` lookup, or switch to `ncipollo/release-action` which handles retags without the duplicate-draft creation.
|
||||
- **Signed binaries** — Windows EV code-signing (~$300/yr, DigiCert or SSL.com) + Apple Developer ID + notarization ($99/yr). Removes SmartScreen/Gatekeeper warnings. Prerequisite for the auto-update path.
|
||||
- **npm registry publication** — future v1.0 distribution work. The package name is local workspace metadata today; current install paths are GitHub Release binaries or local clone + `npm link`.
|
||||
- **HMAC verification on QR payloads** — defer until a client-accessible secret story exists (same deferral as the Android app). Not blocking GA.
|
||||
|
||||
@@ -6,60 +6,118 @@ For shipped work, see `DEVLOG.md`. For architectural decisions, see `docs/decisi
|
||||
|
||||
---
|
||||
|
||||
## User-Added:
|
||||
|
||||
- [ ] Enhance the 'clean chat' view mode to allow more a little more vertical visible text area and scrolling within.
|
||||
- [ ] Look into the voice-settings profile specific capabilities - confirm approach is sound - verify as I noticed that in 'auto' mode it didn't work, it still used the system default despite config despite override voice chosen being displayed to user in voice config in voice setting in app UI. Only switching to 'Relay' specifically allowed the user-override to work/apply.
|
||||
|
||||
- [ ] - analytics and diagnostics pages need cleaned up, improved, enhancements for UI/UX/layout. Diagnostics should have timeline vertical status checks with failure reason etc
|
||||
|
||||
- [x] **Per-profile agent icon + static-image avatar (shipped 2026-06-20 — `d827e46`, see DEVLOG).** Per-profile icon: client-side `ProfileIconStore` (per `(connection, profile)`, never sent to Hermes; stores a copied-file path) → small Coil image beside the agent name in `MessageBubble` via `LocalAgentIconPath`; picker is `AgentIconRow` under the local-name row in `ConnectionInfoSheet`. Static image: "Add a pet" accepts a single image (magic-byte detect → one-frame static pet). Scope shipped: small name-adjacent icon only; big avatar stays global. Follow-ups: on-device smoke (import an image as a pet; set a profile icon, confirm it shows by the name + persists across restart); optionally also show the icon in the profile picker.
|
||||
|
||||
## Hands-free agentic voice backlog
|
||||
|
||||
Goal: make Hermes usable for hands-free work without leaving the operator blind
|
||||
|
||||
to tool state, safety prompts, or the current task.
|
||||
|
||||
- **Waveform output-start sync** — current input waveform timing feels good, but
|
||||
the agent-output waveform can unfold and begin movement before audible speech
|
||||
starts. Split "preparing audio" from "speaking audio" in the visual layer, or
|
||||
gate the unfolded Speaking waveform on the first real playback frame/audio
|
||||
amplitude. Processing can stay as the folded circular spinner until output is
|
||||
actually audible.
|
||||
|
||||
the agent-output waveform can unfold and begin movement before audible speech
|
||||
|
||||
starts. Split "preparing audio" from "speaking audio" in the visual layer, or
|
||||
|
||||
gate the unfolded Speaking waveform on the first real playback frame/audio
|
||||
|
||||
amplitude. Processing can stay as the folded circular spinner until output is
|
||||
|
||||
actually audible.
|
||||
|
||||
- **Voice command layer** — reserve local commands that bypass normal agent
|
||||
routing: "pause", "resume", "stop talking", "cancel", "repeat that", "open
|
||||
overlay", "return to Hermes", and "new chat". These should work while the
|
||||
agent is thinking, speaking, or using tools.
|
||||
|
||||
routing: "pause", "resume", "stop talking", "cancel", "repeat that", "open
|
||||
|
||||
overlay", "return to Hermes", and "new chat". These should work while the
|
||||
|
||||
agent is thinking, speaking, or using tools.
|
||||
|
||||
- **Spoken tool progress** — when Hermes uses tools, voice mode should speak
|
||||
short status updates such as "I'm checking the relay logs" or "I found an
|
||||
error" without waiting for final assistant text. Long tool calls should emit
|
||||
periodic, low-noise progress updates.
|
||||
|
||||
short status updates such as "I'm checking the relay logs" or "I found an
|
||||
|
||||
error" without waiting for final assistant text. Long tool calls should emit
|
||||
|
||||
periodic, low-noise progress updates.
|
||||
|
||||
- **Realtime tool timeline parity** — the voice overlay should render the same
|
||||
live thinking blocks, streaming assistant text, and tool call progress as the
|
||||
normal chat surface without requiring exit/reload.
|
||||
|
||||
live thinking blocks, streaming assistant text, and tool call progress as the
|
||||
|
||||
normal chat surface without requiring exit/reload.
|
||||
|
||||
- **Hands-free confirmation flow** — risky actions need first-class spoken and
|
||||
visual confirmation: "yes", "no", "cancel", "confirm", plus a visible and
|
||||
audible countdown for destructive actions.
|
||||
|
||||
visual confirmation: "yes", "no", "cancel", "confirm", plus a visible and
|
||||
|
||||
audible countdown for destructive actions.
|
||||
|
||||
- **Voice session memory/status** — add a compact "where are we?" summary for
|
||||
the current voice task: active objective, last tool result, pending next step,
|
||||
and whether the agent is waiting on the user.
|
||||
|
||||
the current voice task: active objective, last tool result, pending next step,
|
||||
|
||||
and whether the agent is waiting on the user.
|
||||
|
||||
- **Mode presets** — add presets such as Hands-free, Low latency, Careful tool
|
||||
mode, and Quiet/visual-only. Hands-free should favor Continuous listening,
|
||||
spoken tool progress, confirmations, and overlay availability.
|
||||
|
||||
mode, and Quiet/visual-only. Hands-free should favor Continuous listening,
|
||||
|
||||
spoken tool progress, confirmations, and overlay availability.
|
||||
|
||||
- **Barge-in hardening** — keep barge-in experimental until echo/self-recording
|
||||
is solved. The target path is proper AEC, playback-ducking, and a rule that
|
||||
output audio can never become a user turn.
|
||||
|
||||
is solved. The target path is proper AEC, playback-ducking, and a rule that
|
||||
|
||||
output audio can never become a user turn.
|
||||
|
||||
- **Audio quality guardrails** — normalize output volume across realtime and
|
||||
fallback TTS providers, keep pronunciation hints/profile voice tuning, and
|
||||
measure provider-specific delay, chunk gaps, and tail clipping.
|
||||
|
||||
fallback TTS providers, keep pronunciation hints/profile voice tuning, and
|
||||
|
||||
measure provider-specific delay, chunk gaps, and tail clipping.
|
||||
|
||||
- **Pluggable Realtime Agent media transports** — add an OpenAI-first WebRTC
|
||||
transport option for Realtime Agent so mobile audio can use provider-native
|
||||
jitter buffering, interruption, and media handling instead of only relay
|
||||
WebSocket PCM. Design this as a provider transport interface
|
||||
(`websocket`, `webrtc`, future `livekit`/SIP-style bridges) so other
|
||||
realtime providers can opt in without forking the Hermes broker/tool
|
||||
contract. Hermes must still own tools, memory, confirmations, current data,
|
||||
and durable transcript state.
|
||||
|
||||
transport option for Realtime Agent so mobile audio can use provider-native
|
||||
|
||||
jitter buffering, interruption, and media handling instead of only relay
|
||||
|
||||
WebSocket PCM. Design this as a provider transport interface
|
||||
|
||||
(`websocket`, `webrtc`, future `livekit`/SIP-style bridges) so other
|
||||
|
||||
realtime providers can opt in without forking the Hermes broker/tool
|
||||
|
||||
contract. Hermes must still own tools, memory, confirmations, current data,
|
||||
|
||||
and durable transcript state.
|
||||
|
||||
- **Voice engine selector** — implemented as an opt-in experimental Realtime
|
||||
Agent engine in `docs/plans/2026-05-19-realtime-hermes-voice-agent.md`.
|
||||
Follow-up work is provider-native turn-taking, richer confirmation handling,
|
||||
and quality/latency evaluation before promotion beyond Experimental.
|
||||
|
||||
Agent engine in `docs/plans/2026-05-19-realtime-hermes-voice-agent.md`.
|
||||
|
||||
Follow-up work is provider-native turn-taking, richer confirmation handling,
|
||||
|
||||
and quality/latency evaluation before promotion beyond Experimental.
|
||||
|
||||
- **Realtime-native Hermes bridge prototype** — first relay-brokered slice
|
||||
implemented in `docs/plans/2026-05-19-realtime-hermes-voice-agent.md`.
|
||||
Remaining work: let OpenAI/xAI realtime sessions own more of the live speech
|
||||
turn while still proxying every tool, confirmation, memory, and Android bridge
|
||||
action through Hermes/relay safety.
|
||||
|
||||
implemented in `docs/plans/2026-05-19-realtime-hermes-voice-agent.md`.
|
||||
|
||||
Remaining work: let OpenAI/xAI realtime sessions own more of the live speech
|
||||
|
||||
turn while still proxying every tool, confirmation, memory, and Android bridge
|
||||
|
||||
action through Hermes/relay safety.
|
||||
|
||||
---
|
||||
|
||||
@@ -77,7 +135,7 @@ Things to look into:
|
||||
- **Skill distribution as separate from plugin distribution** — right now skills ride along with the plugin install via `external_dirs`. Should skills be installable independently (e.g. `hermes skill install <git-url>`)? Would that fragment maintenance or improve reuse?
|
||||
- **Tool registration discoverability** — `android_*` tools register at gateway import time. There's no canonical "list installed plugin tools" API. Would adding one to upstream make sense, or is `gateway tool list` already enough?
|
||||
- **Versioning + compatibility ranges** — `pip install -e` doesn't enforce version pins between hermes-agent and our plugin. A breaking change in upstream's plugin loader could silently break us. Do we need a `hermes_compat: ">=0.8.0,<1.0.0"` field somewhere?
|
||||
- **`hermes-relay-self-setup` SKILL.md as a precedent** — we just shipped a self-installing skill that an LLM can fetch from a raw GitHub URL and execute. Does this pattern generalize? Could it become a recommended way for any third-party Hermes project to ship setup automation?
|
||||
- `**hermes-relay-self-setup` SKILL.md as a precedent** — we just shipped a self-installing skill that an LLM can fetch from a raw GitHub URL and execute. Does this pattern generalize? Could it become a recommended way for any third-party Hermes project to ship setup automation?
|
||||
- **Bootstrap injection** — `hermes_relay_bootstrap/` monkey-patches `aiohttp.web.Application` to inject endpoints into vanilla upstream. This is intentional but feels like a hack. Upstream PR #8556 (`feat/session-api`) will eventually let us delete it — verified 2026-04-15 that its scope covers the full bootstrap surface (sessions, memory, skills, config, available-models). Track that PR's status periodically.
|
||||
- **Gateway slash-command preprocessor — upstream Stage 1 PR.** Sibling follow-up to #8556. Intercepts known gateway commands on `/v1/runs` + `/v1/chat/completions`, dispatches the stateless ones (`/help`, `/commands`) via `gateway_help_lines()`, returns a deterministic "use a channel with session state" notice for the stateful majority. Currently being prepared in `C:/Users/Bailey/Desktop/Open-Projects/hermes-agent-pr-prep/` on branch `feat/api-server-gateway-commands`; awaiting subagent's code + draft PR body before pushing. See `docs/upstream-contributions.md` §5.
|
||||
- **Gateway slash-command preprocessor — bootstrap middleware (Stage 1 equivalent).** Sibling shim in `hermes_relay_bootstrap/_command_middleware.py` that mirrors the upstream Stage 1 PR as an aiohttp middleware injected at bootstrap time. Ships the hallucination fix to vanilla-upstream installs before the upstream PR lands. Planned for v0.4.1, after the current bridge feature branch wraps. See `ROADMAP.md` v0.4.1 entry.
|
||||
@@ -94,6 +152,64 @@ When the answer becomes clearer, this section becomes either an ADR in `docs/dec
|
||||
- **Wave 3 voice-bridge multi-turn confirmation** — currently a 5s TTS countdown with cancel; conversational confirmation is the follow-up
|
||||
- **LLM client wiring for `android_navigate`** — `_default_vision_model` is stubbed; production swap to a real Anthropic/OpenAI vision client
|
||||
- **Real screenshots of each flavor's a11y permission dialog** — for `user-docs/guide/release-tracks.md`
|
||||
- **`llms.txt` standard** — explicitly skipped in favor of the `hermes-relay-self-setup` SKILL.md path; revisit if the standard gains traction in the agent ecosystem
|
||||
- **`markdown-renderer` 0.40.x API update** — pinned at `0.30.0` in `gradle/libs.versions.toml` because 0.40.2 introduced breaking API changes that `app/src/main/kotlin/com/hermesandroid/relay/ui/components/MarkdownContent.kt` hasn't been updated for. Specifically: `markdownColor()` drops `codeText`/`linkText`, `MarkdownCodeBlock`/`MarkdownCodeFence` inner lambdas now take a 3rd `TextStyle` arg, and `MarkdownHighlightedCode`'s 3rd param is now `TextStyle` instead of `Highlights.Builder`. Dependabot auto-merged the bump on 2026-04-13 which silently broke CI; reverted for the v0.3.0 release. Update requires reading the new library API docs and testing in Studio — not a blind fix. Consider adding a dependabot ignore rule for `markdown-renderer` major bumps until this is handled.
|
||||
- `**llms.txt` standard** — explicitly skipped in favor of the `hermes-relay-self-setup` SKILL.md path; revisit if the standard gains traction in the agent ecosystem
|
||||
- `**markdown-renderer` 0.40.x API update** — pinned at `0.30.0` in `gradle/libs.versions.toml` because 0.40.2 introduced breaking API changes that `app/src/main/kotlin/com/hermesandroid/relay/ui/components/MarkdownContent.kt` hasn't been updated for. Specifically: `markdownColor()` drops `codeText`/`linkText`, `MarkdownCodeBlock`/`MarkdownCodeFence` inner lambdas now take a 3rd `TextStyle` arg, and `MarkdownHighlightedCode`'s 3rd param is now `TextStyle` instead of `Highlights.Builder`. Dependabot auto-merged the bump on 2026-04-13 which silently broke CI; reverted for the v0.3.0 release. Update requires reading the new library API docs and testing in Studio — not a blind fix. Consider adding a dependabot ignore rule for `markdown-renderer` major bumps until this is handled.
|
||||
- **Dependabot auto-merge guardrails** — Dependabot merged breaking bumps despite CI failing. Investigate why `.github/workflows/dependabot-auto-merge.yml` isn't gating on CI status, and consider adding an ignore rule for packages we know need manual attention on major bumps (`markdown-renderer`, compose BOM, activity-compose).
|
||||
|
||||
---
|
||||
|
||||
## Crash reporting + foldable hardening (shipped 2026-06-20)
|
||||
|
||||
Triggered by a Play Store review: app "keeps crashing" during setup on a Samsung Galaxy Z Fold7 (Android 16 / SDK 36, version code 13). Shipped: in-app crash capture (`util/CrashReporter.kt` — uncaught handler that persists a report then re-raises so Play vitals still collects; `ui/components/CrashReportDialog.kt` — show-once dialog with Copy + pre-filled GitHub-issue "Report"); QR camera-init hardening (`QrPairingScanner.kt` — try/catch around `ProcessCameraProvider.get()` and `InputImage.fromMediaImage()`, graceful `CameraUnavailableCard` → manual pairing instead of force-close).
|
||||
|
||||
Follow-ups:
|
||||
|
||||
- **Confirm the actual crash from Play vitals.** Pull the top crash cluster for Galaxy Z Fold7 / version code 13 (Quality → Android vitals → Crashes & ANRs) to verify the camera path is the real cause vs. another setup-path throw. The hardening is correct regardless, but the trace closes the loop.
|
||||
- **Portrait lock is moot on large screens under SDK 36.** `android:screenOrientation="portrait"` is largely ignored by Android 16's mandatory large-screen orientation override on foldables/tablets. Decide whether to keep the lock (it still applies on phones) or make it conditional; either way it does not *cause* the crash.
|
||||
- **Foldable camera lifecycle races (from the 2026-06-20 audit, not yet fixed).** `QrPairingScanner` can still hit bind/unbind races on rapid fold/unfold recomposition (the `DisposableEffect` `unbindAll()` vs. an in-flight `addListener` bind), and `mapBoxToViewport` runs on possibly-stale `viewportSizePx` during a fold transition. Not crash-fatal after the try/catch hardening (logged + skipped), but worth a fold-aware guard if foldable adoption grows.
|
||||
- **Optional: surface crash history in Settings.** The reporter keeps only the most recent crash (`files/crash/last-crash.json`, consumed on view). If repeat-crash diagnosis becomes common, keep a small ring of recent reports + a Settings entry to view/copy them.
|
||||
|
||||
---
|
||||
|
||||
## Relay enhancement layer + agent-context injection (shipped 2026-06-20 — `docs/plans/2026-06-20-relay-enhancement-layer.md`)
|
||||
|
||||
Shipped: `plugin/enhancements/` (registry + fail-open `context_injection` wrap of `AIAgent._build_system_prompt`), the `media-sensitivity` block, `GET /context/injected` audit route, dashboard toggles, client sensitivity re-thread + "Relay context (server-side)" audit section, and the transport-path UI (`ChatTransportStatusBadge` / `RelayStatusStrip` + tier ladder). OFF by default, removable, vanilla-safe.
|
||||
|
||||
Follow-ups:
|
||||
|
||||
- **Confirm the `AIAgent` seam on the live host before relying on it.** `context_injection._resolve_ai_agent_class()` tries `agent.system_prompt` / `run_agent`. When you flip `RELAY_AGENT_CONTEXT_ENABLED=1`, verify `GET /context/injected` shows the block AND that it actually lands in the prompt (the wrap is fail-open, so a wrong module = inert, not broken). If the class lives elsewhere, widen the module list.
|
||||
- **Retire the monkey-patch when upstream adds a plugin context hook.** Drop `context_injection` (and migrate to the native hook) the moment hermes-agent ships a first-class system-prompt contributor — same as we retire bootstrap routes for native upstream routes.
|
||||
- **Incremental bootstrap migration.** Fold the existing `hermes_relay_bootstrap` route-patches into `plugin/enhancements/` per-surface (startup phase) so patching is one surface; don't big-bang the working compat.
|
||||
- **Structured media channel** — `docs/plans/2026-06-20-structured-media-channel.md` (design only). Replace fragile `MEDIA:`/markdown text markers with a structured channel carrying `sensitive` natively; lead with a relay `relay_send_media(path, sensitive, …)` tool.
|
||||
- **Gateway voice-ephemeral via the same slot.** The enhancement layer's server-side injection can carry per-turn voice instructions on the gateway (which has no ephemeral `system_message`), letting voice stay on the gateway instead of being forced to SSE. Wire when the voice path is revisited.
|
||||
|
||||
---
|
||||
|
||||
## Attachments (shipped 2026-06-18 — `docs/plans/2026-06-18-attachment-experience.md`)
|
||||
|
||||
- **B3 — download progress + cancel.** Inbound fetch is un-cancelable; the previews work scaffolded an indeterminate bar + nullable `onCancel`. Live wiring needs the fetch-path owner (`ChatViewModel`/`Attachment`) to expose determinate progress (Content-Length) + a cancel hook.
|
||||
- **A6 — multi-image gallery.** N images in one message → grid + swipe-across viewer (Telegram media-group parity).
|
||||
- **C5 — agent-side sensitivity config gate.** `RELAY_MEDIA_SENSITIVITY_HINTS` (env or per-profile) instructing the agent to annotate sensitive media via the prompt-builder. Transport (relay `X-Media-Sensitive` header + client blur) already ships; the agent isn't asked to set the bit yet.
|
||||
- **Relay thumbnails (D6).** Server-side thumbnail generation to avoid full-size download for cards/galleries. Needs an image lib (Pillow not currently a dep) — evaluate before adding.
|
||||
- **D5 — outbound upload progress.** No per-attachment progress during the 60s gateway PDF-render window.
|
||||
|
||||
## Voice overhaul (shipped 2026-06-18 — `docs/plans/2026-06-18-voice-overhaul.md`)
|
||||
|
||||
- **Per-profile voice on Standard (upstream PR).** Upstream `/api/profiles/*` has no voice field and `/api/audio/*` is host-global. Long-term: PR a voice section to the profile config + make `/api/audio/*` honor the active/`?profile=` profile. The relay path already carries per-profile voice; ship that first.
|
||||
- **Wire connectionId for per-profile voice namespacing.** `VoicePreferencesRepository` is scope-aware (`base_connId_profile`), but `RelayApp` passes only the profile *name* to `onProfileChanged`, so `connectionId` is null and keys namespace by profile-only. Wire `setVoicePrefsConnection` to `ConnectionViewModel.activeConnectionId` (in `RelayApp`) so two connections with same-named profiles don't share voice settings.
|
||||
- **Realtime-PCM waveform output gating.** The basic-TTS output waveform is now Visualizer-accurate (gated on real playback amplitude), but the realtime path gates `outputAudioActive` on `audioSeen` (first decoded PCM bytes) in `VoiceViewModel.handleRealtimeVoiceEvent`, which can still lead audible output by the `RealtimePcmPlayer` start prebuffer. Gate realtime on actual playback-start (head moved) to match the basic-TTS path.
|
||||
|
||||
## Chat clean-mode + pets (shipped 2026-06-18 — `docs/plans/2026-06-18-chat-clean-mode-and-pets.md`)
|
||||
|
||||
- **Part-A chat polish (optional bundle).** Per-code-block copy + horizontal scroll, visible copy affordance, mid-stream stall feedback, profile/skill-aware empty-state chips, the ~40-flow recomposition hotspot at the top of `ChatScreen`. (Sphere `contentDescription`/reduced-motion was handled by the clean-mode a11y work.)
|
||||
- **Pet hot-load + in-app add/remove (shipped 2026-06-20).** Pets now live-refresh: an `avatarsRefreshTick` keys the avatar `produceState` in `RelayApp`, and Appearance re-scans `pets/` on open and after in-app import/delete — no app restart. Appearance gained "Add a pet" (SAF `.zip` import via `PetImporter`, zip-slip/zip-bomb guarded + validated through `toAvatar`) and an "Installed pets" list with per-pet remove (`PetLoader.deletePet`, confirm dialog, Sphere fallback). Remaining:
|
||||
- **Sphere-skin parity.** Skins are still process-scoped + `adb push` only — the live tick and the importer cover pets, not skins. Extend the tick to `loadUserSkins` and add a `.json` skin import if hot-loading/adding skins in-app is wanted.
|
||||
- **`adb push` into `Android/data` hangs on Samsung scoped storage.** Confirmed: pushing a pet pack to `/sdcard/Android/data/<pkg>/files/pets/` stalls (no bytes written) although `adb shell ls` of the dir works. In-app `.zip` import is the supported path; `/sdcard/Download` pushes fine. Consider softening `docs/pet-spec.md` + user-docs to lead with in-app import over adb.
|
||||
- **On-device import/delete smoke.** Import `/sdcard/Download/lucy.zip` via Add a pet → confirm Lucy appears, selects, and animates all states; then remove it and confirm the avatar falls back to the Sphere.
|
||||
- **Pet state-change re-decode can flash one blank frame.** When the agent state switches clips, the first frame of the new clip may briefly be blank during decode; prewarm/hold-last-frame to smooth it. Root cause is the same as the next item: `PetAvatar.Render` re-decodes from disk on every clip change.
|
||||
- **Pet frame-sequence memory: no cap or downsample (audit 2026-06-19).** `decodeClip` decodes every frame of the selected clip into `List<ImageBitmap>` at full resolution with no `inSampleSize` downscale to the display size and no frame-count/dimension ceiling — a long sequence of large PNGs can use a lot of RAM and a single very large image can OOM `BitmapFactory`. Add `inSampleSize` downsampling to the avatar's draw size and/or a documented hard cap. Spec now warns authors (prefer sprite sheets), but the renderer doesn't enforce it.
|
||||
- **Pet decoded-clip cache (audit 2026-06-19).** `PetAvatar.Render` keys `produceState` on `clip`, so idle→thinking→speaking→idle within one turn re-runs `BitmapFactory.decodeFile` from disk each transition (repeated I/O + GC churn, and the blank-frame flash above). Add a small per-avatar `Map<SphereState, PetFrames>` decode cache.
|
||||
- **Pet behavior model — richer state association (spec'd 2026-06-19, `docs/pet-spec.md` "Agent states & pet behavior").** Shipped: the honesty clamp (declared reactivity ∩ `PET_RENDERER_CAPABILITIES`), the friendly `writing` alias, the `**working`/tool-use overlay** (pet-local sub-state from `toolCallBurst`; opt-in `working` clip drives both the swap and the Tools badge), the **one-shot reaction layer** (`greet`/`wake` on appear, `done`/`celebrate` on turn-finish — opt-in, play-once-then-revert, transition-derived; `ONE_SHOT_MAX_MS` backstop), and `**intensity` modulation** (opt-in `reactive.intensity` → live playback speedup ≤1.6× via `rememberUpdatedState`; un-clamps the Activity badge). Voice · Tools · Activity reactivity is now complete. Remaining:
|
||||
- `**attention` one-shot (only deferred behavior).** A reaction on notification arrival — needs a host event the avatar doesn't yet receive (unlike `greet`/`done`, which ride state transitions). Would plumb a notification edge into `AvatarRenderState` (or a side channel) + a `PetOneShot.Attention`. Low priority: the avatar is rarely on-screen when notifications land (backgrounded) — see the value analysis; revisit only if the avatar becomes an always-on surface (persistent overlay / Quest port).
|
||||
- **On-device verification (working + one-shots + intensity).** Best seen in clean mode (`AgentTextFlow` feeds `toolCallBurst` + `streamingIntensity` + state transitions). Confirm: a `working` clip swaps in during a tool run and releases ~600ms after (`WORKING_BURST_THRESHOLD` 0.5); a `done` clip plays once on reply completion then returns to idle; a `greet` clip plays once when the avatar appears; with `intensity:true`, a writing/working loop visibly quickens while streaming. Watch for the known clip re-decode flash on each swap (separate TODO — decoded-clip cache).
|
||||
- **Undecodable-but-present image appears valid (audit 2026-06-19).** A file that exists but isn't a decodable image passes the loader's `isFile` check, so the pet shows in the picker but renders blank. Documented as a caveat; consider a cheap header sniff at load time if false-valid pets become a support issue.
|
||||
@@ -106,6 +106,19 @@ android {
|
||||
}
|
||||
}
|
||||
|
||||
// Structural guard: the sideload flavor is distributed via GitHub Releases /
|
||||
// F-Droid / ADB and must NEVER be uploaded to Play Console (it declares the
|
||||
// unattended Device Control surface Play forbids). gradle-play-publisher
|
||||
// generates a publish task per variant, so the aggregate `publishReleaseBundle`
|
||||
// would otherwise try BOTH flavors. Disabling sideload here means only
|
||||
// `publishGooglePlayReleaseBundle` can ever reach Play — see the `play { }`
|
||||
// block below and .github/workflows/release-android.yml.
|
||||
playConfigs {
|
||||
register("sideload") {
|
||||
enabled.set(false)
|
||||
}
|
||||
}
|
||||
|
||||
buildTypes {
|
||||
debug {
|
||||
buildConfigField("boolean", "DEV_MODE", "true")
|
||||
@@ -166,6 +179,10 @@ android {
|
||||
// Robolectric (VoicePlayerTest) needs merged Android resources +
|
||||
// manifest on the unit-test classpath to bootstrap its sandbox.
|
||||
unitTests.isIncludeAndroidResources = true
|
||||
// [POC] Roborazzi runs without its Gradle plugin (the plugin needs AGP's
|
||||
// removed TestedExtension). Force record mode via the test-JVM system
|
||||
// property the plugin would otherwise inject, so captureRoboImage writes.
|
||||
unitTests.all { it.systemProperty("roborazzi.test.record", "true") }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -185,6 +202,17 @@ kotlin {
|
||||
jvmToolchain(17)
|
||||
}
|
||||
|
||||
// [screenshots] Host-side screenshot tests render MessageBubble -> MarkdownContent,
|
||||
// whose code-highlighter (dev.snipme.highlights) ships Java-21 bytecode. The build
|
||||
// toolchain pins test execution to JDK 17, which can't load class-file v65, so run
|
||||
// unit tests on a 21 JVM. Compile target stays 17; on-device (dexed) is unaffected.
|
||||
// foojay (settings.gradle.kts) auto-provisions the 21 JDK if absent.
|
||||
tasks.withType<Test>().configureEach {
|
||||
javaLauncher.set(
|
||||
javaToolchains.launcherFor { languageVersion.set(JavaLanguageVersion.of(21)) }
|
||||
)
|
||||
}
|
||||
|
||||
dependencies {
|
||||
// Compose BOM
|
||||
val composeBom = platform(libs.compose.bom)
|
||||
@@ -270,8 +298,19 @@ dependencies {
|
||||
// across priority groups against real local sockets so the behavior we
|
||||
// validate matches on-device.
|
||||
testImplementation(libs.okhttp.mockwebserver)
|
||||
// Konsist — enforces the ADR 34 upstream/relay/shared package fence as a JUnit test
|
||||
testImplementation(libs.konsist)
|
||||
androidTestImplementation(libs.compose.ui.test.junit4)
|
||||
debugImplementation(libs.compose.ui.tooling)
|
||||
debugImplementation(libs.compose.ui.test.manifest)
|
||||
|
||||
// [POC] Roborazzi host-side screenshot rendering (src/test, Robolectric).
|
||||
// Renders real composables on the JVM at an exact canvas — no device, no
|
||||
// status bar, no clipping. See StoreScreenshotTest.
|
||||
testImplementation("io.github.takahirom.roborazzi:roborazzi:1.43.1")
|
||||
testImplementation("io.github.takahirom.roborazzi:roborazzi-compose:1.43.1")
|
||||
testImplementation(libs.compose.ui.test.junit4)
|
||||
testImplementation(libs.compose.ui.test.manifest)
|
||||
testImplementation("androidx.test.ext:junit:1.2.1")
|
||||
}
|
||||
|
||||
|
||||
@@ -126,7 +126,7 @@ class OnboardingFlowTest {
|
||||
navigateToPage(4)
|
||||
|
||||
composeTestRule
|
||||
.onNodeWithText("Standard Hermes")
|
||||
.onNodeWithText("Vanilla Hermes")
|
||||
.assertIsDisplayed()
|
||||
}
|
||||
|
||||
@@ -135,7 +135,7 @@ class OnboardingFlowTest {
|
||||
setOnboardingContent()
|
||||
navigateToPage(4)
|
||||
|
||||
composeTestRule.onNodeWithText("Standard Hermes").performClick()
|
||||
composeTestRule.onNodeWithText("Vanilla Hermes").performClick()
|
||||
composeTestRule.waitForIdle()
|
||||
|
||||
composeTestRule
|
||||
@@ -151,7 +151,7 @@ class OnboardingFlowTest {
|
||||
setOnboardingContent()
|
||||
navigateToPage(4)
|
||||
|
||||
composeTestRule.onNodeWithText("Standard Hermes").performClick()
|
||||
composeTestRule.onNodeWithText("Vanilla Hermes").performClick()
|
||||
composeTestRule.waitForIdle()
|
||||
|
||||
composeTestRule
|
||||
@@ -172,6 +172,17 @@ class OnboardingFlowTest {
|
||||
.assertIsDisplayed()
|
||||
}
|
||||
|
||||
@Test
|
||||
fun powerPage_linksToPermissionReview() {
|
||||
setOnboardingContent()
|
||||
navigateToPage(3)
|
||||
|
||||
composeTestRule
|
||||
.onNodeWithText("Review permissions")
|
||||
.assertIsDisplayed()
|
||||
.assertIsEnabled()
|
||||
}
|
||||
|
||||
@Test
|
||||
fun skipButton_visibleOnIntroPages_andWizardSkipOnConnectPage() {
|
||||
setOnboardingContent()
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
package com.hermesandroid.relay.ui.screens
|
||||
|
||||
import androidx.compose.ui.test.assertIsDisplayed
|
||||
import androidx.compose.ui.test.junit4.createComposeRule
|
||||
import androidx.compose.ui.test.onNodeWithText
|
||||
import androidx.compose.ui.test.performScrollTo
|
||||
import com.hermesandroid.relay.ui.theme.HermesRelayTheme
|
||||
import org.junit.Rule
|
||||
import org.junit.Test
|
||||
|
||||
class PermissionsStatusScreenTest {
|
||||
|
||||
@get:Rule
|
||||
val composeTestRule = createComposeRule()
|
||||
|
||||
@Test
|
||||
fun permissionsScreen_showsStandardAndOnDemandRows() {
|
||||
composeTestRule.setContent {
|
||||
HermesRelayTheme {
|
||||
PermissionsStatusScreen(
|
||||
onBack = {},
|
||||
onOpenBridge = {},
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
composeTestRule
|
||||
.onNodeWithText("Permissions and capabilities")
|
||||
.assertIsDisplayed()
|
||||
composeTestRule
|
||||
.onNodeWithText("Chat and Manage")
|
||||
.assertIsDisplayed()
|
||||
composeTestRule
|
||||
.onNodeWithText("No Android runtime permission needed. API/dashboard auth is configured separately.")
|
||||
.assertIsDisplayed()
|
||||
composeTestRule
|
||||
.onNodeWithText("Camera")
|
||||
.performScrollTo()
|
||||
.assertIsDisplayed()
|
||||
composeTestRule
|
||||
.onNodeWithText("Microphone")
|
||||
.performScrollTo()
|
||||
.assertIsDisplayed()
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,8 @@
|
||||
package com.hermesandroid.relay.voice
|
||||
|
||||
import com.hermesandroid.relay.network.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.handlers.LocalDispatchResult
|
||||
import com.hermesandroid.relay.network.models.Envelope
|
||||
import com.hermesandroid.relay.network.relay.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.shared.LocalDispatchResult
|
||||
import com.hermesandroid.relay.network.relay.models.Envelope
|
||||
|
||||
/**
|
||||
* Local in-process bridge dispatcher type. The Play flavor never invokes
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
en-US
|
||||
@@ -0,0 +1,62 @@
|
||||
Hermes-Relay is the native Android client for the Hermes agent platform. Point it at your own Hermes instance and chat with your agent, talk to it hands-free, and manage models, keys, skills, and profiles from anywhere.
|
||||
|
||||
It is not a hosted AI service. It is a companion app for the Hermes agent you run, and it talks only to the instances you configure.
|
||||
|
||||
QUICK START
|
||||
|
||||
1. Run hermes-agent with its API server and dashboard enabled on your computer or home server.
|
||||
2. Install Hermes-Relay and enter your server address, for example http://192.168.1.100:8642.
|
||||
3. The setup wizard checks what your server supports and shows a readiness card, then you are ready to chat.
|
||||
|
||||
A plain Hermes install is enough. Chat, management, and voice work with no plugin or extra service.
|
||||
|
||||
HOW IT WORKS
|
||||
|
||||
Chat streams directly from your Hermes API Server or dashboard gateway in real time. Manage and voice use your Hermes dashboard with one sign-in. Run the optional relay service and the app can pair by QR code to add power tools: remote terminal, notification companion, media handoff, relay-session management, and additional voice engines.
|
||||
|
||||
GOOGLE PLAY BUILD
|
||||
|
||||
The Google Play build ships Hermes Bridge Core only. It has no AccessibilityService Device Control: it cannot read your screen, tap, type, swipe, screenshot, send SMS, place calls, or access contacts or location. Device Control is reserved for sideload builds distributed outside Google Play.
|
||||
|
||||
FEATURES
|
||||
|
||||
- Streaming Chat: real-time responses with reasoning, markdown, tool-call visibility, attachments, mid-turn steering, edit-and-resend, and a searchable command palette.
|
||||
|
||||
- Manage Your Agent: use your Hermes dashboard from your phone to switch models, manage provider keys, edit profiles, and browse, install, and update skills.
|
||||
|
||||
- Voice Mode: talk hands-free using your server's speech providers. Relay-paired setups add per-profile voices and an experimental realtime engine.
|
||||
|
||||
- Works Away From Home: add LAN, Tailscale, or public routes and the app chooses the best available path on connect.
|
||||
|
||||
- Sessions: create, switch, rename, and delete chats. Message history loads on demand.
|
||||
|
||||
- Multiple Servers and Profiles: connect to more than one server and switch in a tap; overlay an agent profile or personality per conversation.
|
||||
|
||||
- Relay Power Tools: optional QR pairing for remote terminal, relay-session management, media handoff, and per-feature grants.
|
||||
|
||||
- Notification Companion: optionally forward notification metadata to your paired relay so your assistant can summarize it. Toggle it anytime in system settings.
|
||||
|
||||
- Stats for Nerds: local-only counters for response timing, token usage, cost, and stream health.
|
||||
|
||||
- Material You: Material 3 dynamic color, light/dark/system themes, and haptics.
|
||||
|
||||
SECURITY AND PRIVACY
|
||||
|
||||
- API keys and relay tokens are stored in encrypted Android storage.
|
||||
- HTTPS is enforced for remote connections; cleartext is limited to localhost or LAN setups.
|
||||
- No telemetry, ads, tracking, or third-party analytics SDKs.
|
||||
- Notification access and the microphone are optional and user-controlled.
|
||||
- All app traffic goes only to servers you configure.
|
||||
|
||||
REQUIREMENTS
|
||||
|
||||
- Android 8.0 or later.
|
||||
- A running Hermes agent for chat, management, and voice.
|
||||
- Optional Hermes relay service for power tools such as terminal, notifications, and media.
|
||||
- Network access to your server by local network, VPN, or internet.
|
||||
|
||||
OPEN SOURCE
|
||||
|
||||
Hermes-Relay is MIT licensed. Source, docs, and issue tracking are on GitHub.
|
||||
|
||||
This app is a community project and is not affiliated with or endorsed by NousResearch.
|
||||
|
After Width: | Height: | Size: 44 KiB |
|
After Width: | Height: | Size: 37 KiB |
|
After Width: | Height: | Size: 152 KiB |
|
After Width: | Height: | Size: 182 KiB |
|
After Width: | Height: | Size: 112 KiB |
|
After Width: | Height: | Size: 131 KiB |
|
After Width: | Height: | Size: 129 KiB |
|
After Width: | Height: | Size: 246 KiB |
|
After Width: | Height: | Size: 140 KiB |
|
After Width: | Height: | Size: 165 KiB |
@@ -0,0 +1 @@
|
||||
Your Hermes AI agent, in your pocket - chat, voice, and control.
|
||||
@@ -0,0 +1 @@
|
||||
Hermes-Relay
|
||||
@@ -0,0 +1,7 @@
|
||||
v1.2.0 — Make it yours.
|
||||
|
||||
• Eight app themes, swappable sphere skins, and animated agent "pets" that react to what your agent is doing.
|
||||
• See which streaming path you're on, plus a "What the agent sees" sheet showing the agent's exact context.
|
||||
• ~3× faster cold start and honest loading states.
|
||||
• In-app crash reporting with one-tap bug reports.
|
||||
• Fixes: QR pairing on foldables, server-image & PDF crashes, in-chat model picks now apply.
|
||||
@@ -1,5 +1,6 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
|
||||
xmlns:tools="http://schemas.android.com/tools">
|
||||
|
||||
<uses-permission android:name="android.permission.INTERNET" />
|
||||
<uses-permission android:name="android.permission.ACCESS_NETWORK_STATE" />
|
||||
@@ -36,6 +37,8 @@
|
||||
android:name=".MainActivity"
|
||||
android:exported="true"
|
||||
android:launchMode="singleTask"
|
||||
android:screenOrientation="portrait"
|
||||
tools:ignore="LockedOrientationActivity"
|
||||
android:configChanges="uiMode|fontScale|locale|density|orientation|screenSize|screenLayout|keyboardHidden"
|
||||
android:windowSoftInputMode="adjustResize"
|
||||
android:theme="@style/Theme.HermesRelay.Splash">
|
||||
@@ -73,7 +76,7 @@
|
||||
runs while the user has explicitly enabled the toggle. specialUse
|
||||
needs a Play Console foreground-service declaration at submission. -->
|
||||
<service
|
||||
android:name=".network.GatewayKeepAliveService"
|
||||
android:name=".network.upstream.GatewayKeepAliveService"
|
||||
android:exported="false"
|
||||
android:foregroundServiceType="specialUse">
|
||||
<property
|
||||
|
||||
@@ -26,7 +26,9 @@
|
||||
left: 0;
|
||||
right: 0;
|
||||
bottom: 0;
|
||||
padding: 8px 6px 0 8px;
|
||||
/* Bottom gap so xterm's last row clears the extra-keys footer
|
||||
instead of butting flush against it (read as an overlap). */
|
||||
padding: 8px 6px 8px 8px;
|
||||
box-sizing: border-box;
|
||||
}
|
||||
.xterm .xterm-viewport {
|
||||
@@ -149,6 +151,18 @@
|
||||
}
|
||||
});
|
||||
|
||||
// Report scroll position so the host can show a "jump to latest" pill
|
||||
// while the user is scrolled up into scrollback. atBottom is true when
|
||||
// the viewport is pinned to the live tail.
|
||||
const reportScroll = function () {
|
||||
if (!(window.AndroidBridge && window.AndroidBridge.onScrollPosition)) return;
|
||||
try {
|
||||
const buf = term.buffer.active;
|
||||
window.AndroidBridge.onScrollPosition(buf.viewportY >= buf.baseY);
|
||||
} catch (_) {}
|
||||
};
|
||||
term.onScroll(function () { reportScroll(); });
|
||||
|
||||
// ── Inbound: Android → terminal ───────────────────────────────────
|
||||
// Base64-encoded payloads avoid JS string-escaping headaches when the
|
||||
// stream contains control characters, raw escape sequences, or bytes
|
||||
@@ -223,6 +237,13 @@
|
||||
try { term.focus(); } catch (_) {}
|
||||
};
|
||||
|
||||
// Current xterm selection as plain text ('' when nothing selected).
|
||||
// Read back via WebView.evaluateJavascript for the toolbar Copy key,
|
||||
// since long-press copy is unreliable inside an Android WebView.
|
||||
window.getSelectionText = function () {
|
||||
try { return term.getSelection() || ''; } catch (_) { return ''; }
|
||||
};
|
||||
|
||||
window.clearTerminal = function () {
|
||||
try { term.clear(); } catch (_) {}
|
||||
};
|
||||
@@ -239,6 +260,36 @@
|
||||
}
|
||||
};
|
||||
|
||||
// Mode-aware encoder for the on-screen toolbar's special keys
|
||||
// (arrows / Home / End / Page). Arrows must follow xterm's current
|
||||
// DECCKM (application cursor keys) mode: when an app like vim, less,
|
||||
// or readline has requested it, an arrow is SS3-encoded (\eOA) rather
|
||||
// than CSI (\e[A). The old path always sent CSI from Kotlin, which the
|
||||
// running TUI could misread. We read term.modes here (where the mode
|
||||
// actually lives) and route bytes back through onInput so sticky
|
||||
// modifiers still apply. Page keys are mode-independent.
|
||||
window.termSendKey = function (name) {
|
||||
var appCursor = false;
|
||||
try {
|
||||
appCursor = !!(term.modes && term.modes.applicationCursorKeysMode);
|
||||
} catch (_) {}
|
||||
var p = appCursor ? 'O' : '[';
|
||||
var map = {
|
||||
ArrowUp: p + 'A',
|
||||
ArrowDown: p + 'B',
|
||||
ArrowRight: p + 'C',
|
||||
ArrowLeft: p + 'D',
|
||||
Home: p + 'H',
|
||||
End: p + 'F',
|
||||
PageUp: '[5~',
|
||||
PageDown: '[6~',
|
||||
};
|
||||
var seq = map[name];
|
||||
if (seq && window.AndroidBridge && window.AndroidBridge.onInput) {
|
||||
window.AndroidBridge.onInput(seq);
|
||||
}
|
||||
};
|
||||
|
||||
// ── Scroll shims + gesture ────────────────────────────────────────
|
||||
// xterm.js ships a scrollback buffer (scrollback: 10000 above) but
|
||||
// has no built-in mobile touch-to-scroll — its input handlers are
|
||||
|
||||
@@ -1,36 +1,32 @@
|
||||
v1.0.0 - The 1.0 release
|
||||
v1.2.0 - Make it yours
|
||||
|
||||
Standard path
|
||||
* Chat, Manage, and voice now work on a plain Hermes agent — no relay
|
||||
plugin required. The plugin is optional and only adds power tools.
|
||||
Personalize
|
||||
* Eight app themes in Settings → Appearance — the Hermes Relay brand plus
|
||||
ports of the Nous Hermes looks (Teal, Nous Blue, Midnight, Ember, Mono,
|
||||
Cyberpunk, Rosé), with light/dark.
|
||||
* Swap the agent orb for an animated pet that reacts to what the agent is
|
||||
doing — add, preview, and tune pets right in the app, or generate one
|
||||
from sprite art with the AI authoring kit.
|
||||
* Reskin the sphere, and give each agent profile its own icon.
|
||||
|
||||
Chat
|
||||
* New gateway transport streams the agent's reasoning live, so the
|
||||
Thinking block fills in during generation instead of after.
|
||||
* Warm-start + opt-in "Keep connected in background" make returning to a
|
||||
conversation fast.
|
||||
* Attachments at desktop parity: images, PDFs, and files upload over the
|
||||
gateway. If a connection can't carry a file, you'll see a notice
|
||||
instead of a silent drop.
|
||||
* Steer a running turn, edit & resend your messages, watch subagent
|
||||
lanes, and a context-window meter — plus turn-complete notifications.
|
||||
* Tap an image to open it full-screen (pinch to zoom); save or share
|
||||
images and other attachments.
|
||||
* Redesigned input bar: pill field, one morphing Send/Voice/Stop button.
|
||||
See what's happening
|
||||
* The chat status strip names the actual streaming path (Gateway, Sessions,
|
||||
Completions, Runs), with a basic→best tier ladder in Chat Settings.
|
||||
* Tap the context meter for a "What the agent sees" sheet — the exact extra
|
||||
context prepended to your next turn.
|
||||
* Voice and Realtime turns are badged in the scrollback.
|
||||
|
||||
Profiles
|
||||
* Switch the whole agent — model, persona, and skills — per conversation.
|
||||
The drawer scopes to the active profile, and switching is ephemeral: it
|
||||
never changes your server's default agent.
|
||||
Privacy
|
||||
* When paired to the relay, the agent can mark private media and the phone
|
||||
blurs it per your setting — sensitivity stays model-emitted.
|
||||
|
||||
Manage
|
||||
* Models, provider keys, profiles + SOUL.md, and a skills hub — parity
|
||||
with the desktop dashboard. Cached for instant cold-launch.
|
||||
Faster & more reliable
|
||||
* Cold start is about 3× faster, and model/personality/approvals load
|
||||
honestly instead of showing a maybe-wrong value.
|
||||
* In-app crash reporting offers a one-tap, pre-filled bug report.
|
||||
* QR pairing no longer force-closes on unusual cameras (foldables); fixed
|
||||
crashes opening server images and PDFs; in-chat model picks now apply.
|
||||
|
||||
Voice
|
||||
* Realtime Agent keeps one session across turns; long runs continue in
|
||||
the background and are spoken when ready.
|
||||
|
||||
Polish
|
||||
* Seamless LAN/Tailscale handoffs (no chat reload), slide-down status
|
||||
toasts, and a broad round of fixes.
|
||||
Voice & terminal
|
||||
* Enhanced voice control for Gemini and xAI providers.
|
||||
* Leaner terminal with TUI-correct input and an isolated, tuned tmux.
|
||||
|
||||
@@ -1,9 +1,6 @@
|
||||
package com.hermesandroid.relay
|
||||
|
||||
import android.app.Application
|
||||
import android.os.Build
|
||||
import androidx.compose.ui.ComposeUiFlags
|
||||
import androidx.compose.ui.ExperimentalComposeUiApi
|
||||
import coil3.ImageLoader
|
||||
import coil3.PlatformContext
|
||||
import coil3.SingletonImageLoader
|
||||
@@ -13,6 +10,7 @@ import com.hermesandroid.relay.bridge.UnattendedAccessManager
|
||||
import com.hermesandroid.relay.data.AppAnalytics
|
||||
import com.hermesandroid.relay.power.WakeLockManager
|
||||
import com.hermesandroid.relay.util.AppForegroundTracker
|
||||
import com.hermesandroid.relay.util.CrashReporter
|
||||
|
||||
class HermesRelayApp : Application(), SingletonImageLoader.Factory {
|
||||
|
||||
@@ -28,24 +26,12 @@ class HermesRelayApp : Application(), SingletonImageLoader.Factory {
|
||||
.crossfade(true)
|
||||
.build()
|
||||
|
||||
@OptIn(ExperimentalComposeUiApi::class)
|
||||
override fun attachBaseContext(base: android.content.Context?) {
|
||||
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.VANILLA_ICE_CREAM) {
|
||||
ComposeUiFlags.isAdaptiveRefreshRateEnabled = false
|
||||
}
|
||||
super.attachBaseContext(base)
|
||||
}
|
||||
|
||||
@OptIn(ExperimentalComposeUiApi::class)
|
||||
override fun onCreate() {
|
||||
super.onCreate()
|
||||
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.VANILLA_ICE_CREAM) {
|
||||
// Compose's adaptive refresh-rate hint path on API 35 can emit
|
||||
// `setRequestedFrameRate frameRate=NaN` from inside AndroidComposeView
|
||||
// on every draw pass. Disable ARR globally until the upstream fix lands.
|
||||
ComposeUiFlags.isAdaptiveRefreshRateEnabled = false
|
||||
}
|
||||
instance = this
|
||||
// Install the crash handler FIRST so any failure in the rest of app
|
||||
// init (or anywhere later) is captured and surfaced on next launch.
|
||||
CrashReporter.install(this)
|
||||
AppAnalytics.initialize(this)
|
||||
// A8 — wire the bridge-gesture wake-lock wrapper so
|
||||
// ActionExecutor.tap/tapText/typeText/swipe/scroll can hold
|
||||
|
||||
@@ -21,7 +21,6 @@ import com.hermesandroid.relay.bridge.UnattendedAccessManager
|
||||
import com.hermesandroid.relay.data.BuildFlavor
|
||||
import com.hermesandroid.relay.notifications.TurnCompleteNotifier
|
||||
import com.hermesandroid.relay.ui.RelayApp
|
||||
import com.hermesandroid.relay.util.ComposeArrWorkaround
|
||||
import com.hermesandroid.relay.util.NavRouteRequest
|
||||
import com.hermesandroid.relay.viewmodel.ConnectionViewModel
|
||||
|
||||
@@ -118,9 +117,6 @@ class MainActivity : ComponentActivity() {
|
||||
setContent {
|
||||
RelayApp()
|
||||
}
|
||||
window.decorView.post {
|
||||
ComposeArrWorkaround.disableForViewTree(window.decorView)
|
||||
}
|
||||
}
|
||||
|
||||
override fun onNewIntent(intent: Intent) {
|
||||
|
||||
@@ -1551,7 +1551,7 @@ class ActionExecutor(private val service: HermesAccessibilityService) {
|
||||
* googlePlay as a dialer-opener" per the plan.
|
||||
*
|
||||
* The destructive-verb confirmation modal is fired in
|
||||
* [com.hermesandroid.relay.network.handlers.BridgeCommandHandler]
|
||||
* [com.hermesandroid.relay.network.relay.BridgeCommandHandler]
|
||||
* before we even get here — by the time this method runs, the user
|
||||
* has explicitly approved the call.
|
||||
*/
|
||||
|
||||
@@ -12,8 +12,8 @@ import android.util.Log
|
||||
import com.hermesandroid.relay.bridge.BridgeSafetyManager
|
||||
import com.hermesandroid.relay.bridge.UnattendedAccessManager
|
||||
import com.hermesandroid.relay.data.BuildFlavor
|
||||
import com.hermesandroid.relay.network.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.models.Envelope
|
||||
import com.hermesandroid.relay.network.relay.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.relay.models.Envelope
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.Job
|
||||
import kotlinx.coroutines.delay
|
||||
|
||||
@@ -68,7 +68,7 @@ class HermesAccessibilityService : AccessibilityService() {
|
||||
* service is not running. Written on [onServiceConnected],
|
||||
* cleared on [onUnbind] / [onDestroy].
|
||||
*
|
||||
* Read by [com.hermesandroid.relay.network.handlers.BridgeCommandHandler]
|
||||
* Read by [com.hermesandroid.relay.network.relay.BridgeCommandHandler]
|
||||
* and by the Bridge UI screen (bridge-ui) to check live status.
|
||||
*/
|
||||
@Volatile
|
||||
|
||||
@@ -38,7 +38,13 @@ import kotlin.math.sqrt
|
||||
* The Visualizer is attached exactly once against the ExoPlayer's
|
||||
* [ExoPlayer.getAudioSessionId]. There is a known gotcha where re-attaching
|
||||
* the Visualizer on every track transition invalidates the session id — the
|
||||
* single-attach lifecycle here sidesteps it entirely.
|
||||
* single-attach lifecycle here sidesteps it entirely. The single attach is
|
||||
* triggered by whichever of {playback became live, a real session id landed}
|
||||
* arrives last, so a late AudioTrack allocation (deep-buffer cold-start) can't
|
||||
* leave amplitude pinned at 0 for the turn — see [attachVisualizerIfPlaying].
|
||||
* That promptness matters because the voice overlay gates its output waveform
|
||||
* on the first real playback-amplitude frame, so the visual follows audible
|
||||
* speech instead of leading it.
|
||||
*
|
||||
* @param context used for [ExoPlayer.Builder]. Application context is fine;
|
||||
* the player holds no view references.
|
||||
@@ -109,6 +115,21 @@ class VoicePlayer(
|
||||
audioSessionId: Int,
|
||||
) {
|
||||
cachedAudioSessionId = audioSessionId
|
||||
// Deep-buffer cold-start guard. On some OEM pipelines the
|
||||
// AudioTrack — and therefore a real (non-zero) session id —
|
||||
// isn't allocated until *after* onIsPlayingChanged(true) has
|
||||
// already fired. In that race the isPlaying-driven attach
|
||||
// below ran with id == 0, no-oped, and isPlaying will not
|
||||
// toggle again for the rest of a continuous TTS turn, so the
|
||||
// Visualizer would never attach and [amplitude] would stay
|
||||
// pinned at 0 for the whole turn. The output waveform gates
|
||||
// its unfold on the first real playback-amplitude frame, so a
|
||||
// never-firing amplitude leaves it stuck in the folded
|
||||
// processing/spinner shape even though audio is audible.
|
||||
// Attaching here — the moment a real session id lands while
|
||||
// playback is already live — makes the first-audible-frame
|
||||
// signal reliable regardless of when the track allocates.
|
||||
attachVisualizerIfPlaying()
|
||||
}
|
||||
})
|
||||
exoPlayer.addListener(object : Player.Listener {
|
||||
@@ -124,11 +145,11 @@ class VoicePlayer(
|
||||
// runs on the main thread too, so reading the getter here
|
||||
// is safe and guarantees the cache is warm by the time
|
||||
// playback is audible (and thus by the time barge-in
|
||||
// starts its IO reader).
|
||||
// starts its IO reader). If the id isn't ready yet, the
|
||||
// analytics callback above re-tries the attach the instant
|
||||
// it lands (see attachVisualizerIfPlaying).
|
||||
cachedAudioSessionId = exoPlayer.audioSessionId
|
||||
if (!visualizerAttached) {
|
||||
attachVisualizer(cachedAudioSessionId)
|
||||
}
|
||||
attachVisualizerIfPlaying()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -308,6 +329,24 @@ class VoicePlayer(
|
||||
exoPlayer.release()
|
||||
}
|
||||
|
||||
/**
|
||||
* Attach the [Visualizer] iff playback is live and we haven't attached for
|
||||
* this session yet. Idempotent and main-thread-only: both call sites
|
||||
* ([Player.Listener.onIsPlayingChanged] and the [AnalyticsListener]'s
|
||||
* `onAudioSessionIdChanged`) are delivered on the player's application
|
||||
* thread, so the [visualizerAttached] check needs no extra synchronization.
|
||||
*
|
||||
* The delegate [attachVisualizer] still no-ops (without latching
|
||||
* [visualizerAttached]) when the cached session id is 0, which preserves
|
||||
* the retry: whichever of {isPlaying, valid session id} arrives last drives
|
||||
* the single attach. This is the cold-start race fix — see the
|
||||
* `onAudioSessionIdChanged` comment in `init`.
|
||||
*/
|
||||
private fun attachVisualizerIfPlaying() {
|
||||
if (visualizerAttached || !_isPlaying.value) return
|
||||
attachVisualizer(cachedAudioSessionId)
|
||||
}
|
||||
|
||||
private fun attachVisualizer(audioSessionId: Int) {
|
||||
if (audioSessionId == 0) {
|
||||
// ExoPlayer returns 0 before the audio track is allocated; retry
|
||||
|
||||
@@ -6,8 +6,8 @@ import com.hermesandroid.relay.data.Connection
|
||||
import com.hermesandroid.relay.data.EndpointCandidate
|
||||
import com.hermesandroid.relay.data.PairingPreferences
|
||||
import com.hermesandroid.relay.data.Profile
|
||||
import com.hermesandroid.relay.network.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.models.Envelope
|
||||
import com.hermesandroid.relay.network.relay.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.relay.models.Envelope
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.flow.MutableStateFlow
|
||||
@@ -92,6 +92,19 @@ class AuthManager(
|
||||
* legacy connection intentionally keeps [Connection.LEGACY_TOKEN_STORE_KEY].
|
||||
*/
|
||||
private val tokenStoreKey: String? = null,
|
||||
/**
|
||||
* When false, [init] skips the eager session-token hydration (and the
|
||||
* keyset decrypt it forces). Used for the throwaway LEGACY SENTINEL manager
|
||||
* that `ConnectionViewModel` builds at field-init and replaces as soon as
|
||||
* the active connection hydrates — decrypting its keyset only to discard it
|
||||
* is a measured ~600 ms of wasted startup keystore work, and on a device
|
||||
* whose active connection isn't connection 0 the sentinel's file has no
|
||||
* token anyway. The real per-connection manager (created via the active
|
||||
* connection, [eagerHydrate] = true) hydrates normally; the
|
||||
* `restorePersistedActiveConnectionContext` path even awaits its
|
||||
* Paired/Failed state. Channel handlers are still registered either way.
|
||||
*/
|
||||
private val eagerHydrate: Boolean = true,
|
||||
) : ChannelMultiplexer.ChannelHandler {
|
||||
|
||||
companion object {
|
||||
@@ -102,6 +115,10 @@ class AuthManager(
|
||||
private const val KEY_API_KEY = "api_server_key"
|
||||
private const val HINT_API_KEY_PRESENT = "api_key_present"
|
||||
private const val KEY_PAIRED_META = "paired_session_meta_json"
|
||||
// Marker (in the connection-0 token store) recording that the one-shot
|
||||
// pre-StrongBox `hermes_companion_auth` → `hermes_companion_auth_hw`
|
||||
// migration has run, so we never rebuild the legacy keyset to re-check.
|
||||
private const val KEY_LEGACY_MIGRATED = "legacy_migrated"
|
||||
private const val PAIRING_CODE_LENGTH = 6
|
||||
private val PAIRING_CODE_CHARS = ('A'..'Z') + ('0'..'9')
|
||||
|
||||
@@ -329,19 +346,29 @@ class AuthManager(
|
||||
_store?.let { return it }
|
||||
return storeMutex.withLock {
|
||||
_store?.let { return it }
|
||||
withContext(Dispatchers.IO) {
|
||||
// Multi-connection: [tokenPrefsName] picks the
|
||||
// EncryptedSharedPreferences filename for the bound
|
||||
// connection. The legacy sentinel keeps the pre-multi-
|
||||
// connection install on its original file so the existing
|
||||
// paired device keeps working with no migration.
|
||||
val picked: SessionTokenStore =
|
||||
KeystoreTokenStore.tryCreate(context, tokenPrefsName)
|
||||
?: LegacyEncryptedPrefsTokenStore(context, tokenPrefsName)
|
||||
migrateFromLegacyIfNeeded(picked)
|
||||
_store = picked
|
||||
picked
|
||||
val picked = withContext(Dispatchers.IO) {
|
||||
// One keyset build per file, process-wide (see [SecureStoreCache]).
|
||||
// The legacy sentinel is deferred (eagerHydrate=false) and the
|
||||
// dashboard cookie store now shares this same file, so the active
|
||||
// connection's token keyset is the ONLY one built on the cold-
|
||||
// start critical path. [tokenPrefsName] picks the file.
|
||||
//
|
||||
// The build decrypts its Tink keyset eagerly, so a corrupt file
|
||||
// can throw AEADBadTagException — KeystoreTokenStore.tryCreate
|
||||
// degrades to null, the legacy store self-heals in its ctor, and
|
||||
// a fundamentally broken keystore falls back to InMemory (the app
|
||||
// stays up; the user re-pairs). See [buildRawTokenStore].
|
||||
val s = SecureStoreCache.getOrBuild(tokenPrefsName) {
|
||||
buildRawTokenStore(context, tokenPrefsName)
|
||||
}
|
||||
// Migration runs AFTER the (shared) build so the cookie store can
|
||||
// trigger the build without needing token-migration logic; a
|
||||
// marker makes it read the legacy file at most once ever.
|
||||
migrateFromLegacyIfNeeded(s)
|
||||
s
|
||||
}
|
||||
_store = picked
|
||||
picked
|
||||
}
|
||||
}
|
||||
|
||||
@@ -353,14 +380,33 @@ class AuthManager(
|
||||
*/
|
||||
private fun migrateFromLegacyIfNeeded(picked: SessionTokenStore) {
|
||||
if (picked is LegacyEncryptedPrefsTokenStore) return
|
||||
// Multi-connection: only the legacy connection inherits from the pre-
|
||||
// multi-connection `hermes_companion_auth` file. A freshly-minted
|
||||
// per-connection store must NOT be seeded from the legacy file or
|
||||
// Gate on the FILE, not the connection id. Only the legacy connection-0
|
||||
// file (`hermes_companion_auth_hw`) inherits from the pre-multi-
|
||||
// connection `hermes_companion_auth` file; a freshly-minted per-
|
||||
// connection store (`hermes_auth_<id>`) must NOT be seeded from it or
|
||||
// we'd copy connection 0's token into every new connection.
|
||||
if (connectionId != CONNECTION_ID_LEGACY) return
|
||||
//
|
||||
// Why file-gated rather than `connectionId == CONNECTION_ID_LEGACY`:
|
||||
// the store build is now cached/deduped across the legacy sentinel and
|
||||
// the real connection-0 manager, so whichever one builds the file first
|
||||
// runs this migration. Both share `tokenPrefsName == LEGACY_TOKEN_STORE_KEY`
|
||||
// but only the sentinel had `connectionId == CONNECTION_ID_LEGACY`, so
|
||||
// the old id-based gate would skip migration whenever the real manager
|
||||
// won the race — dropping a pre-StrongBox user's token. The file name is
|
||||
// the same for both, so gating on it is race-proof.
|
||||
if (tokenPrefsName != Connection.LEGACY_TOKEN_STORE_KEY) return
|
||||
// Read the legacy file at most ONCE ever. The build is now cache-shared
|
||||
// (and the cookie store can trigger it without migrating), so without
|
||||
// this marker every freshly-rebuilt connection-0 AuthManager would
|
||||
// re-build the legacy `hermes_companion_auth` keyset just to find it
|
||||
// already drained — re-introducing the startup cost we just removed.
|
||||
if (picked.contains(KEY_LEGACY_MIGRATED)) return
|
||||
val legacy = try {
|
||||
LegacyEncryptedPrefsTokenStore(context)
|
||||
} catch (_: Exception) {
|
||||
// Legacy file unreadable/corrupt — nothing to inherit. Still mark
|
||||
// done so its keyset isn't rebuilt on every launch.
|
||||
picked.putString(KEY_LEGACY_MIGRATED, "1")
|
||||
return
|
||||
}
|
||||
|
||||
@@ -384,6 +430,7 @@ class AuthManager(
|
||||
// backup copies of the session token lying around.
|
||||
legacy.clearAll()
|
||||
}
|
||||
picked.putString(KEY_LEGACY_MIGRATED, "1")
|
||||
}
|
||||
|
||||
/** Cert pin store — shared across all relay connections. */
|
||||
@@ -512,24 +559,28 @@ class AuthManager(
|
||||
// one-line change in [onMessage].
|
||||
multiplexer.registerHandler("pairing", this)
|
||||
|
||||
// Check for existing session token off main thread
|
||||
scope.launch {
|
||||
val s = store()
|
||||
val existingToken = s.getString(KEY_SESSION_TOKEN)
|
||||
if (existingToken != null) {
|
||||
_authState.value = AuthState.Paired(existingToken)
|
||||
_currentPairedSession.value = loadStoredMetadata(existingToken)
|
||||
Log.i(
|
||||
TAG,
|
||||
"init: hydrated existing session_token=${existingToken.take(8)}… " +
|
||||
"→ authState=Paired (stale-at-startup unless this is a real continuous session)"
|
||||
)
|
||||
} else {
|
||||
Log.i(TAG, "init: no stored session_token → authState stays Unpaired")
|
||||
// Check for existing session token off main thread. Skipped for the
|
||||
// throwaway sentinel (eagerHydrate=false) so it never pays the keyset
|
||||
// decrypt for a store that's about to be replaced (see [eagerHydrate]).
|
||||
if (eagerHydrate) {
|
||||
scope.launch {
|
||||
val s = store()
|
||||
val existingToken = s.getString(KEY_SESSION_TOKEN)
|
||||
if (existingToken != null) {
|
||||
_authState.value = AuthState.Paired(existingToken)
|
||||
_currentPairedSession.value = loadStoredMetadata(existingToken)
|
||||
Log.i(
|
||||
TAG,
|
||||
"init: hydrated existing session_token=${existingToken.take(8)}… " +
|
||||
"→ authState=Paired (stale-at-startup unless this is a real continuous session)"
|
||||
)
|
||||
} else {
|
||||
Log.i(TAG, "init: no stored session_token → authState stays Unpaired")
|
||||
}
|
||||
// Converge the plain api-key-present hint with the decrypted
|
||||
// truth (also repairs a hint that predates legacy migration).
|
||||
recordApiKeyHint(!s.getString(KEY_API_KEY).isNullOrBlank())
|
||||
}
|
||||
// Converge the plain api-key-present hint with the decrypted
|
||||
// truth (also repairs a hint that predates legacy migration).
|
||||
recordApiKeyHint(!s.getString(KEY_API_KEY).isNullOrBlank())
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -6,6 +6,44 @@ import android.os.Build
|
||||
import android.util.Log
|
||||
import androidx.security.crypto.EncryptedSharedPreferences
|
||||
import androidx.security.crypto.MasterKey
|
||||
import java.util.concurrent.ConcurrentHashMap
|
||||
|
||||
/**
|
||||
* Process-global cache for encrypted stores, keyed by prefs-file name.
|
||||
*
|
||||
* `EncryptedSharedPreferences.create()` unwraps a Tink keyset via a KeyStore op
|
||||
* (~0.6–1 s on StrongBox), and Tink serializes those process-globally — so a
|
||||
* second build of the SAME file is pure waste (the measured cold-start
|
||||
* `Long monitor contention … AndroidKeysetManager.build()` with `waiters=1..4`).
|
||||
*
|
||||
* Caching by file name means each file's keyset builds ONCE process-wide. The
|
||||
* cache is **synchronous** ([ConcurrentHashMap.computeIfAbsent], which holds a
|
||||
* per-key lock so the build runs at most once per file) precisely so the SAME
|
||||
* instance serves both the suspend token path (callers wrap this in
|
||||
* [kotlinx.coroutines.Dispatchers.IO]) AND the synchronous OkHttp cookie-jar
|
||||
* path — which is how the dashboard cookies now ride the connection's
|
||||
* already-built token keyset instead of building a second one.
|
||||
*
|
||||
* The build is ~1 s on StrongBox: call only from IO / OkHttp threads, never the
|
||||
* main thread.
|
||||
*/
|
||||
internal object SecureStoreCache {
|
||||
private val instances = ConcurrentHashMap<String, SessionTokenStore>()
|
||||
|
||||
fun getOrBuild(prefsName: String, build: () -> SessionTokenStore): SessionTokenStore =
|
||||
instances.computeIfAbsent(prefsName) { build() }
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the raw encrypted store for [prefsName] — Keystore-backed when possible,
|
||||
* self-healing legacy fallback, in-memory last resort. No migration. Shared by
|
||||
* the token store and the dashboard cookie store so a given file always yields
|
||||
* the SAME backend, via [SecureStoreCache].
|
||||
*/
|
||||
internal fun buildRawTokenStore(context: Context, prefsName: String): SessionTokenStore =
|
||||
KeystoreTokenStore.tryCreate(context, prefsName)
|
||||
?: runCatching { LegacyEncryptedPrefsTokenStore(context, prefsName) }
|
||||
.getOrElse { InMemoryTokenStore() }
|
||||
|
||||
/**
|
||||
* Abstraction over the storage backend for the relay session token + API key
|
||||
@@ -72,9 +110,12 @@ class KeystoreTokenStore private constructor(
|
||||
) : SessionTokenStore {
|
||||
|
||||
// Mutable so [resetPrefs] can swap in a fresh instance after a corrupted
|
||||
// file is deleted. Built lazily via [buildPrefs] so the constructor can't
|
||||
// throw — [tryCreate] still controls the "is this device usable at all"
|
||||
// decision via its init probe below.
|
||||
// file is deleted. This field initializer runs [buildPrefs] eagerly, so it
|
||||
// CAN throw (e.g. AEADBadTagException on a corrupt keyset) — but the
|
||||
// constructor is private and only reachable via [tryCreate], which wraps
|
||||
// construction in try/catch and degrades to the legacy store. The
|
||||
// directly-constructed legacy path self-heals instead; see
|
||||
// [LegacyEncryptedPrefsTokenStore.buildPrefsResilient].
|
||||
private var prefs: SharedPreferences = buildPrefs()
|
||||
|
||||
private fun buildPrefs(): SharedPreferences {
|
||||
@@ -257,7 +298,38 @@ class LegacyEncryptedPrefsTokenStore(
|
||||
|
||||
// Mutable so [resetPrefs] can swap in a fresh instance after a corrupted
|
||||
// file is deleted. See [KeystoreTokenStore.resetPrefs] for the rationale.
|
||||
private var prefs: SharedPreferences = buildPrefs()
|
||||
//
|
||||
// Built via [buildPrefsResilient] so a corrupt keyset can't crash the
|
||||
// constructor. Unlike [KeystoreTokenStore], this class is `new`-ed
|
||||
// directly (it's the fallback when KeystoreTokenStore.tryCreate returns
|
||||
// null, and the migration source), so there's no tryCreate-style guard
|
||||
// upstream — the healing has to live here.
|
||||
private var prefs: SharedPreferences = buildPrefsResilient()
|
||||
|
||||
/**
|
||||
* Build the encrypted prefs, healing a corrupted keyset on the way.
|
||||
*
|
||||
* [EncryptedSharedPreferences.create] decrypts the Tink keyset eagerly, so
|
||||
* a stale/corrupt legacy file throws [javax.crypto.AEADBadTagException]
|
||||
* (AES-GCM tag mismatch) right here in the constructor. This is the classic
|
||||
* post-upgrade / post-restore failure: the encrypted blob persists but the
|
||||
* hardware master key it was sealed against is gone or rotated. Delete the
|
||||
* file and rebuild a fresh keyset against the current master key rather
|
||||
* than letting the exception escape and force-close the app — the token in
|
||||
* the unreadable file was lost anyway, so the user simply re-pairs.
|
||||
*/
|
||||
private fun buildPrefsResilient(): SharedPreferences =
|
||||
try {
|
||||
buildPrefs()
|
||||
} catch (e: Exception) {
|
||||
Log.w(TAG, "Initial legacy prefs build failed — wiping corrupted file and rebuilding: ${e.message}")
|
||||
try {
|
||||
appContext.deleteSharedPreferences(prefsName)
|
||||
} catch (e2: Exception) {
|
||||
Log.w(TAG, "deleteSharedPreferences($prefsName) failed: ${e2.message}")
|
||||
}
|
||||
buildPrefs()
|
||||
}
|
||||
|
||||
private fun buildPrefs(): SharedPreferences {
|
||||
val masterKey = MasterKey.Builder(appContext)
|
||||
@@ -342,3 +414,24 @@ class LegacyEncryptedPrefsTokenStore(
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// In-memory last-resort implementation
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Non-persistent [SessionTokenStore]. Used only when BOTH the Keystore and the
|
||||
* (self-healing) legacy encrypted store fail to construct — i.e. the device's
|
||||
* AndroidKeystore is so broken it can't even build a fresh key. Tokens live for
|
||||
* the process lifetime only, so the user re-pairs on the next cold start, but
|
||||
* the app stays up instead of force-closing. See [AuthManager.store].
|
||||
*/
|
||||
class InMemoryTokenStore : SessionTokenStore {
|
||||
private val map = java.util.concurrent.ConcurrentHashMap<String, String>()
|
||||
override val hasHardwareBackedStorage: Boolean = false
|
||||
override fun getString(key: String): String? = map[key]
|
||||
override fun putString(key: String, value: String) { map[key] = value }
|
||||
override fun remove(key: String) { map.remove(key) }
|
||||
override fun contains(key: String): Boolean = map.containsKey(key)
|
||||
override fun clearAll() { map.clear() }
|
||||
}
|
||||
|
||||
@@ -25,7 +25,6 @@ import androidx.savedstate.SavedStateRegistryOwner
|
||||
import androidx.savedstate.setViewTreeSavedStateRegistryOwner
|
||||
import com.hermesandroid.relay.ui.components.BridgeStatusOverlayChip
|
||||
import com.hermesandroid.relay.ui.components.DestructiveVerbConfirmDialog
|
||||
import com.hermesandroid.relay.util.ComposeArrWorkaround
|
||||
import java.util.concurrent.ConcurrentHashMap
|
||||
|
||||
/**
|
||||
@@ -158,7 +157,6 @@ class BridgeStatusOverlay(context: Context) : ConfirmationOverlayHost {
|
||||
Log.w(TAG, "addView(chip) failed", it)
|
||||
return
|
||||
}
|
||||
compose.post { ComposeArrWorkaround.disableForViewTree(compose) }
|
||||
chipView = compose
|
||||
chipUnattended = unattended
|
||||
}
|
||||
@@ -227,7 +225,6 @@ class BridgeStatusOverlay(context: Context) : ConfirmationOverlayHost {
|
||||
onResult(false)
|
||||
return
|
||||
}
|
||||
compose.post { ComposeArrWorkaround.disableForViewTree(compose) }
|
||||
activeConfirmations[request.id] = compose
|
||||
}
|
||||
|
||||
|
||||
@@ -233,7 +233,7 @@ object UnattendedAccessManager {
|
||||
* Acquire the screen-bright wake lock + opportunistically request
|
||||
* keyguard dismiss. Synchronous — does not suspend. The caller (
|
||||
* [com.hermesandroid.relay.accessibility.ActionExecutor] wrapper, or
|
||||
* [com.hermesandroid.relay.network.handlers.BridgeCommandHandler]
|
||||
* [com.hermesandroid.relay.network.relay.BridgeCommandHandler]
|
||||
* pre-dispatch hook) holds onto the result and decides whether to
|
||||
* proceed with the action.
|
||||
*
|
||||
|
||||
@@ -10,24 +10,38 @@ package com.hermesandroid.relay.data
|
||||
*/
|
||||
object AgentDisplay {
|
||||
const val SERVER_DEFAULT_PROFILE_KEY: String = "__server_default__"
|
||||
private val GENERIC_MODEL_ALIASES = setOf(
|
||||
"hermes-agent",
|
||||
"hermes_agent",
|
||||
"hermes agent",
|
||||
)
|
||||
|
||||
// Only an EXPLICIT pick drives the effective profile. We deliberately do
|
||||
// NOT fall back to the advertised "default" profile: a dashboard profile's
|
||||
// description is a verbose SOUL summary ("Builds and maintains…"), and
|
||||
// resolving it here replaced the clean agent name (the personality, e.g.
|
||||
// "Victor") with that summary in the header. With no explicit pick, the
|
||||
// name comes from the personality. ([profiles] kept for call-site symmetry.)
|
||||
// Only an EXPLICIT pick drives request/session identity. The advertised
|
||||
// "default" profile is an alias for server default, so falling back to it
|
||||
// here would split chat, voice, or session scope.
|
||||
@Suppress("UNUSED_PARAMETER")
|
||||
fun effectiveProfile(
|
||||
selectedProfile: Profile?,
|
||||
profiles: List<Profile>,
|
||||
): Profile? = selectedProfile
|
||||
|
||||
// The NAME goes in the name slot. A profile's description is a SOUL summary
|
||||
// ("Builds and maintains…"), far too verbose for the agent-name label, so
|
||||
// the profile name wins; description is only a last resort when name is blank.
|
||||
// Display can use the synthetic default profile's metadata without making
|
||||
// it a request/session override. Verbose SOUL summaries are filtered by
|
||||
// profileDisplayName below, so this is safe for headers/cards.
|
||||
fun effectiveDisplayProfile(
|
||||
selectedProfile: Profile?,
|
||||
profiles: List<Profile>,
|
||||
): Profile? = selectedProfile ?: profiles.firstOrNull { isServerDefaultAlias(it.name) }
|
||||
|
||||
// The NAME goes in the name slot. Non-default profiles use their profile
|
||||
// name first. The synthetic default profile uses its description only when
|
||||
// that description looks like a concise human agent name ("Victor"), not a
|
||||
// verbose SOUL summary.
|
||||
fun profileDisplayName(profile: Profile?): String? {
|
||||
if (profile == null) return null
|
||||
if (isServerDefaultAlias(profile.name)) {
|
||||
return defaultProfileDisplayName(profile)
|
||||
}
|
||||
return when {
|
||||
profile.name.isNotBlank() -> titleCase(profile.name.trim())
|
||||
profile.description.isNotBlank() -> profile.description.trim()
|
||||
@@ -35,19 +49,34 @@ object AgentDisplay {
|
||||
}
|
||||
}
|
||||
|
||||
fun defaultProfileDisplayName(profile: Profile?): String? =
|
||||
profile
|
||||
?.description
|
||||
?.trim()
|
||||
?.takeIf { it.looksLikeConciseAgentName() }
|
||||
?.let(::titleCase)
|
||||
|
||||
fun agentName(
|
||||
profile: Profile?,
|
||||
selectedPersonality: String,
|
||||
defaultPersonality: String,
|
||||
connectionLabel: String?,
|
||||
localDisplayAlias: String? = null,
|
||||
): String {
|
||||
localDisplayAlias(localDisplayAlias)?.let { return it }
|
||||
profileDisplayName(profile)?.let { return it }
|
||||
|
||||
// "none"/"neutral" are the upstream "cleared overlay" aliases — treat
|
||||
// them like "default" for identity: fall through to the server default
|
||||
// (or the base connection identity) rather than rendering the literal
|
||||
// word as an agent name.
|
||||
val personalityName = if (
|
||||
selectedPersonality == "default" &&
|
||||
isClearedPersonality(selectedPersonality) &&
|
||||
defaultPersonality.isNotBlank()
|
||||
) {
|
||||
defaultPersonality
|
||||
} else if (isClearedPersonality(selectedPersonality)) {
|
||||
""
|
||||
} else {
|
||||
selectedPersonality
|
||||
}
|
||||
@@ -60,16 +89,43 @@ object AgentDisplay {
|
||||
}
|
||||
}
|
||||
|
||||
/** True for the upstream "clear the overlay" aliases (default == none == neutral). */
|
||||
fun isClearedPersonality(value: String): Boolean =
|
||||
value.trim().lowercase() in setOf("default", "none", "neutral", "")
|
||||
|
||||
fun personalityLabel(
|
||||
selectedPersonality: String,
|
||||
defaultPersonality: String,
|
||||
): String = when {
|
||||
// Explicit "none" — show "None" (or the configured default name, if any)
|
||||
// so the cleared-overlay state is legible in the picker header.
|
||||
selectedPersonality.trim().lowercase() in setOf("none", "neutral") ->
|
||||
if (defaultPersonality.isNotBlank()) titleCase(defaultPersonality.trim()) else "None"
|
||||
selectedPersonality != "default" && selectedPersonality.isNotBlank() ->
|
||||
titleCase(selectedPersonality.trim())
|
||||
defaultPersonality.isNotBlank() -> titleCase(defaultPersonality.trim())
|
||||
else -> "Default"
|
||||
}
|
||||
|
||||
fun displayModelName(model: String?): String? =
|
||||
model
|
||||
?.trim()
|
||||
?.takeIf { it.isNotEmpty() }
|
||||
?.takeUnless { it.lowercase() in GENERIC_MODEL_ALIASES }
|
||||
|
||||
/**
|
||||
* A model string safe to SEND to the server as a model override or
|
||||
* `config.set model=…`. Returns null for the generic agent placeholders
|
||||
* ("hermes-agent", …) which are NOT real models — the server rejects them
|
||||
* (HTTP 400) and falls back. Null means "send no model; use the server's
|
||||
* configured default."
|
||||
*/
|
||||
fun requestModelName(model: String?): String? =
|
||||
model
|
||||
?.trim()
|
||||
?.takeIf { it.isNotEmpty() }
|
||||
?.takeUnless { it.lowercase() in GENERIC_MODEL_ALIASES }
|
||||
|
||||
fun isServerDefaultAlias(profileName: String?): Boolean =
|
||||
profileName?.trim()?.equals("default", ignoreCase = true) == true
|
||||
|
||||
@@ -87,6 +143,22 @@ object AgentDisplay {
|
||||
fun profileContextKey(connectionId: String?, profileName: String?): String =
|
||||
"${connectionId.orEmpty()}::${profileSessionKey(profileName)}"
|
||||
|
||||
fun localDisplayAlias(value: String?): String? =
|
||||
value
|
||||
?.trim()
|
||||
?.replace(Regex("\\s+"), " ")
|
||||
?.takeIf { it.isNotEmpty() }
|
||||
|
||||
private fun String.looksLikeConciseAgentName(): Boolean {
|
||||
if (isBlank() || length > 40 || contains('\n') || contains('\r')) {
|
||||
return false
|
||||
}
|
||||
if (any { it == '.' || it == ':' || it == ';' }) {
|
||||
return false
|
||||
}
|
||||
return trim().split(Regex("\\s+")).size <= 4
|
||||
}
|
||||
|
||||
private fun titleCase(value: String): String =
|
||||
value.replaceFirstChar { it.uppercase() }
|
||||
}
|
||||
|
||||
@@ -46,7 +46,7 @@ data class ChatMessage(
|
||||
/**
|
||||
* Rich content cards emitted by the agent via `CARD:{json}` line
|
||||
* markers in the text stream. Parsed in
|
||||
* [com.hermesandroid.relay.network.handlers.ChatHandler.scanForCardMarkers]
|
||||
* [com.hermesandroid.relay.network.upstream.ChatHandler.scanForCardMarkers]
|
||||
* and rendered inline by
|
||||
* [com.hermesandroid.relay.ui.components.HermesCardBubble]. Mirrors
|
||||
* [attachments]' lifecycle — the marker line is stripped from
|
||||
@@ -76,7 +76,7 @@ data class ChatMessage(
|
||||
* The sync builder treats messages with [voiceIntent] != null and
|
||||
* [VoiceIntentTrace.syncedToServer] == false as the inputs to its
|
||||
* synthesis pass; on a successful send we flip [VoiceIntentTrace.syncedToServer]
|
||||
* to true via [com.hermesandroid.relay.network.handlers.ChatHandler.markVoiceIntentsSynced]
|
||||
* to true via [com.hermesandroid.relay.network.upstream.ChatHandler.markVoiceIntentsSynced]
|
||||
* so they're not re-sent on the next turn.
|
||||
*/
|
||||
val voiceIntent: VoiceIntentTrace? = null,
|
||||
@@ -89,12 +89,33 @@ data class ChatMessage(
|
||||
* the durable session turn; the provider's spoken summary is UI/runtime
|
||||
* provenance, not another canonical assistant message.
|
||||
*/
|
||||
val realtimeTurn: RealtimeTurnTrace? = null
|
||||
val realtimeTurn: RealtimeTurnTrace? = null,
|
||||
/**
|
||||
* True for bubbles that exist ONLY on the client and have no server-side
|
||||
* row — slash-command notices, voice-intent traces, the steer echo, gateway
|
||||
* ask cards, an errored turn the server never persisted, and a provider-only
|
||||
* (non-Hermes-backed) realtime turn. The post-turn history reload
|
||||
* ([com.hermesandroid.relay.network.upstream.ChatHandler.loadMessageHistory])
|
||||
* preserves any client-only message whose id is absent from the reloaded
|
||||
* server transcript; without the flag those orphans would be silently
|
||||
* wiped by the reconcile.
|
||||
*
|
||||
* Replaces the old id-prefix whitelist (`voice-intent-`/`steer-`/`ask-`/
|
||||
* `system-notice-`) + "Error"-badge sniffing: each creator now declares its
|
||||
* own provenance instead of the reconcile having to know every id
|
||||
* convention. Defaults false so every server-backed message and existing
|
||||
* call site stays correct.
|
||||
*
|
||||
* NOTE: an "Error" badge alone does NOT make a message preservable — a turn
|
||||
* can error *after* persisting server-side, and that message must still
|
||||
* reconcile normally. Only [clientOnly] gates orphan preservation.
|
||||
*/
|
||||
val clientOnly: Boolean = false,
|
||||
)
|
||||
|
||||
/**
|
||||
* Structured details about a phone-local voice intent that was dispatched
|
||||
* in-process via [com.hermesandroid.relay.network.handlers.BridgeCommandHandler.handleLocalCommand].
|
||||
* in-process via [com.hermesandroid.relay.network.relay.BridgeCommandHandler.handleLocalCommand].
|
||||
*
|
||||
* Captured on a [ChatMessage] (id prefix `voice-intent-`) so the next chat
|
||||
* payload can include synthetic OpenAI-format `assistant` + `tool` message
|
||||
@@ -123,12 +144,12 @@ data class ChatMessage(
|
||||
* includes an `error` field.
|
||||
* @property resultJson Compact JSON object describing the dispatch outcome.
|
||||
* On success, typically `{"ok":true,...}` with any tool-specific fields
|
||||
* from [com.hermesandroid.relay.network.handlers.LocalDispatchResult.resultJson].
|
||||
* from [com.hermesandroid.relay.network.shared.LocalDispatchResult.resultJson].
|
||||
* On failure, an error envelope including `ok:false`, `error`, optionally
|
||||
* `error_code`. Stored as a string and rendered verbatim into the
|
||||
* synthetic `tool`-role message's `content` field.
|
||||
* @property syncedToServer Idempotency guard. Flipped to true by
|
||||
* [com.hermesandroid.relay.network.handlers.ChatHandler.markVoiceIntentsSynced]
|
||||
* [com.hermesandroid.relay.network.upstream.ChatHandler.markVoiceIntentsSynced]
|
||||
* the moment we hand the request payload to the API client. Once true,
|
||||
* the trace is excluded from future sync passes — the server-side
|
||||
* session has already absorbed it.
|
||||
@@ -186,7 +207,23 @@ data class Attachment(
|
||||
/** Opaque token from `MEDIA:hermes-relay://<token>` — identifies the file on the relay. */
|
||||
val relayToken: String? = null,
|
||||
/** content:// URI from the FileProvider once bytes are cached to disk. */
|
||||
val cachedUri: String? = null
|
||||
val cachedUri: String? = null,
|
||||
/**
|
||||
* Whether this attachment was flagged sensitive (NSFW / spoiler) and should
|
||||
* render blurred until the user taps to reveal — honored per the user's
|
||||
* `MediaSettings.blurMode`.
|
||||
*
|
||||
* The flag is **model-emitted metadata, never an on-device or relay-side
|
||||
* classifier** (see `docs/plans/2026-06-18-attachment-experience.md` §C): the
|
||||
* agent annotates media it surfaces, the relay transports the bit
|
||||
* authoritatively via the `X-Media-Sensitive` response header, and the
|
||||
* client merely renders the blur. Populated for inbound attachments from
|
||||
* [com.hermesandroid.relay.network.relay.RelayHttpClient.FetchedMedia.sensitive]
|
||||
* when the bytes flip to [AttachmentState.LOADED]. Defaults false so every
|
||||
* existing outbound/inbound call site stays valid and unflagged media
|
||||
* renders exactly as before.
|
||||
*/
|
||||
val sensitive: Boolean = false
|
||||
) {
|
||||
val isImage: Boolean get() = contentType.startsWith("image/")
|
||||
|
||||
@@ -272,5 +309,16 @@ data class ChatSession(
|
||||
val title: String?,
|
||||
val model: String?,
|
||||
val messageCount: Int = 0,
|
||||
val updatedAt: Long = 0L
|
||||
)
|
||||
val updatedAt: Long = 0L,
|
||||
val startedAt: Long = 0L,
|
||||
val lastActivityAt: Long = 0L
|
||||
) {
|
||||
val activityTimestamp: Long
|
||||
get() = firstPositive(lastActivityAt, updatedAt, startedAt)
|
||||
|
||||
val startTimestamp: Long
|
||||
get() = firstPositive(startedAt, updatedAt, lastActivityAt)
|
||||
|
||||
private fun firstPositive(vararg values: Long): Long =
|
||||
values.firstOrNull { it > 0L } ?: 0L
|
||||
}
|
||||
|
||||
@@ -30,7 +30,7 @@ data class DashboardConnectionStatus(
|
||||
* open the token store.
|
||||
*
|
||||
* Switching connection is a HEAVY context swap — caller is expected to tear down
|
||||
* the current [com.hermesandroid.relay.network.ConnectionManager],
|
||||
* the current [com.hermesandroid.relay.network.relay.ConnectionManager],
|
||||
* [com.hermesandroid.relay.auth.AuthManager], and API client, then construct
|
||||
* fresh ones pointed at the new connection's `tokenStoreKey`.
|
||||
*
|
||||
@@ -290,6 +290,8 @@ data class Connection(
|
||||
role = role.ifBlank { inferRouteRole(apiServerUrl) },
|
||||
priority = priority,
|
||||
api = ApiEndpoint(host = host, port = port, tls = tls),
|
||||
dashboard = deriveDefaultDashboardUrl(apiServerUrl)
|
||||
?.let { DashboardEndpoint(url = it) },
|
||||
relay = RelayEndpoint(url = resolvedRelayUrl, transportHint = transportHint),
|
||||
)
|
||||
}
|
||||
|
||||
@@ -8,8 +8,8 @@ import androidx.datastore.preferences.core.edit
|
||||
import androidx.datastore.preferences.core.stringPreferencesKey
|
||||
import com.hermesandroid.relay.auth.AuthManager
|
||||
import com.hermesandroid.relay.auth.ConnectionAuthSecrets
|
||||
import com.hermesandroid.relay.network.EncryptedDashboardCookieStore
|
||||
import com.hermesandroid.relay.network.StoredDashboardCookie
|
||||
import com.hermesandroid.relay.network.upstream.EncryptedDashboardCookieStore
|
||||
import com.hermesandroid.relay.network.upstream.StoredDashboardCookie
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.flow.first
|
||||
import kotlinx.coroutines.flow.map
|
||||
@@ -147,6 +147,7 @@ class DataManager(
|
||||
dashboardCookies = EncryptedDashboardCookieStore(
|
||||
context = context,
|
||||
connectionId = connection.id,
|
||||
tokenStoreKey = connection.tokenStoreKey,
|
||||
).load().map { it.toBackup() },
|
||||
)
|
||||
}
|
||||
@@ -185,6 +186,7 @@ class DataManager(
|
||||
EncryptedDashboardCookieStore(
|
||||
context = context,
|
||||
connectionId = connection.id,
|
||||
tokenStoreKey = connection.tokenStoreKey,
|
||||
).save(secret.dashboardCookies.map { it.toStoredCookie() })
|
||||
}
|
||||
}
|
||||
|
||||
@@ -40,6 +40,10 @@ data class EndpointCandidate(
|
||||
val priority: Int = 0,
|
||||
val api: ApiEndpoint,
|
||||
val relay: RelayEndpoint,
|
||||
val dashboard: DashboardEndpoint? = null,
|
||||
val proxy: ProxyEndpoint? = null,
|
||||
val security: String? = null,
|
||||
val recommended: Boolean = false,
|
||||
)
|
||||
|
||||
/**
|
||||
@@ -61,6 +65,17 @@ data class ApiEndpoint(
|
||||
get() = "${if (tls) "https" else "http"}://$host:$port"
|
||||
}
|
||||
|
||||
/**
|
||||
* Dashboard/admin surface for an [EndpointCandidate]. This is optional so
|
||||
* older v3 payloads that only carried API + Relay endpoints keep
|
||||
* deserializing; when absent, Android derives the conventional same-host
|
||||
* `:9119` dashboard URL from [ApiEndpoint].
|
||||
*/
|
||||
@Serializable
|
||||
data class DashboardEndpoint(
|
||||
val url: String,
|
||||
)
|
||||
|
||||
/**
|
||||
* The relay-server half of an [EndpointCandidate] — the WSS URL the phone
|
||||
* opens for the bridge + terminal channels.
|
||||
@@ -78,6 +93,22 @@ data class RelayEndpoint(
|
||||
val transportHint: String? = null,
|
||||
)
|
||||
|
||||
/**
|
||||
* Optional plugin-owned secure proxy route. Unlike [api], [dashboard], and
|
||||
* [relay], this is one app-facing base that can cover all Hermes-Relay
|
||||
* supported traffic after pairing. It is deliberately optional so plugin
|
||||
* proxy support can be advertised by newer payloads without changing the
|
||||
* standard upstream connection model.
|
||||
*/
|
||||
@Serializable
|
||||
data class ProxyEndpoint(
|
||||
val url: String,
|
||||
@SerialName("transport_hint")
|
||||
val transportHint: String? = null,
|
||||
@SerialName("pin_sha256")
|
||||
val pinSha256: String? = null,
|
||||
)
|
||||
|
||||
/**
|
||||
* Returns true when [EndpointCandidate.role] is one of the built-in, styled
|
||||
* roles: `lan`, `tailscale`, or `public`. Case-insensitive match — but the
|
||||
@@ -89,7 +120,7 @@ data class RelayEndpoint(
|
||||
*/
|
||||
fun EndpointCandidate.isKnownRole(): Boolean {
|
||||
return when (role.lowercase()) {
|
||||
"lan", "tailscale", "public" -> true
|
||||
"lan", "tailscale", "public", "plugin_proxy", "plugin-proxy", "https" -> true
|
||||
else -> false
|
||||
}
|
||||
}
|
||||
@@ -106,7 +137,15 @@ fun EndpointCandidate.displayLabel(): String {
|
||||
return when (role.lowercase()) {
|
||||
"lan" -> "LAN"
|
||||
"tailscale" -> "Tailscale"
|
||||
"public" -> "Public"
|
||||
"public" -> if (api.tls) "HTTPS" else "Public"
|
||||
"https" -> "HTTPS"
|
||||
"plugin_proxy", "plugin-proxy" -> "Plugin proxy"
|
||||
else -> "Custom VPN ($role)"
|
||||
}
|
||||
}
|
||||
|
||||
fun EndpointCandidate.hasSecureProxy(): Boolean =
|
||||
proxy?.url?.startsWith("https://", ignoreCase = true) == true ||
|
||||
proxy?.url?.startsWith("wss://", ignoreCase = true) == true ||
|
||||
role.equals("plugin_proxy", ignoreCase = true) ||
|
||||
role.equals("plugin-proxy", ignoreCase = true)
|
||||
|
||||
@@ -11,7 +11,7 @@ import androidx.datastore.preferences.core.edit
|
||||
* Shared by [com.hermesandroid.relay.viewmodel.ConnectionViewModel] (the
|
||||
* StateFlow + setter that drive the foreground service and the client's
|
||||
* no-background-close flag) and
|
||||
* [com.hermesandroid.relay.network.GatewayKeepAliveService]'s Stop notification
|
||||
* [com.hermesandroid.relay.network.upstream.GatewayKeepAliveService]'s Stop notification
|
||||
* action, so both read/write the same key.
|
||||
*/
|
||||
val KEY_GATEWAY_KEEP_ALIVE = booleanPreferencesKey("gateway_keep_alive_background")
|
||||
|
||||
@@ -6,7 +6,7 @@ import kotlinx.serialization.Serializable
|
||||
/**
|
||||
* A rich content card emitted inline in an assistant message via the
|
||||
* `CARD:{json}` line marker. Parsed by
|
||||
* [com.hermesandroid.relay.network.handlers.ChatHandler] and rendered by
|
||||
* [com.hermesandroid.relay.network.upstream.ChatHandler] and rendered by
|
||||
* [com.hermesandroid.relay.ui.components.HermesCardBubble].
|
||||
*
|
||||
* The marker lives in the text stream alongside `MEDIA:...` for the same
|
||||
@@ -226,7 +226,7 @@ data class HermesCardAction(
|
||||
* (with structured `tool_calls`) + `tool` message pairs under a synthetic
|
||||
* `hermes_card_action` tool name, splicing them into the session history
|
||||
* the LLM sees. After the API client takes ownership of the request,
|
||||
* [com.hermesandroid.relay.network.handlers.ChatHandler.markCardDispatchesSynced]
|
||||
* [com.hermesandroid.relay.network.upstream.ChatHandler.markCardDispatchesSynced]
|
||||
* flips [syncedToServer] so subsequent turns don't re-send the same
|
||||
* trace.
|
||||
*/
|
||||
@@ -238,7 +238,7 @@ data class HermesCardDispatch(
|
||||
/**
|
||||
* Idempotency guard for the server-side session sync path.
|
||||
* Flipped to true by
|
||||
* [com.hermesandroid.relay.network.handlers.ChatHandler.markCardDispatchesSynced]
|
||||
* [com.hermesandroid.relay.network.upstream.ChatHandler.markCardDispatchesSynced]
|
||||
* once the API client has accepted the request that carried this
|
||||
* dispatch's synthetic message pair. Once true, the dispatch is
|
||||
* excluded from future
|
||||
|
||||
@@ -4,9 +4,26 @@ import android.content.Context
|
||||
import androidx.datastore.preferences.core.booleanPreferencesKey
|
||||
import androidx.datastore.preferences.core.edit
|
||||
import androidx.datastore.preferences.core.intPreferencesKey
|
||||
import androidx.datastore.preferences.core.stringPreferencesKey
|
||||
import kotlinx.coroutines.flow.Flow
|
||||
import kotlinx.coroutines.flow.map
|
||||
|
||||
/**
|
||||
* How aggressively inbound media is blurred behind a "tap to reveal" gate.
|
||||
*
|
||||
* - [OFF] never blur — show everything immediately.
|
||||
* - [FLAGGED] blur only media the agent flagged sensitive (the model-emitted
|
||||
* `X-Media-Sensitive` bit; see
|
||||
* `docs/plans/2026-06-18-attachment-experience.md` §C). This is
|
||||
* the product default: zero blur when nothing is flagged.
|
||||
* - [ALL_IMAGES] blur every inbound image regardless of source. Works on the
|
||||
* pure standard path with no server support at all.
|
||||
*
|
||||
* Persisted by [Enum.name] so adding cases later is forward-safe; an unknown
|
||||
* stored value decodes back to the default rather than throwing.
|
||||
*/
|
||||
enum class BlurMode { OFF, FLAGGED, ALL_IMAGES }
|
||||
|
||||
/**
|
||||
* User-tunable limits for inbound media attachments fetched from the relay.
|
||||
*
|
||||
@@ -20,12 +37,17 @@ import kotlinx.coroutines.flow.map
|
||||
* - [autoFetchOnCellular] master switch: when false, the cellular-network
|
||||
* case always inserts a manual-download placeholder.
|
||||
* - [cachedMediaCapMb] LRU cap on the `hermes-media/` cache directory.
|
||||
* - [blurSensitive] whether (and which) inbound images render behind a
|
||||
* tap-to-reveal blur — see [BlurMode]. Unlike the four knobs above this one
|
||||
* also applies on the standard (no-Relay) path, since [BlurMode.ALL_IMAGES]
|
||||
* needs no server cooperation.
|
||||
*/
|
||||
data class MediaSettings(
|
||||
val maxInboundSizeMb: Int = 25,
|
||||
val autoFetchThresholdMb: Int = 2,
|
||||
val autoFetchOnCellular: Boolean = false,
|
||||
val cachedMediaCapMb: Int = 200
|
||||
val cachedMediaCapMb: Int = 200,
|
||||
val blurSensitive: BlurMode = BlurMode.FLAGGED
|
||||
)
|
||||
|
||||
/**
|
||||
@@ -39,11 +61,18 @@ class MediaSettingsRepository(private val context: Context) {
|
||||
private val KEY_AUTO_FETCH_THRESHOLD_MB = intPreferencesKey("media_auto_fetch_threshold_mb")
|
||||
private val KEY_AUTO_FETCH_ON_CELLULAR = booleanPreferencesKey("media_auto_fetch_on_cellular")
|
||||
private val KEY_CACHED_MEDIA_CAP_MB = intPreferencesKey("media_cached_cap_mb")
|
||||
private val KEY_BLUR_SENSITIVE = stringPreferencesKey("media_blur_sensitive")
|
||||
|
||||
const val DEFAULT_MAX_INBOUND_MB = 25
|
||||
const val DEFAULT_AUTO_FETCH_THRESHOLD_MB = 2
|
||||
const val DEFAULT_AUTO_FETCH_ON_CELLULAR = false
|
||||
const val DEFAULT_CACHED_MEDIA_CAP_MB = 200
|
||||
val DEFAULT_BLUR_SENSITIVE = BlurMode.FLAGGED
|
||||
|
||||
/** Decode a persisted [BlurMode] name, falling back to the default. */
|
||||
private fun parseBlurMode(raw: String?): BlurMode =
|
||||
raw?.let { name -> BlurMode.entries.firstOrNull { it.name == name } }
|
||||
?: DEFAULT_BLUR_SENSITIVE
|
||||
}
|
||||
|
||||
val settings: Flow<MediaSettings> = context.relayDataStore.data.map { prefs ->
|
||||
@@ -51,10 +80,21 @@ class MediaSettingsRepository(private val context: Context) {
|
||||
maxInboundSizeMb = prefs[KEY_MAX_INBOUND_MB] ?: DEFAULT_MAX_INBOUND_MB,
|
||||
autoFetchThresholdMb = prefs[KEY_AUTO_FETCH_THRESHOLD_MB] ?: DEFAULT_AUTO_FETCH_THRESHOLD_MB,
|
||||
autoFetchOnCellular = prefs[KEY_AUTO_FETCH_ON_CELLULAR] ?: DEFAULT_AUTO_FETCH_ON_CELLULAR,
|
||||
cachedMediaCapMb = prefs[KEY_CACHED_MEDIA_CAP_MB] ?: DEFAULT_CACHED_MEDIA_CAP_MB
|
||||
cachedMediaCapMb = prefs[KEY_CACHED_MEDIA_CAP_MB] ?: DEFAULT_CACHED_MEDIA_CAP_MB,
|
||||
blurSensitive = parseBlurMode(prefs[KEY_BLUR_SENSITIVE])
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Just the blur knob — a standalone flow so per-bubble UI can observe it
|
||||
* without collecting (and recomposing on) the whole [MediaSettings].
|
||||
* Built here (outside composition) on purpose so callers can
|
||||
* `collectAsState()` it without tripping `FlowOperatorInvokedInComposition`.
|
||||
*/
|
||||
val blurMode: Flow<BlurMode> = context.relayDataStore.data.map { prefs ->
|
||||
parseBlurMode(prefs[KEY_BLUR_SENSITIVE])
|
||||
}
|
||||
|
||||
suspend fun setMaxInboundSize(mb: Int) {
|
||||
context.relayDataStore.edit { it[KEY_MAX_INBOUND_MB] = mb.coerceAtLeast(1) }
|
||||
}
|
||||
@@ -70,4 +110,8 @@ class MediaSettingsRepository(private val context: Context) {
|
||||
suspend fun setCachedMediaCap(mb: Int) {
|
||||
context.relayDataStore.edit { it[KEY_CACHED_MEDIA_CAP_MB] = mb.coerceAtLeast(10) }
|
||||
}
|
||||
|
||||
suspend fun setBlurSensitive(mode: BlurMode) {
|
||||
context.relayDataStore.edit { it[KEY_BLUR_SENSITIVE] = mode.name }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
package com.hermesandroid.relay.data
|
||||
|
||||
import android.content.Context
|
||||
import androidx.datastore.core.DataStore
|
||||
import androidx.datastore.preferences.core.Preferences
|
||||
import androidx.datastore.preferences.core.edit
|
||||
import androidx.datastore.preferences.core.stringPreferencesKey
|
||||
import androidx.datastore.preferences.preferencesDataStore
|
||||
import kotlinx.coroutines.flow.Flow
|
||||
import kotlinx.coroutines.flow.map
|
||||
|
||||
/**
|
||||
* Local-only display aliases for agent profiles.
|
||||
*
|
||||
* These names are phone UI labels. They are never sent to Hermes and are keyed
|
||||
* by connection + profile context so the server-default agent can be called
|
||||
* something different on each configured Hermes host.
|
||||
*/
|
||||
class ProfileDisplayAliasStore(
|
||||
private val dataStore: DataStore<Preferences>,
|
||||
) {
|
||||
constructor(context: Context) : this(context.profileDisplayAliasesDataStore)
|
||||
|
||||
companion object {
|
||||
private const val PREFIX = "profile_alias__"
|
||||
|
||||
private fun keyName(connectionId: String, profileName: String?): String =
|
||||
"$PREFIX${connectionId}__${AgentDisplay.profileSessionKey(profileName)}"
|
||||
|
||||
private fun keyFor(connectionId: String, profileName: String?) =
|
||||
stringPreferencesKey(keyName(connectionId, profileName))
|
||||
|
||||
private fun connectionPrefix(connectionId: String): String =
|
||||
"$PREFIX${connectionId}__"
|
||||
}
|
||||
|
||||
suspend fun setAlias(connectionId: String, profileName: String?, alias: String?) {
|
||||
dataStore.edit { prefs ->
|
||||
val key = keyFor(connectionId, profileName)
|
||||
if (alias.isNullOrBlank()) {
|
||||
prefs.remove(key)
|
||||
} else {
|
||||
prefs[key] = alias
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fun aliasFlow(connectionId: String, profileName: String?): Flow<String?> {
|
||||
val key = keyFor(connectionId, profileName)
|
||||
return dataStore.data.map { prefs -> prefs[key] }
|
||||
}
|
||||
|
||||
suspend fun clearConnection(connectionId: String) {
|
||||
val prefix = connectionPrefix(connectionId)
|
||||
dataStore.edit { prefs ->
|
||||
prefs.asMap().keys
|
||||
.filter { it.name.startsWith(prefix) }
|
||||
.forEach { prefs.remove(it) }
|
||||
}
|
||||
}
|
||||
|
||||
suspend fun clearAll() {
|
||||
dataStore.edit { prefs -> prefs.clear() }
|
||||
}
|
||||
}
|
||||
|
||||
internal val Context.profileDisplayAliasesDataStore: DataStore<Preferences>
|
||||
by preferencesDataStore(name = "profile_display_aliases")
|
||||
@@ -0,0 +1,70 @@
|
||||
package com.hermesandroid.relay.data
|
||||
|
||||
import android.content.Context
|
||||
import androidx.datastore.core.DataStore
|
||||
import androidx.datastore.preferences.core.Preferences
|
||||
import androidx.datastore.preferences.core.edit
|
||||
import androidx.datastore.preferences.core.stringPreferencesKey
|
||||
import androidx.datastore.preferences.preferencesDataStore
|
||||
import kotlinx.coroutines.flow.Flow
|
||||
import kotlinx.coroutines.flow.map
|
||||
|
||||
/**
|
||||
* Local-only per-profile agent icons — the visual twin of [ProfileDisplayAliasStore].
|
||||
*
|
||||
* Stores a **file path** to an image that was copied into app storage (not a SAF
|
||||
* content URI, so it survives without a persistable-permission grant). Like the
|
||||
* name alias, these are phone-UI labels only: never sent to Hermes, and keyed by
|
||||
* connection + profile context so the same server-default agent can wear a
|
||||
* different face on each configured host.
|
||||
*/
|
||||
class ProfileIconStore(
|
||||
private val dataStore: DataStore<Preferences>,
|
||||
) {
|
||||
constructor(context: Context) : this(context.profileIconsDataStore)
|
||||
|
||||
companion object {
|
||||
private const val PREFIX = "profile_icon__"
|
||||
|
||||
private fun keyName(connectionId: String, profileName: String?): String =
|
||||
"$PREFIX${connectionId}__${AgentDisplay.profileSessionKey(profileName)}"
|
||||
|
||||
private fun keyFor(connectionId: String, profileName: String?) =
|
||||
stringPreferencesKey(keyName(connectionId, profileName))
|
||||
|
||||
private fun connectionPrefix(connectionId: String): String =
|
||||
"$PREFIX${connectionId}__"
|
||||
}
|
||||
|
||||
suspend fun setIcon(connectionId: String, profileName: String?, path: String?) {
|
||||
dataStore.edit { prefs ->
|
||||
val key = keyFor(connectionId, profileName)
|
||||
if (path.isNullOrBlank()) {
|
||||
prefs.remove(key)
|
||||
} else {
|
||||
prefs[key] = path
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fun iconFlow(connectionId: String, profileName: String?): Flow<String?> {
|
||||
val key = keyFor(connectionId, profileName)
|
||||
return dataStore.data.map { prefs -> prefs[key] }
|
||||
}
|
||||
|
||||
suspend fun clearConnection(connectionId: String) {
|
||||
val prefix = connectionPrefix(connectionId)
|
||||
dataStore.edit { prefs ->
|
||||
prefs.asMap().keys
|
||||
.filter { it.name.startsWith(prefix) }
|
||||
.forEach { prefs.remove(it) }
|
||||
}
|
||||
}
|
||||
|
||||
suspend fun clearAll() {
|
||||
dataStore.edit { prefs -> prefs.clear() }
|
||||
}
|
||||
}
|
||||
|
||||
internal val Context.profileIconsDataStore: DataStore<Preferences>
|
||||
by preferencesDataStore(name = "profile_icons")
|
||||
@@ -8,8 +8,11 @@ import androidx.datastore.preferences.core.longPreferencesKey
|
||||
import androidx.datastore.preferences.core.Preferences
|
||||
import androidx.datastore.preferences.core.stringPreferencesKey
|
||||
import kotlinx.coroutines.flow.Flow
|
||||
import kotlinx.coroutines.flow.MutableStateFlow
|
||||
import kotlinx.coroutines.flow.StateFlow
|
||||
import kotlinx.coroutines.flow.asStateFlow
|
||||
import kotlinx.coroutines.flow.combine
|
||||
import kotlinx.coroutines.flow.distinctUntilChanged
|
||||
import kotlinx.coroutines.flow.map
|
||||
|
||||
/**
|
||||
* User-tunable voice mode preferences.
|
||||
@@ -37,8 +40,60 @@ data class VoiceSettings(
|
||||
* docs/plans/2026-05-24-realtime-persistent-session.md.
|
||||
*/
|
||||
val realtimePersistentSession: Boolean = true,
|
||||
/**
|
||||
* Enhanced-voice overrides for the relay TTS path, mapped onto the active
|
||||
* provider (Gemini / xAI). Empty string / false means "use the server's
|
||||
* saved config" — the relay only applies a field when it is set. Surfaced
|
||||
* in Voice Settings only when the relay advertises an enhanced provider
|
||||
* (`/voice/config` `tts.enhanced.supported`). Field meaning is generic:
|
||||
* `enhancedVoice` → Gemini voice / xAI voice_id; `enhancedAudioTags` →
|
||||
* Gemini audio_tags / xAI auto_speech_tags; `enhancedPersona` is Gemini-only
|
||||
* and `enhancedLanguage` is xAI-only.
|
||||
*/
|
||||
val enhancedVoice: String = "",
|
||||
val enhancedModel: String = "",
|
||||
val enhancedAudioTags: Boolean = false,
|
||||
val enhancedPersona: String = "",
|
||||
val enhancedLanguage: String = "",
|
||||
)
|
||||
|
||||
/**
|
||||
* Per-request enhanced-voice overrides forwarded to the relay's
|
||||
* `/voice/synthesize`. Mirrors the generic fields recognized by
|
||||
* `plugin/relay/voice.py:_extract_voice_overrides`; the relay maps them onto
|
||||
* the active provider's config.
|
||||
*/
|
||||
data class EnhancedVoiceOverrides(
|
||||
val voice: String? = null,
|
||||
val model: String? = null,
|
||||
val audioTags: Boolean? = null,
|
||||
val personaPrompt: String? = null,
|
||||
val language: String? = null,
|
||||
) {
|
||||
val isEmpty: Boolean
|
||||
get() = voice == null && model == null && audioTags == null &&
|
||||
personaPrompt == null && language == null
|
||||
|
||||
companion object {
|
||||
/**
|
||||
* Build overrides from persisted settings, or null when nothing is set
|
||||
* (so the relay falls back to the server's saved config). The audio-tags
|
||||
* toggle only sends `true` — leaving it off defers to the server default
|
||||
* rather than forcing it off.
|
||||
*/
|
||||
fun fromSettings(s: VoiceSettings): EnhancedVoiceOverrides? {
|
||||
val overrides = EnhancedVoiceOverrides(
|
||||
voice = s.enhancedVoice.takeIf { it.isNotBlank() },
|
||||
model = s.enhancedModel.takeIf { it.isNotBlank() },
|
||||
audioTags = true.takeIf { s.enhancedAudioTags },
|
||||
personaPrompt = s.enhancedPersona.takeIf { it.isNotBlank() },
|
||||
language = s.enhancedLanguage.takeIf { it.isNotBlank() },
|
||||
)
|
||||
return overrides.takeUnless { it.isEmpty }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
enum class VoiceEngineMode(val storageValue: String) {
|
||||
HermesVoiceOutput("hermes_voice_output"),
|
||||
RealtimeAgent("realtime_agent");
|
||||
@@ -60,13 +115,63 @@ enum class VoiceAudioRoute(val storageValue: String) {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Active scope for per-profile voice prefs.
|
||||
*
|
||||
* Mirrors [ProfileSelectionStore]'s `_<connectionId>` keying and extends it to
|
||||
* `_<connectionId>_<profile>` so per-profile voice picks don't leak across
|
||||
* profiles (or across connections that expose a same-named profile).
|
||||
*
|
||||
* A null/blank [profileName] is the "default / launch profile" and resolves to
|
||||
* the un-namespaced global keys — i.e. the default profile *is* the base layer
|
||||
* that named profiles override. A null/blank [connectionId] degrades to
|
||||
* profile-only namespacing, which still isolates profiles within one
|
||||
* connection; it just can't disambiguate two connections with a same-named
|
||||
* profile. See [VoicePreferencesRepository.setActiveScope].
|
||||
*/
|
||||
data class VoiceProfileScope(
|
||||
val connectionId: String? = null,
|
||||
val profileName: String? = null,
|
||||
) {
|
||||
companion object {
|
||||
val Global = VoiceProfileScope()
|
||||
}
|
||||
}
|
||||
|
||||
class VoicePreferencesRepository(private val dataStore: DataStore<Preferences>) {
|
||||
|
||||
constructor(context: Context) : this(context.relayDataStore)
|
||||
|
||||
companion object {
|
||||
private val KEY_ENGINE_MODE = stringPreferencesKey("voice_engine_mode")
|
||||
private val KEY_AUDIO_ROUTE = stringPreferencesKey("voice_audio_route")
|
||||
// --- Per-profile keys (override map; namespaced by active scope) -----
|
||||
// These are stored as base NAME strings (not typed Key<>s) so the
|
||||
// scoped key can be built per (connectionId, profile) at read/write
|
||||
// time. Resolution layers a per-profile value over the global value
|
||||
// over the hard default — see [scopedName] / [resolveString].
|
||||
//
|
||||
// Why these are per-profile: engine mode, audio route, and the
|
||||
// enhanced-voice overrides describe *which voice the agent speaks
|
||||
// with*, which is a property of the profile (the relay already
|
||||
// persists `voice_output:`/`realtime_voice:` per profile and
|
||||
// `RelayVoiceClient` already sends `?profile=`). Keeping them global
|
||||
// leaked one profile's voice onto every other profile.
|
||||
private const val KEY_ENGINE_MODE = "voice_engine_mode"
|
||||
private const val KEY_AUDIO_ROUTE = "voice_audio_route"
|
||||
private const val KEY_ENH_VOICE = "voice_enh_voice"
|
||||
private const val KEY_ENH_MODEL = "voice_enh_model"
|
||||
private const val KEY_ENH_AUDIO_TAGS = "voice_enh_audio_tags"
|
||||
private const val KEY_ENH_PERSONA = "voice_enh_persona"
|
||||
private const val KEY_ENH_LANGUAGE = "voice_enh_language"
|
||||
|
||||
// --- Global keys (shared across profiles; never namespaced) ----------
|
||||
// Why these stay global: interaction-mode and silence-threshold are
|
||||
// ergonomic input preferences about *how the user drives the mic*, not
|
||||
// about the agent's voice — a user wants the same tap/hold/continuous
|
||||
// habit regardless of which profile is active. auto-tts and the STT
|
||||
// language hint are dead/experimental controls today, and the two
|
||||
// realtime diagnostic toggles (trace details, persistent session) are
|
||||
// engine-behaviour switches that aren't profile-specific. Keeping them
|
||||
// un-namespaced means switching profiles never churns these.
|
||||
private val KEY_INTERACTION_MODE = stringPreferencesKey("voice_interaction_mode")
|
||||
private val KEY_SILENCE_THRESHOLD_MS = longPreferencesKey("voice_silence_threshold_ms")
|
||||
private val KEY_AUTO_TTS = booleanPreferencesKey("voice_auto_tts")
|
||||
@@ -83,37 +188,149 @@ class VoicePreferencesRepository(private val dataStore: DataStore<Preferences>)
|
||||
const val DEFAULT_LANGUAGE = ""
|
||||
const val DEFAULT_REALTIME_TRACE_DETAILS = false
|
||||
const val DEFAULT_REALTIME_PERSISTENT_SESSION = true
|
||||
|
||||
/**
|
||||
* Build the storage name for a per-profile [base] key under [scope].
|
||||
*
|
||||
* - null/blank profile → returns [base] verbatim (the global base
|
||||
* layer; the default profile reads/writes the un-namespaced key).
|
||||
* - profile set, no connection → `<base>_<profile>`.
|
||||
* - profile + connection set → `<base>_<connectionId>_<profile>`,
|
||||
* matching [ProfileSelectionStore]'s connection-first ordering.
|
||||
*/
|
||||
internal fun scopedName(base: String, scope: VoiceProfileScope): String {
|
||||
val profile = scope.profileName?.trim()?.takeIf { it.isNotEmpty() } ?: return base
|
||||
val conn = scope.connectionId?.trim()?.takeIf { it.isNotEmpty() }
|
||||
return if (conn != null) "${base}_${conn}_$profile" else "${base}_$profile"
|
||||
}
|
||||
}
|
||||
|
||||
val settings: Flow<VoiceSettings> = dataStore.data
|
||||
.map { prefs ->
|
||||
VoiceSettings(
|
||||
engineMode = VoiceEngineMode.fromStorage(
|
||||
prefs[KEY_ENGINE_MODE] ?: DEFAULT_ENGINE_MODE,
|
||||
).storageValue,
|
||||
audioRoute = VoiceAudioRoute.fromStorage(
|
||||
prefs[KEY_AUDIO_ROUTE] ?: DEFAULT_AUDIO_ROUTE,
|
||||
).storageValue,
|
||||
interactionMode = prefs[KEY_INTERACTION_MODE] ?: DEFAULT_INTERACTION_MODE,
|
||||
silenceThresholdMs = prefs[KEY_SILENCE_THRESHOLD_MS] ?: DEFAULT_SILENCE_THRESHOLD_MS,
|
||||
autoTts = prefs[KEY_AUTO_TTS] ?: DEFAULT_AUTO_TTS,
|
||||
language = prefs[KEY_LANGUAGE] ?: DEFAULT_LANGUAGE,
|
||||
realtimeTraceDetails = prefs[KEY_REALTIME_TRACE_DETAILS]
|
||||
?: DEFAULT_REALTIME_TRACE_DETAILS,
|
||||
realtimePersistentSession = prefs[KEY_REALTIME_PERSISTENT_SESSION]
|
||||
?: DEFAULT_REALTIME_PERSISTENT_SESSION,
|
||||
)
|
||||
// In-memory active scope. Defaults to global so un-scoped consumers (and
|
||||
// every existing call site) behave exactly as before until a scope is set.
|
||||
private val _scope = MutableStateFlow(VoiceProfileScope.Global)
|
||||
|
||||
/** The active per-profile scope. Set via [setActiveScope]. */
|
||||
val activeScope: StateFlow<VoiceProfileScope> = _scope.asStateFlow()
|
||||
|
||||
/**
|
||||
* Point the repository at a (connection, profile) scope. Per-profile reads
|
||||
* and writes (engine/route/enhanced) re-target the namespaced keys for that
|
||||
* profile; global prefs are unaffected. Passing a null/blank profile name
|
||||
* reverts per-profile reads/writes to the global base layer (the default
|
||||
* profile). Idempotent — a no-op when the normalized scope is unchanged.
|
||||
*/
|
||||
fun setActiveScope(connectionId: String?, profileName: String?) {
|
||||
val next = VoiceProfileScope(
|
||||
connectionId = connectionId?.trim()?.takeIf { it.isNotEmpty() },
|
||||
profileName = profileName?.trim()?.takeIf { it.isNotEmpty() },
|
||||
)
|
||||
if (_scope.value != next) {
|
||||
_scope.value = next
|
||||
}
|
||||
.distinctUntilChanged()
|
||||
}
|
||||
|
||||
/**
|
||||
* Emits the resolved [VoiceSettings] for the [activeScope]. Re-emits when
|
||||
* either the underlying DataStore or the active scope changes. Per-profile
|
||||
* fields are resolved as: per-profile key → global key → hard default.
|
||||
*/
|
||||
val settings: Flow<VoiceSettings> = combine(_scope, dataStore.data) { scope, prefs ->
|
||||
VoiceSettings(
|
||||
// --- per-profile (override map) ---
|
||||
engineMode = VoiceEngineMode.fromStorage(
|
||||
resolveString(prefs, KEY_ENGINE_MODE, scope, DEFAULT_ENGINE_MODE),
|
||||
).storageValue,
|
||||
audioRoute = VoiceAudioRoute.fromStorage(
|
||||
resolveString(prefs, KEY_AUDIO_ROUTE, scope, DEFAULT_AUDIO_ROUTE),
|
||||
).storageValue,
|
||||
enhancedVoice = resolveString(prefs, KEY_ENH_VOICE, scope, ""),
|
||||
enhancedModel = resolveString(prefs, KEY_ENH_MODEL, scope, ""),
|
||||
enhancedAudioTags = resolveBoolean(prefs, KEY_ENH_AUDIO_TAGS, scope, false),
|
||||
enhancedPersona = resolveString(prefs, KEY_ENH_PERSONA, scope, ""),
|
||||
enhancedLanguage = resolveString(prefs, KEY_ENH_LANGUAGE, scope, ""),
|
||||
// --- global (shared across profiles) ---
|
||||
interactionMode = prefs[KEY_INTERACTION_MODE] ?: DEFAULT_INTERACTION_MODE,
|
||||
silenceThresholdMs = prefs[KEY_SILENCE_THRESHOLD_MS] ?: DEFAULT_SILENCE_THRESHOLD_MS,
|
||||
autoTts = prefs[KEY_AUTO_TTS] ?: DEFAULT_AUTO_TTS,
|
||||
language = prefs[KEY_LANGUAGE] ?: DEFAULT_LANGUAGE,
|
||||
realtimeTraceDetails = prefs[KEY_REALTIME_TRACE_DETAILS]
|
||||
?: DEFAULT_REALTIME_TRACE_DETAILS,
|
||||
realtimePersistentSession = prefs[KEY_REALTIME_PERSISTENT_SESSION]
|
||||
?: DEFAULT_REALTIME_PERSISTENT_SESSION,
|
||||
)
|
||||
}.distinctUntilChanged()
|
||||
|
||||
// --- per-profile resolution (per-profile key → global key → default) -----
|
||||
|
||||
private fun resolveString(
|
||||
prefs: Preferences,
|
||||
base: String,
|
||||
scope: VoiceProfileScope,
|
||||
default: String,
|
||||
): String {
|
||||
val scopedName = scopedName(base, scope)
|
||||
if (scopedName != base) {
|
||||
prefs[stringPreferencesKey(scopedName)]?.let { return it }
|
||||
}
|
||||
return prefs[stringPreferencesKey(base)] ?: default
|
||||
}
|
||||
|
||||
private fun resolveBoolean(
|
||||
prefs: Preferences,
|
||||
base: String,
|
||||
scope: VoiceProfileScope,
|
||||
default: Boolean,
|
||||
): Boolean {
|
||||
val scopedName = scopedName(base, scope)
|
||||
if (scopedName != base) {
|
||||
prefs[booleanPreferencesKey(scopedName)]?.let { return it }
|
||||
}
|
||||
return prefs[booleanPreferencesKey(base)] ?: default
|
||||
}
|
||||
|
||||
// --- per-profile setters (write the namespaced key for the active scope) -
|
||||
|
||||
suspend fun setEngineMode(mode: VoiceEngineMode) {
|
||||
dataStore.edit { it[KEY_ENGINE_MODE] = mode.storageValue }
|
||||
val key = stringPreferencesKey(scopedName(KEY_ENGINE_MODE, _scope.value))
|
||||
dataStore.edit { it[key] = mode.storageValue }
|
||||
}
|
||||
|
||||
suspend fun setAudioRoute(route: VoiceAudioRoute) {
|
||||
dataStore.edit { it[KEY_AUDIO_ROUTE] = route.storageValue }
|
||||
val key = stringPreferencesKey(scopedName(KEY_AUDIO_ROUTE, _scope.value))
|
||||
dataStore.edit { it[key] = route.storageValue }
|
||||
}
|
||||
|
||||
/** "" clears the override (relay falls back to the server's saved voice). */
|
||||
suspend fun setEnhancedVoice(voice: String) {
|
||||
val key = stringPreferencesKey(scopedName(KEY_ENH_VOICE, _scope.value))
|
||||
dataStore.edit { it[key] = voice.trim() }
|
||||
}
|
||||
|
||||
/** "" clears the override (relay falls back to the server's saved model). */
|
||||
suspend fun setEnhancedModel(model: String) {
|
||||
val key = stringPreferencesKey(scopedName(KEY_ENH_MODEL, _scope.value))
|
||||
dataStore.edit { it[key] = model.trim() }
|
||||
}
|
||||
|
||||
suspend fun setEnhancedAudioTags(enabled: Boolean) {
|
||||
val key = booleanPreferencesKey(scopedName(KEY_ENH_AUDIO_TAGS, _scope.value))
|
||||
dataStore.edit { it[key] = enabled }
|
||||
}
|
||||
|
||||
/** "" clears the inline persona/style direction (Gemini). */
|
||||
suspend fun setEnhancedPersona(persona: String) {
|
||||
val key = stringPreferencesKey(scopedName(KEY_ENH_PERSONA, _scope.value))
|
||||
dataStore.edit { it[key] = persona }
|
||||
}
|
||||
|
||||
/** "" clears the language override (xAI). */
|
||||
suspend fun setEnhancedLanguage(language: String) {
|
||||
val key = stringPreferencesKey(scopedName(KEY_ENH_LANGUAGE, _scope.value))
|
||||
dataStore.edit { it[key] = language.trim() }
|
||||
}
|
||||
|
||||
// --- global setters (always the un-namespaced key) -----------------------
|
||||
|
||||
suspend fun setInteractionMode(mode: String) {
|
||||
dataStore.edit { it[KEY_INTERACTION_MODE] = mode }
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network.handlers
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import android.content.ActivityNotFoundException
|
||||
import android.content.ClipData
|
||||
@@ -23,9 +23,10 @@ import kotlinx.serialization.json.booleanOrNull
|
||||
// === PHASE3-tier-C: flavor gate for sideload-only tools ===
|
||||
import com.hermesandroid.relay.data.BuildFlavor
|
||||
// === END PHASE3-tier-C ===
|
||||
import com.hermesandroid.relay.network.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.RelayHttpClient
|
||||
import com.hermesandroid.relay.network.models.Envelope
|
||||
import com.hermesandroid.relay.network.relay.ChannelMultiplexer
|
||||
import com.hermesandroid.relay.network.relay.RelayHttpClient
|
||||
import com.hermesandroid.relay.network.relay.models.Envelope
|
||||
import com.hermesandroid.relay.network.shared.LocalDispatchResult
|
||||
import com.hermesandroid.relay.util.MediaCacheWriter
|
||||
import kotlin.coroutines.AbstractCoroutineContextElement
|
||||
import kotlin.coroutines.CoroutineContext
|
||||
@@ -2418,33 +2419,9 @@ class BridgeCommandHandler(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Captured outcome of a local bridge dispatch. Voice mode reads this to
|
||||
* emit follow-up chat traces showing the real success/failure state of
|
||||
* an action after the safety modal resolves and the underlying
|
||||
* [ActionExecutor] method returns. The fields mirror what the LLM path
|
||||
* would see on a `bridge.response` envelope:
|
||||
*
|
||||
* - [status] — HTTP-style status: 200 success, 400 client error,
|
||||
* 403 user denial / bridge disabled, 500 executor error
|
||||
* - [errorMessage] — free-text error from the response payload, or null
|
||||
* on success. Safe to speak / display verbatim to the user.
|
||||
* - [errorCode] — structured classification (e.g. `permission_denied`,
|
||||
* `bridge_disabled`, `user_denied`) when `respondFromResult` or a
|
||||
* direct respond call includes one. Null for errors we haven't
|
||||
* classified yet.
|
||||
* - [resultJson] — the raw result object, for callers that need
|
||||
* action-specific fields (e.g. the resolved phone number from
|
||||
* /search_contacts). Optional.
|
||||
*/
|
||||
data class LocalDispatchResult(
|
||||
val status: Int,
|
||||
val errorMessage: String?,
|
||||
val errorCode: String?,
|
||||
val resultJson: JsonObject?,
|
||||
) {
|
||||
val isSuccess: Boolean get() = status in 200..299
|
||||
}
|
||||
// LocalDispatchResult moved to network.shared (ADR 34 fence): it is a passive
|
||||
// DTO shared with the upstream chat path (ChatHandler), so it cannot live in
|
||||
// this relay-package file without creating an upstream -> relay import.
|
||||
|
||||
/**
|
||||
* Coroutine context marker installed by [BridgeCommandHandler.handleLocalCommand]
|
||||
@@ -1,6 +1,6 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import com.hermesandroid.relay.network.models.Envelope
|
||||
import com.hermesandroid.relay.network.relay.models.Envelope
|
||||
import kotlinx.serialization.json.Json
|
||||
import kotlinx.serialization.json.buildJsonObject
|
||||
import kotlinx.serialization.json.put
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import android.content.Context
|
||||
import android.net.ConnectivityManager
|
||||
@@ -12,7 +12,8 @@ import com.hermesandroid.relay.data.PairingPreferences
|
||||
import com.hermesandroid.relay.diagnostics.DiagnosticCategory
|
||||
import com.hermesandroid.relay.diagnostics.DiagnosticSeverity
|
||||
import com.hermesandroid.relay.diagnostics.DiagnosticsLog
|
||||
import com.hermesandroid.relay.network.models.Envelope
|
||||
import com.hermesandroid.relay.network.relay.models.Envelope
|
||||
import com.hermesandroid.relay.network.shared.EndpointResolver
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.SupervisorJob
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import android.util.Log
|
||||
import com.hermesandroid.relay.auth.PairedDeviceInfo
|
||||
@@ -7,6 +7,7 @@ import com.hermesandroid.relay.diagnostics.DiagnosticSeverity
|
||||
import com.hermesandroid.relay.diagnostics.DiagnosticsLog
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.Serializable
|
||||
import kotlinx.serialization.builtins.ListSerializer
|
||||
import kotlinx.serialization.json.Json
|
||||
import kotlinx.serialization.json.jsonObject
|
||||
@@ -22,7 +23,7 @@ import java.io.IOException
|
||||
*
|
||||
* The chat SSE stream can emit tool output containing a marker of the form
|
||||
* `MEDIA:hermes-relay://<opaque-token>`
|
||||
* [ChatHandler][com.hermesandroid.relay.network.handlers.ChatHandler] parses
|
||||
* [ChatHandler][com.hermesandroid.relay.network.upstream.ChatHandler] parses
|
||||
* the marker, and [ChatViewModel][com.hermesandroid.relay.viewmodel.ChatViewModel]
|
||||
* calls [fetchMedia] to pull the actual bytes over plain HTTP(S). The relay
|
||||
* base URL is the WSS relay URL with `ws`/`wss` swapped for `http`/`https`.
|
||||
@@ -40,7 +41,11 @@ import java.io.IOException
|
||||
class RelayHttpClient(
|
||||
private val okHttpClient: OkHttpClient,
|
||||
private val relayUrlProvider: () -> String?,
|
||||
private val sessionTokenProvider: suspend () -> String?
|
||||
private val sessionTokenProvider: suspend () -> String?,
|
||||
/** Synchronous snapshot of the paired session token (null when not currently
|
||||
* paired). Lets [mediaUrlConfigured] check fetch-readiness without
|
||||
* suspending; mirrors what [sessionTokenProvider] resolves. */
|
||||
private val pairedTokenSnapshot: () -> String? = { null },
|
||||
) {
|
||||
|
||||
companion object {
|
||||
@@ -53,6 +58,19 @@ class RelayHttpClient(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* True when relay media is actually FETCHABLE right now: a non-blank relay
|
||||
* URL AND a current paired session token. Synchronous. The token check
|
||||
* matters because the relay's SessionManager is in-memory and wiped on
|
||||
* restart, so a configured relay URL can outlive the pairing — gating on URL
|
||||
* alone made the media-capability badge read "available" while every
|
||||
* `/media/by-path` fetch failed for a missing token. Now the badge (and the
|
||||
* SSE media hint) agree with what the fetch can do, and self-correct on
|
||||
* re-pair.
|
||||
*/
|
||||
fun mediaUrlConfigured(): Boolean =
|
||||
!relayUrlProvider().isNullOrBlank() && !pairedTokenSnapshot().isNullOrBlank()
|
||||
|
||||
/**
|
||||
* The result of a successful [fetchMedia] call.
|
||||
*
|
||||
@@ -61,28 +79,53 @@ class RelayHttpClient(
|
||||
* @property bytes raw response body.
|
||||
* @property fileName best-effort filename parsed from
|
||||
* `Content-Disposition: inline; filename="..."`, or null.
|
||||
* @property sensitive model-emitted sensitivity hint, read from the
|
||||
* relay's `X-Media-Sensitive` response header (`"1"`/`"true"`
|
||||
* → true). The relay never classifies media — it transports
|
||||
* whatever the producing tool/agent declared. Absent header →
|
||||
* false. Consumed by `ChatViewModel` to blur per the user's
|
||||
* setting.
|
||||
*/
|
||||
data class FetchedMedia(
|
||||
val contentType: String,
|
||||
val bytes: ByteArray,
|
||||
val fileName: String?
|
||||
val fileName: String?,
|
||||
val sensitive: Boolean = false
|
||||
) {
|
||||
override fun equals(other: Any?): Boolean {
|
||||
if (this === other) return true
|
||||
if (other !is FetchedMedia) return false
|
||||
return contentType == other.contentType &&
|
||||
bytes.contentEquals(other.bytes) &&
|
||||
fileName == other.fileName
|
||||
fileName == other.fileName &&
|
||||
sensitive == other.sensitive
|
||||
}
|
||||
|
||||
override fun hashCode(): Int {
|
||||
var result = contentType.hashCode()
|
||||
result = 31 * result + bytes.contentHashCode()
|
||||
result = 31 * result + (fileName?.hashCode() ?: 0)
|
||||
result = 31 * result + sensitive.hashCode()
|
||||
return result
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Server-side relay context that would be injected into the next agent turn.
|
||||
* Mirrors `GET /context/injected`; Android treats it as audit-only state.
|
||||
*/
|
||||
@Serializable
|
||||
data class InjectedContextAudit(
|
||||
val enabled: Boolean = false,
|
||||
val blocks: List<InjectedContextBlock> = emptyList(),
|
||||
)
|
||||
|
||||
@Serializable
|
||||
data class InjectedContextBlock(
|
||||
val name: String,
|
||||
val text: String,
|
||||
)
|
||||
|
||||
/**
|
||||
* Fetch `GET /media/<token>` from the relay over HTTP(S). Returns a
|
||||
* [Result] — success carries a [FetchedMedia], failure wraps the
|
||||
@@ -141,12 +184,16 @@ class RelayHttpClient(
|
||||
response.header("Content-Disposition")
|
||||
)
|
||||
|
||||
val sensitive = parseSensitiveHeader(
|
||||
response.header("X-Media-Sensitive")
|
||||
)
|
||||
|
||||
val body = response.body
|
||||
if (body == null) {
|
||||
return@withContext Result.failure(IOException("Empty response body"))
|
||||
}
|
||||
val bytes = body.bytes()
|
||||
Result.success(FetchedMedia(contentType, bytes, fileName))
|
||||
Result.success(FetchedMedia(contentType, bytes, fileName, sensitive))
|
||||
}
|
||||
} catch (e: IOException) {
|
||||
Log.w(TAG, "fetchMedia failed for $token: ${e.message}")
|
||||
@@ -248,12 +295,16 @@ class RelayHttpClient(
|
||||
response.header("Content-Disposition")
|
||||
)
|
||||
|
||||
val sensitive = parseSensitiveHeader(
|
||||
response.header("X-Media-Sensitive")
|
||||
)
|
||||
|
||||
val body = response.body
|
||||
if (body == null) {
|
||||
return@withContext Result.failure(IOException("Empty response body"))
|
||||
}
|
||||
val bytes = body.bytes()
|
||||
Result.success(FetchedMedia(contentType, bytes, fileName))
|
||||
Result.success(FetchedMedia(contentType, bytes, fileName, sensitive))
|
||||
}
|
||||
} catch (e: IOException) {
|
||||
Log.w(TAG, "fetchMediaByPath failed for $path: ${e.message}")
|
||||
@@ -264,6 +315,85 @@ class RelayHttpClient(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch the relay's server-side injected-context audit. This endpoint is
|
||||
* optional and fail-open: old/plugin-absent relays return an empty disabled
|
||||
* audit rather than breaking the client-side context sheet.
|
||||
*/
|
||||
suspend fun fetchInjectedContext(): Result<InjectedContextAudit> = withContext(Dispatchers.IO) {
|
||||
val relayUrl = relayUrlProvider()?.trim().orEmpty()
|
||||
if (relayUrl.isEmpty()) {
|
||||
return@withContext Result.failure(
|
||||
IllegalStateException("Relay URL not configured")
|
||||
)
|
||||
}
|
||||
|
||||
val sessionToken = sessionTokenProvider()
|
||||
if (sessionToken.isNullOrBlank()) {
|
||||
return@withContext Result.failure(
|
||||
IllegalStateException("Relay not paired — session token missing")
|
||||
)
|
||||
}
|
||||
|
||||
val httpBase = relayUrl
|
||||
.replace(Regex("^wss://", RegexOption.IGNORE_CASE), "https://")
|
||||
.replace(Regex("^ws://", RegexOption.IGNORE_CASE), "http://")
|
||||
.trimEnd('/')
|
||||
|
||||
val url = try {
|
||||
"$httpBase/context/injected".toHttpUrl()
|
||||
} catch (e: IllegalArgumentException) {
|
||||
return@withContext Result.failure(
|
||||
IOException("Invalid relay URL: ${e.message}")
|
||||
)
|
||||
}
|
||||
|
||||
val request = Request.Builder()
|
||||
.url(url)
|
||||
.get()
|
||||
.header("Authorization", "Bearer $sessionToken")
|
||||
.header("Accept", "application/json")
|
||||
.build()
|
||||
|
||||
val auditClient = okHttpClient.newBuilder()
|
||||
.callTimeout(3, java.util.concurrent.TimeUnit.SECONDS)
|
||||
.build()
|
||||
|
||||
try {
|
||||
auditClient.newCall(request).execute().use { response ->
|
||||
if (response.code == 404) {
|
||||
return@withContext Result.success(InjectedContextAudit())
|
||||
}
|
||||
if (!response.isSuccessful) {
|
||||
val reason = when (response.code) {
|
||||
401, 403 -> "Unauthorized — re-pair with the relay"
|
||||
in 500..599 -> "Relay error (HTTP ${response.code})"
|
||||
else -> "HTTP ${response.code}: ${response.message.ifBlank { "request failed" }}"
|
||||
}
|
||||
return@withContext Result.failure(IOException(reason))
|
||||
}
|
||||
|
||||
val body = response.body?.string().orEmpty()
|
||||
if (body.isBlank()) {
|
||||
return@withContext Result.failure(IOException("Empty response body"))
|
||||
}
|
||||
|
||||
Result.success(
|
||||
sessionsJson.decodeFromString(
|
||||
InjectedContextAudit.serializer(),
|
||||
body,
|
||||
)
|
||||
)
|
||||
}
|
||||
} catch (e: IOException) {
|
||||
Log.w(TAG, "fetchInjectedContext failed: ${e.message}")
|
||||
Result.failure(e)
|
||||
} catch (e: Exception) {
|
||||
Log.w(TAG, "fetchInjectedContext parse error: ${e.message}")
|
||||
Result.failure(e)
|
||||
}
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// Paired-device management (2026-04-11 security overhaul)
|
||||
// ------------------------------------------------------------------
|
||||
@@ -777,4 +907,16 @@ class RelayHttpClient(
|
||||
val match = Regex("""filename\s*=\s*"?([^";]+)"?""", RegexOption.IGNORE_CASE).find(header)
|
||||
return match?.groupValues?.get(1)?.trim()?.ifBlank { null }
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the relay's `X-Media-Sensitive` response header into a bool.
|
||||
*
|
||||
* The relay emits the header only when the media was flagged sensitive,
|
||||
* with value `"1"` (and tolerates `"true"`). Any other value — or an
|
||||
* absent header — means "not sensitive", so when in doubt we don't blur.
|
||||
*/
|
||||
private fun parseSensitiveHeader(header: String?): Boolean {
|
||||
val value = header?.trim()?.lowercase() ?: return false
|
||||
return value == "1" || value == "true"
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import android.util.Log
|
||||
import com.hermesandroid.relay.data.ProfileConfigResponse
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import java.net.URI
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import com.hermesandroid.relay.data.EnhancedVoiceOverrides
|
||||
import com.hermesandroid.relay.data.VoiceAudioRoute
|
||||
import com.hermesandroid.relay.network.shared.VoiceAudioClient
|
||||
import java.io.File
|
||||
|
||||
/**
|
||||
* Adapts the relay-only [RelayVoiceClient] (same package) to the neutral
|
||||
* [VoiceAudioClient] routing seam in `network.shared`. Relay → shared is an
|
||||
* allowed dependency direction under the ADR 34 package fence.
|
||||
*/
|
||||
class RelayVoiceAudioClientAdapter(
|
||||
private val relayVoiceClient: RelayVoiceClient,
|
||||
private val enhancedOverridesProvider: () -> EnhancedVoiceOverrides? = { null },
|
||||
) : VoiceAudioClient {
|
||||
override val route: VoiceAudioRoute = VoiceAudioRoute.Relay
|
||||
|
||||
override suspend fun transcribe(audioFile: File): Result<String> =
|
||||
relayVoiceClient.transcribe(audioFile)
|
||||
|
||||
override suspend fun synthesize(text: String): Result<File> =
|
||||
relayVoiceClient.synthesize(text, enhancedOverridesProvider())
|
||||
}
|
||||
@@ -1,7 +1,8 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.relay
|
||||
|
||||
import android.content.Context
|
||||
import android.util.Log
|
||||
import com.hermesandroid.relay.data.EnhancedVoiceOverrides
|
||||
import com.hermesandroid.relay.data.MessageRole
|
||||
import com.hermesandroid.relay.data.RealtimeConversationContextMessage
|
||||
import kotlinx.coroutines.CompletableDeferred
|
||||
@@ -208,7 +209,10 @@ class RelayVoiceClient(
|
||||
* when done (typical pattern: keep the last N mp3s in the cache dir and
|
||||
* let the OS reclaim on cache pressure).
|
||||
*/
|
||||
suspend fun synthesize(text: String): Result<File> = withContext(Dispatchers.IO) {
|
||||
suspend fun synthesize(
|
||||
text: String,
|
||||
enhanced: EnhancedVoiceOverrides? = null,
|
||||
): Result<File> = withContext(Dispatchers.IO) {
|
||||
val httpBase = resolveHttpBase()
|
||||
?: return@withContext Result.failure(IllegalStateException("Relay URL not configured"))
|
||||
val token = resolveBearerToken()
|
||||
@@ -223,6 +227,16 @@ class RelayVoiceClient(
|
||||
|
||||
val bodyJson = buildJsonObject {
|
||||
put("text", JsonPrimitive(text))
|
||||
// Per-request enhanced-voice overrides. The relay maps these generic
|
||||
// fields onto the active provider (Gemini/xAI) and ignores them for
|
||||
// others — see voice.py:_extract_voice_overrides.
|
||||
enhanced?.let { ov ->
|
||||
ov.voice?.let { put("voice", JsonPrimitive(it)) }
|
||||
ov.model?.let { put("model", JsonPrimitive(it)) }
|
||||
ov.audioTags?.let { put("audio_tags", JsonPrimitive(it)) }
|
||||
ov.personaPrompt?.let { put("persona_prompt", JsonPrimitive(it)) }
|
||||
ov.language?.let { put("language", JsonPrimitive(it)) }
|
||||
}
|
||||
putProfile()
|
||||
}.toString()
|
||||
|
||||
@@ -724,6 +738,7 @@ class RelayVoiceClient(
|
||||
codec: String? = null,
|
||||
optimizeStreamingLatency: Int? = null,
|
||||
textNormalization: Boolean? = null,
|
||||
autoSpeechTags: Boolean? = null,
|
||||
fallbackEnabled: Boolean? = null,
|
||||
): Result<VoiceOutputConfig> = withContext(Dispatchers.IO) {
|
||||
val httpBase = resolveHttpBase()
|
||||
@@ -755,6 +770,7 @@ class RelayVoiceClient(
|
||||
put("optimize_streaming_latency", JsonPrimitive(it))
|
||||
}
|
||||
textNormalization?.let { put("text_normalization", JsonPrimitive(it)) }
|
||||
autoSpeechTags?.let { put("auto_speech_tags", JsonPrimitive(it)) }
|
||||
fallbackEnabled?.let { put("fallback_enabled", JsonPrimitive(it)) }
|
||||
}
|
||||
|
||||
@@ -2304,11 +2320,44 @@ data class VoiceProviderInfo(
|
||||
val voiceId: String? = null,
|
||||
val enabled: Boolean = false,
|
||||
val available: Boolean = true,
|
||||
/**
|
||||
* Provider-specific enhanced-voice capability hint. Present (non-null) only
|
||||
* for the TTS block when the relay's active provider supports per-request
|
||||
* enhanced control (today: Gemini and xAI).
|
||||
*/
|
||||
val enhanced: EnhancedVoiceCapabilities? = null,
|
||||
) {
|
||||
val displayVoice: String? get() = voice ?: voiceId
|
||||
val isEnabled: Boolean get() = enabled || (!provider.isNullOrBlank() && available)
|
||||
}
|
||||
|
||||
/**
|
||||
* Wire shape of the `tts.enhanced` block on `GET /voice/config` — the relay's
|
||||
* provider-aware enhanced-voice capability advertisement. Mirrors
|
||||
* `plugin/relay/voice.py:_enhanced_voice_block`. `voices`/`models` may be empty
|
||||
* (e.g. xAI uses a free-text voice field); the UI renders from the flags.
|
||||
*/
|
||||
@Serializable
|
||||
data class EnhancedVoiceCapabilities(
|
||||
val provider: String? = null,
|
||||
val supported: Boolean = false,
|
||||
val voices: List<String> = emptyList(),
|
||||
val models: List<String> = emptyList(),
|
||||
@SerialName("audio_tag_models")
|
||||
val audioTagModels: List<String> = emptyList(),
|
||||
@SerialName("audio_tags_enabled")
|
||||
val audioTagsEnabled: Boolean = false,
|
||||
@SerialName("audio_tags_label")
|
||||
val audioTagsLabel: String = "Expressive tone tags",
|
||||
@SerialName("supports_persona")
|
||||
val supportsPersona: Boolean = false,
|
||||
@SerialName("supports_language")
|
||||
val supportsLanguage: Boolean = false,
|
||||
@SerialName("persona_prompt_file")
|
||||
val personaPromptFile: String? = null,
|
||||
val overrides: List<String> = emptyList(),
|
||||
)
|
||||
|
||||
@Serializable
|
||||
data class RealtimeVoiceConfig(
|
||||
val success: Boolean = false,
|
||||
@@ -2481,6 +2530,8 @@ data class VoiceOutputConfig(
|
||||
val codec: String = "pcm",
|
||||
val optimize_streaming_latency: Int = 1,
|
||||
val text_normalization: Boolean = false,
|
||||
/** xAI expressive speech tags on the streaming renderer (xai_tts only). */
|
||||
val auto_speech_tags: Boolean = false,
|
||||
val fallback_enabled: Boolean = true,
|
||||
val fallback_provider: String? = null,
|
||||
val providers: List<RealtimeProviderInfo> = emptyList(),
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network.models
|
||||
package com.hermesandroid.relay.network.relay.models
|
||||
|
||||
import kotlinx.serialization.Serializable
|
||||
import kotlinx.serialization.json.JsonArray
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.shared
|
||||
|
||||
import android.content.Context
|
||||
import android.net.ConnectivityManager
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.shared
|
||||
|
||||
import android.util.Log
|
||||
import com.hermesandroid.relay.data.EndpointCandidate
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.shared
|
||||
|
||||
import android.content.Context
|
||||
import android.net.ConnectivityManager
|
||||
@@ -0,0 +1,31 @@
|
||||
package com.hermesandroid.relay.network.shared
|
||||
|
||||
import kotlinx.serialization.json.JsonObject
|
||||
|
||||
/**
|
||||
* Transport-neutral result of a phone-control dispatch.
|
||||
*
|
||||
* Shared vocabulary between the relay bridge path
|
||||
* ([com.hermesandroid.relay.network.relay.BridgeCommandHandler], which produces
|
||||
* it) and the upstream chat path
|
||||
* ([com.hermesandroid.relay.network.upstream.ChatHandler], which renders a
|
||||
* phone-action bubble from it). It is a passive DTO — not a client — so it
|
||||
* lives in `network.shared` to keep the upstream↔relay package fence intact
|
||||
* (ADR 34); neither side depends on the other to speak it.
|
||||
*
|
||||
* - [status] — HTTP-style status of the dispatch (200 = ok).
|
||||
* - [errorMessage] — human-readable failure text, or null on success.
|
||||
* - [errorCode] — machine error code when the dispatch failed and was
|
||||
* classified, or null.
|
||||
* - [resultJson] — the raw result object, for callers that need
|
||||
* action-specific fields (e.g. the resolved phone number from
|
||||
* /search_contacts). Optional.
|
||||
*/
|
||||
data class LocalDispatchResult(
|
||||
val status: Int,
|
||||
val errorMessage: String?,
|
||||
val errorCode: String?,
|
||||
val resultJson: JsonObject?,
|
||||
) {
|
||||
val isSuccess: Boolean get() = status in 200..299
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.shared
|
||||
|
||||
import java.net.URI
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
package com.hermesandroid.relay.network.shared
|
||||
|
||||
import com.hermesandroid.relay.data.VoiceAudioRoute
|
||||
import java.io.File
|
||||
|
||||
/**
|
||||
* Transport-neutral STT/TTS contract. The routing seam between the Standard
|
||||
* (dashboard) and Relay voice clients — implementations live in `network.upstream`
|
||||
* (`StandardHermesVoiceClient`) and `network.relay` (`RelayVoiceAudioClientAdapter`),
|
||||
* while this interface and the [AutoVoiceAudioClient] router stay dependency-neutral
|
||||
* so neither voice backend leaks across the upstream/relay package fence (ADR 34).
|
||||
*/
|
||||
interface VoiceAudioClient {
|
||||
val route: VoiceAudioRoute
|
||||
|
||||
/**
|
||||
* The route a call would ACTUALLY use right now. For a concrete backend this
|
||||
* equals [route]; for the [AutoVoiceAudioClient] router it resolves `Auto`
|
||||
* against live readiness (relay-first). Callers that need to reason about
|
||||
* the backend's capabilities (e.g. "is standard global-TTS in play?") must
|
||||
* use this, not the configured preference.
|
||||
*/
|
||||
val effectiveRoute: VoiceAudioRoute
|
||||
get() = route
|
||||
|
||||
suspend fun transcribe(audioFile: File): Result<String>
|
||||
suspend fun synthesize(text: String): Result<File>
|
||||
}
|
||||
|
||||
/**
|
||||
* Routes each STT/TTS call to the Standard (dashboard) or Relay voice client.
|
||||
*
|
||||
* Auto preference order is **Relay first, then Standard**: a paired Relay is
|
||||
* the purpose-built mobile facade — profile-aware voice config, no dashboard
|
||||
* sign-in dependency — so users who installed the plugin keep the richer
|
||||
* path. Standard is the zero-plugin route for vanilla Hermes installs and is
|
||||
* used whenever Relay isn't configured/paired (or fails mid-call). Power
|
||||
* users can force either route in Voice Settings.
|
||||
*
|
||||
* Depends only on the [VoiceAudioClient] abstraction (both backends are passed
|
||||
* in as the interface), so this router carries no upstream or relay imports.
|
||||
*/
|
||||
class AutoVoiceAudioClient(
|
||||
private val standardClient: VoiceAudioClient,
|
||||
private val relayClient: VoiceAudioClient,
|
||||
private val routeProvider: () -> VoiceAudioRoute,
|
||||
private val standardReadyProvider: () -> Boolean,
|
||||
private val relayReadyProvider: () -> Boolean,
|
||||
) : VoiceAudioClient {
|
||||
override val route: VoiceAudioRoute
|
||||
get() = routeProvider()
|
||||
|
||||
/**
|
||||
* Resolve the configured preference to the backend a call would land on:
|
||||
* `Standard`/`Relay` are honored verbatim; `Auto` prefers Relay when it's
|
||||
* ready (matching [runAuto]) and falls back to Standard otherwise. Used to
|
||||
* decide whether standard-only limitations (global TTS) currently apply.
|
||||
*/
|
||||
override val effectiveRoute: VoiceAudioRoute
|
||||
get() = when (routeProvider()) {
|
||||
VoiceAudioRoute.Standard -> VoiceAudioRoute.Standard
|
||||
VoiceAudioRoute.Relay -> VoiceAudioRoute.Relay
|
||||
VoiceAudioRoute.Auto ->
|
||||
if (relayReadyProvider()) VoiceAudioRoute.Relay else VoiceAudioRoute.Standard
|
||||
}
|
||||
|
||||
override suspend fun transcribe(audioFile: File): Result<String> =
|
||||
runWithSelectedRoute { it.transcribe(audioFile) }
|
||||
|
||||
override suspend fun synthesize(text: String): Result<File> =
|
||||
runWithSelectedRoute { it.synthesize(text) }
|
||||
|
||||
private suspend fun <T> runWithSelectedRoute(
|
||||
block: suspend (VoiceAudioClient) -> Result<T>,
|
||||
): Result<T> {
|
||||
return when (routeProvider()) {
|
||||
VoiceAudioRoute.Standard -> {
|
||||
if (!standardReadyProvider()) {
|
||||
Result.failure(
|
||||
IllegalStateException(
|
||||
"Vanilla Hermes voice is not available — check dashboard sign-in in Manage",
|
||||
),
|
||||
)
|
||||
} else {
|
||||
block(standardClient)
|
||||
}
|
||||
}
|
||||
VoiceAudioRoute.Relay -> {
|
||||
if (!relayReadyProvider()) {
|
||||
Result.failure(IllegalStateException("Relay voice is not available"))
|
||||
} else {
|
||||
block(relayClient)
|
||||
}
|
||||
}
|
||||
VoiceAudioRoute.Auto -> runAuto(block)
|
||||
}
|
||||
}
|
||||
|
||||
private suspend fun <T> runAuto(
|
||||
block: suspend (VoiceAudioClient) -> Result<T>,
|
||||
): Result<T> {
|
||||
var relayFailure: Result<T>? = null
|
||||
if (relayReadyProvider()) {
|
||||
val result = block(relayClient)
|
||||
if (result.isSuccess || !standardReadyProvider()) return result
|
||||
relayFailure = result
|
||||
}
|
||||
if (standardReadyProvider()) {
|
||||
val result = block(standardClient)
|
||||
if (result.isSuccess) return result
|
||||
return relayFailure ?: result
|
||||
}
|
||||
return relayFailure ?: Result.failure(
|
||||
IllegalStateException("Voice needs a reachable Hermes dashboard or Relay voice route"),
|
||||
)
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
package com.hermesandroid.relay.network.handlers
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import android.util.Log
|
||||
import com.hermesandroid.relay.data.Attachment
|
||||
import com.hermesandroid.relay.data.ChatMessage
|
||||
import com.hermesandroid.relay.data.ChatSession
|
||||
import com.hermesandroid.relay.data.HermesCard
|
||||
@@ -8,9 +9,10 @@ import com.hermesandroid.relay.data.MessageRole
|
||||
import com.hermesandroid.relay.data.RealtimeTurnTrace
|
||||
import com.hermesandroid.relay.data.ToolCall
|
||||
import com.hermesandroid.relay.data.VoiceIntentTrace
|
||||
import com.hermesandroid.relay.network.GatewaySubagentEvent
|
||||
import com.hermesandroid.relay.network.models.MessageItem
|
||||
import com.hermesandroid.relay.network.models.SessionItem
|
||||
import com.hermesandroid.relay.network.shared.LocalDispatchResult
|
||||
import com.hermesandroid.relay.network.upstream.GatewaySubagentEvent
|
||||
import com.hermesandroid.relay.network.upstream.models.MessageItem
|
||||
import com.hermesandroid.relay.network.upstream.models.SessionItem
|
||||
import kotlinx.coroutines.flow.MutableStateFlow
|
||||
import kotlinx.coroutines.flow.StateFlow
|
||||
import kotlinx.coroutines.flow.asStateFlow
|
||||
@@ -36,6 +38,14 @@ class ChatHandler {
|
||||
/** Maximum number of messages kept in memory per session. Oldest are trimmed. */
|
||||
internal const val MAX_MESSAGES = 500
|
||||
|
||||
private fun timestampToMillis(timestamp: Double?): Long {
|
||||
val value = timestamp ?: return 0L
|
||||
return if (value > 1e12) value.toLong() else (value * 1000).toLong()
|
||||
}
|
||||
|
||||
private fun firstPositive(vararg values: Long): Long =
|
||||
values.firstOrNull { it > 0L } ?: 0L
|
||||
|
||||
// Tool annotation patterns embedded as text markers by Hermes.
|
||||
//
|
||||
// Hermes /v1/chat/completions injects tool progress as inline markdown:
|
||||
@@ -72,7 +82,11 @@ class ChatHandler {
|
||||
// reachable when the tool fired, so we render an "unavailable"
|
||||
// placeholder instead of attempting a fetch.
|
||||
private val mediaRelayRegex = Regex("""MEDIA:hermes-relay://([A-Za-z0-9_-]+)""")
|
||||
private val mediaBarePathRegex = Regex("""^\s*MEDIA:(/\S+)\s*$""")
|
||||
// `/.+?` (not `/\S+`) so absolute paths containing spaces — e.g.
|
||||
// `MEDIA:/mnt/media/Coralee Adshade/undressher.jpg` — still match. The
|
||||
// trailing `\s*$` trims any trailing whitespace; non-greedy keeps the
|
||||
// capture to the path. OkHttp re-encodes the space for /media/by-path.
|
||||
private val mediaBarePathRegex = Regex("""^\s*MEDIA:(/.+?)\s*$""")
|
||||
// Rich card marker — single line, full JSON object payload.
|
||||
//
|
||||
// Agents emit:
|
||||
@@ -97,6 +111,14 @@ class ChatHandler {
|
||||
/** Whether to parse tool annotations from assistant text (for servers that don't emit tool events). */
|
||||
var parseToolAnnotations: Boolean = false
|
||||
|
||||
/**
|
||||
* When false (default — TUI/desktop parity), server-injected role:system
|
||||
* STEERING markers ("[System: The active model … has changed …]") are
|
||||
* hidden from the rendered transcript. A developer toggle flips this to
|
||||
* surface them for debugging. They always remain in server-side history.
|
||||
*/
|
||||
var showSystemMarkers: Boolean = false
|
||||
|
||||
/** Active personality/agent name — set by ChatViewModel before each stream. Included on new assistant messages. */
|
||||
var activeAgentName: String? = null
|
||||
|
||||
@@ -181,6 +203,18 @@ class ChatHandler {
|
||||
private val _messages = MutableStateFlow<List<ChatMessage>>(emptyList())
|
||||
val messages: StateFlow<List<ChatMessage>> = _messages.asStateFlow()
|
||||
|
||||
/**
|
||||
* Latest gateway `status.update` lifecycle line for the in-flight turn
|
||||
* (model fallback, retries, errors). Surfaced as a transient status line
|
||||
* above the composer; cleared when the turn completes.
|
||||
*/
|
||||
private val _turnStatus = MutableStateFlow<String?>(null)
|
||||
val turnStatus: StateFlow<String?> = _turnStatus.asStateFlow()
|
||||
|
||||
fun setTurnStatus(text: String) {
|
||||
_turnStatus.value = text
|
||||
}
|
||||
|
||||
private val _isStreaming = MutableStateFlow(false)
|
||||
val isStreaming: StateFlow<Boolean> = _isStreaming.asStateFlow()
|
||||
|
||||
@@ -213,6 +247,7 @@ class ChatHandler {
|
||||
role = MessageRole.SYSTEM,
|
||||
content = text,
|
||||
timestamp = System.currentTimeMillis(),
|
||||
clientOnly = true,
|
||||
)
|
||||
(list + notice).let { if (it.size > MAX_MESSAGES) it.drop(it.size - MAX_MESSAGES) else it }
|
||||
}
|
||||
@@ -221,8 +256,9 @@ class ChatHandler {
|
||||
/**
|
||||
* Append an assistant message that carries ONLY a gateway ask card
|
||||
* (clarify / approval / sudo / secret). Local-only — the server never
|
||||
* stores the ask as a message, so [loadMessageHistory] preserves the
|
||||
* `ask-` id prefix the same way it preserves voice-intent traces.
|
||||
* stores the ask as a message, so the bubble is flagged
|
||||
* [ChatMessage.clientOnly] and [loadMessageHistory] preserves it across the
|
||||
* reload the same way it preserves voice-intent traces.
|
||||
* Idempotent on [messageId] so a re-emitted ask never duplicates.
|
||||
*/
|
||||
fun appendAskCardMessage(messageId: String, card: HermesCard) {
|
||||
@@ -235,6 +271,7 @@ class ChatHandler {
|
||||
timestamp = System.currentTimeMillis(),
|
||||
cards = listOf(card),
|
||||
agentName = activeAgentName,
|
||||
clientOnly = true,
|
||||
)
|
||||
(list + msg).let { if (it.size > MAX_MESSAGES) it.drop(it.size - MAX_MESSAGES) else it }
|
||||
}
|
||||
@@ -306,6 +343,7 @@ class ChatHandler {
|
||||
role = MessageRole.USER,
|
||||
content = userText,
|
||||
timestamp = ts,
|
||||
clientOnly = true,
|
||||
)
|
||||
val assistantMsg = ChatMessage(
|
||||
id = "voice-intent-action-$ts",
|
||||
@@ -323,6 +361,7 @@ class ChatHandler {
|
||||
// session memory. Null for the pre-dispatch user bubble (the
|
||||
// raw transcribed utterance carries no structure on its own).
|
||||
voiceIntent = voiceIntent,
|
||||
clientOnly = true,
|
||||
)
|
||||
_messages.update { list ->
|
||||
(list + userMsg + assistantMsg).let {
|
||||
@@ -371,6 +410,7 @@ class ChatHandler {
|
||||
// pre-dispatch trace was appended) leave this null and rely
|
||||
// on the pre-dispatch trace's voiceIntent field.
|
||||
voiceIntent = voiceIntent,
|
||||
clientOnly = true,
|
||||
)
|
||||
_messages.update { list ->
|
||||
(list + resultMsg).let {
|
||||
@@ -443,7 +483,13 @@ class ChatHandler {
|
||||
val mapped = messages.map { msg ->
|
||||
if (msg.id == messageId && msg.role == MessageRole.ASSISTANT) {
|
||||
changed = true
|
||||
msg.copy(realtimeTurn = trace)
|
||||
// A realtimeTurn trace is attached ONLY for provider-only
|
||||
// (non-Hermes-backed) turns — Hermes-backed ones leave it
|
||||
// null because the server owns that turn. So a trace ⟺ a
|
||||
// purely local bubble: mark it clientOnly so the post-turn
|
||||
// reload preserves it instead of wiping the only record of
|
||||
// the turn (and the trace the next chat payload still needs).
|
||||
msg.copy(realtimeTurn = trace, clientOnly = true)
|
||||
} else {
|
||||
msg
|
||||
}
|
||||
@@ -712,8 +758,19 @@ class ChatHandler {
|
||||
}
|
||||
|
||||
/**
|
||||
* Load message history from API response into the messages list.
|
||||
* Replaces current messages with the loaded history.
|
||||
* Reconcile the in-memory transcript against the server message history.
|
||||
*
|
||||
* This is a surgical DELTA-MERGE keyed by message id, not a wholesale
|
||||
* replace:
|
||||
* - a server message whose id matches a local row UPDATES that row's
|
||||
* server-authoritative fields (content, tool calls, cards, reasoning)
|
||||
* in place while keeping every client-only field;
|
||||
* - a server message with no local row is INSERTED in timestamp order;
|
||||
* - a client-only orphan ([ChatMessage.clientOnly], no server row) is KEPT;
|
||||
* - a local row that WAS server-backed (not clientOnly) but is no longer in
|
||||
* the transcript is DROPPED (genuinely deleted / forked / truncated
|
||||
* server-side).
|
||||
*
|
||||
* Reconstructs tool calls from assistant messages' tool_calls field.
|
||||
*
|
||||
* Server-persisted message content still contains raw `MEDIA:...` markers
|
||||
@@ -738,11 +795,69 @@ class ChatHandler {
|
||||
// mutateMessage lookups find the newly-loaded messages.
|
||||
val pendingMediaHits = mutableListOf<Pair<String, MediaMarkerHit>>()
|
||||
|
||||
// Reconcile optimistic (client-UUID) live ids to their server ids BEFORE
|
||||
// building the carry map, so the id-keyed delta-merge updates rows in
|
||||
// place instead of dropping-and-reinserting them. SSE assistant rows are
|
||||
// already reconciled mid-turn (replaceMessageId ← message.started); but
|
||||
// GATEWAY assistant rows keep a local UUID (the gateway exposes no
|
||||
// per-message server id during the turn) and USER rows of every transport
|
||||
// keep a local UUID. Without this, the id-keyed carry-forward below
|
||||
// silently misses those rows, so a gateway turn's tokens/badges survived
|
||||
// only if a content match happened to cover them. See
|
||||
// [reconcileLiveIdsToServer].
|
||||
val serverItemIds = items.mapNotNullTo(HashSet()) { it.id }
|
||||
val idRemap = reconcileLiveIdsToServer(items, serverItemIds)
|
||||
|
||||
// Carry CLIENT-ONLY enrichment forward across the reload, keyed by the
|
||||
// RECONCILED message id. The server transcript (MessageItem) rebuilds
|
||||
// content, tool calls, and reasoning — but it does NOT persist per-message
|
||||
// token usage/cost, provenance badges, tapped-card confirmations, or the
|
||||
// voice/realtime sync traces. Carrying these forward is preserve-by-default
|
||||
// (not a per-field whitelist the next new field forgets). With the remap
|
||||
// above, an id-matched (SSE) row keeps its server id and a positionally
|
||||
// reconciled (gateway/user) row adopts its server id, so both match the
|
||||
// reloaded item id and update in place.
|
||||
val priorById = _messages.value.associateBy { idRemap[it.id] ?: it.id }
|
||||
|
||||
// Outbound (user-authored) attachments are the safety net for any USER row
|
||||
// that did NOT reconcile to a server id at merge time (they live on USER
|
||||
// messages, are not echoed in server content, and are not re-dispatched —
|
||||
// only inbound `MEDIA:` markers are). Reconciled rows already carry their
|
||||
// attachment by id via priorById, so we queue ONLY the still-unreconciled
|
||||
// rows here — otherwise a later same-content row could pull a duplicate
|
||||
// from the queue. Match by content, consume-once so repeated identical
|
||||
// sends don't cross-assign; inbound (relayToken != null) attachments are
|
||||
// excluded since the marker re-dispatch re-fetches them.
|
||||
val priorOutboundByContent = HashMap<String, ArrayDeque<List<Attachment>>>()
|
||||
for (msg in _messages.value) {
|
||||
if (msg.role != MessageRole.USER) continue
|
||||
if ((idRemap[msg.id] ?: msg.id) in serverItemIds) continue // carried by id already
|
||||
val outbound = msg.attachments.filter { it.relayToken == null }
|
||||
if (outbound.isNotEmpty()) {
|
||||
priorOutboundByContent.getOrPut(msg.content) { ArrayDeque() }.addLast(outbound)
|
||||
}
|
||||
}
|
||||
|
||||
val loaded = items.mapNotNull { item ->
|
||||
val role = when (item.role) {
|
||||
"user" -> MessageRole.USER
|
||||
"assistant" -> MessageRole.ASSISTANT
|
||||
"system" -> MessageRole.SYSTEM
|
||||
"system" ->
|
||||
// Upstream injects role:system STEERING markers into the
|
||||
// session history on model/personality change — e.g.
|
||||
// "[System: The active model for this chat has changed to …]"
|
||||
// (tui_gateway/server.py) — so the LLM picks up the new
|
||||
// runtime/persona. They are NOT user-facing; the TUI/desktop
|
||||
// keep them invisible. Hide them for parity UNLESS the
|
||||
// developer "Show system messages" toggle is on (debugging).
|
||||
// They remain in server-side history for the model regardless.
|
||||
if (!showSystemMarkers &&
|
||||
item.contentText?.trimStart()?.startsWith("[System:") == true
|
||||
) {
|
||||
return@mapNotNull null
|
||||
} else {
|
||||
MessageRole.SYSTEM
|
||||
}
|
||||
"tool" -> return@mapNotNull null // Merged into assistant tool calls above
|
||||
else -> return@mapNotNull null
|
||||
}
|
||||
@@ -780,24 +895,77 @@ class ChatHandler {
|
||||
afterMedia to emptyList()
|
||||
}
|
||||
|
||||
ChatMessage(
|
||||
id = messageId,
|
||||
role = role,
|
||||
content = cleanedContent,
|
||||
timestamp = timestampMs,
|
||||
isStreaming = false,
|
||||
toolCalls = toolCalls,
|
||||
cards = extractedCards,
|
||||
agentName = if (role == MessageRole.ASSISTANT) activeAgentName else null,
|
||||
// Server persists per-message reasoning — restore it so the
|
||||
// Thought-process block survives returning to the chat
|
||||
// instead of existing only for the live turn.
|
||||
thinkingContent = if (role == MessageRole.ASSISTANT) {
|
||||
item.resolvedReasoning?.trim() ?: ""
|
||||
} else {
|
||||
""
|
||||
},
|
||||
)
|
||||
val prior = priorById[messageId]
|
||||
// Outbound attachments: prefer an id-match (covers any future
|
||||
// user-message id reconciliation), else fall back to the
|
||||
// content-keyed queue. Inbound attachments are intentionally
|
||||
// excluded — they come back via the marker re-dispatch.
|
||||
val carriedAttachments = run {
|
||||
val byId = prior?.attachments.orEmpty().filter { it.relayToken == null }
|
||||
when {
|
||||
byId.isNotEmpty() -> byId
|
||||
role == MessageRole.USER ->
|
||||
priorOutboundByContent[cleanedContent]?.removeFirstOrNull().orEmpty()
|
||||
else -> emptyList()
|
||||
}
|
||||
}
|
||||
// Server reasoning is authoritative when present; absent, keep the
|
||||
// live-streamed thinking rather than blanking it on reload.
|
||||
val serverThinking =
|
||||
if (role == MessageRole.ASSISTANT) item.resolvedReasoning?.trim() else null
|
||||
|
||||
if (prior != null) {
|
||||
// DELTA-MERGE UPDATE — a local row with this id already exists.
|
||||
// Refresh ONLY the server-authoritative fields (content, tool
|
||||
// calls, cards, reasoning, role, timestamp) and keep every
|
||||
// client-only field (tokens, cost, badges, tapped-card
|
||||
// confirmations, voice/realtime traces, the clientOnly flag, …)
|
||||
// by copying the existing message. This is strictly broader than
|
||||
// the old curated carry list — a new client-only field is
|
||||
// preserved automatically — and produces an object equal to the
|
||||
// prior one when nothing server-side changed, so unchanged rows
|
||||
// don't churn.
|
||||
//
|
||||
// `id = messageId` adopts the server id: for an id-matched (SSE)
|
||||
// row it's a no-op, but for a positionally reconciled (gateway /
|
||||
// user) row whose `prior` still carries a client UUID it swaps in
|
||||
// the server id so EVERY future reload matches by id.
|
||||
prior.copy(
|
||||
id = messageId,
|
||||
role = role,
|
||||
content = cleanedContent,
|
||||
attachments = carriedAttachments,
|
||||
timestamp = timestampMs,
|
||||
isStreaming = false,
|
||||
isThinkingStreaming = false,
|
||||
toolCalls = toolCalls,
|
||||
cards = extractedCards,
|
||||
agentName = if (role == MessageRole.ASSISTANT) activeAgentName else null,
|
||||
thinkingContent = if (role == MessageRole.ASSISTANT) {
|
||||
serverThinking ?: prior.thinkingContent
|
||||
} else {
|
||||
""
|
||||
},
|
||||
)
|
||||
} else {
|
||||
// INSERT — a server message with no local row yet. Built from
|
||||
// server data; client-only enrichment defaults empty (there is
|
||||
// nothing local to carry).
|
||||
ChatMessage(
|
||||
id = messageId,
|
||||
role = role,
|
||||
content = cleanedContent,
|
||||
attachments = carriedAttachments,
|
||||
timestamp = timestampMs,
|
||||
isStreaming = false,
|
||||
toolCalls = toolCalls,
|
||||
cards = extractedCards,
|
||||
agentName = if (role == MessageRole.ASSISTANT) activeAgentName else null,
|
||||
// Server persists per-message reasoning — restore it so the
|
||||
// Thought-process block survives returning to the chat.
|
||||
thinkingContent = if (role == MessageRole.ASSISTANT) serverThinking ?: "" else "",
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Reload swaps the entire message list — any stale dedupe entries keyed
|
||||
@@ -805,41 +973,37 @@ class ChatHandler {
|
||||
// just collected against the reloaded IDs are guaranteed to fire.
|
||||
dispatchedMediaMarkers.clear()
|
||||
|
||||
// === PHASE3-voice-intents-chathistory ===
|
||||
// Preserve local-only voice-intent trace messages across a reload.
|
||||
// These messages are injected by [appendLocalVoiceIntentTrace] with
|
||||
// IDs prefixed "voice-intent-" and never reach the server-side
|
||||
// session, so a wholesale `_messages.value = loaded` assignment
|
||||
// would wipe them. Bailey hit this 2026-04-15: voice fall-through
|
||||
// ("proceed" → not a recognized intent → chat.sendMessage) triggered
|
||||
// a history reload on stream complete and the previous voice trace
|
||||
// vanished, making it look like "the chat cleared". Server-side
|
||||
// sync (so these traces reach the LLM's session memory too) is
|
||||
// still a v0.4.1 follow-up, but preserving them client-side is
|
||||
// enough to fix the disappearing-scrollback bug today.
|
||||
// Gateway-local bubbles ride the same preservation: steered text
|
||||
// (id "steer-…") lives inside a server-side tool result, never as a
|
||||
// user message, and ask cards (id "ask-…") are built from gateway
|
||||
// events that have no server-side message at all — a wholesale
|
||||
// reload would silently erase both.
|
||||
val preservedVoiceTraces = _messages.value.filter {
|
||||
it.id.startsWith("voice-intent-") ||
|
||||
it.id.startsWith("steer-") ||
|
||||
it.id.startsWith("ask-")
|
||||
}
|
||||
val merged = if (preservedVoiceTraces.isEmpty()) {
|
||||
// Preserve client-only orphans across the reload. A bubble flagged
|
||||
// [ChatMessage.clientOnly] has no server-side row — slash-command
|
||||
// notices (addSystemNotice), voice-intent traces
|
||||
// (appendLocalVoiceIntent*), the steer echo, gateway ask cards
|
||||
// (appendAskCardMessage), an errored turn the server never persisted
|
||||
// (markError), and provider-only realtime turns (attachRealtimeTurnTrace).
|
||||
// A wholesale `_messages.value = loaded` assignment would silently wipe
|
||||
// any whose id is absent from the reloaded transcript: the
|
||||
// disappearing-scrollback / "reply appears then vanishes" class of bug.
|
||||
//
|
||||
// This replaces the old id-prefix whitelist (voice-intent-/steer-/ask-/
|
||||
// system-notice-) plus "Error"-badge sniffing — each creator now declares
|
||||
// its own provenance, so a new client-only bubble type is preserved the
|
||||
// moment it sets the flag, with no reconcile-side change. Note the badge
|
||||
// subtlety: a turn that errors *after* persisting keeps an "Error" badge
|
||||
// but IS in the transcript, so it reconciles normally; only clientOnly +
|
||||
// absent-from-transcript marks a preservable orphan.
|
||||
val loadedIds = loaded.mapTo(HashSet()) { it.id }
|
||||
val preservedLocal = _messages.value.filter { it.clientOnly && it.id !in loadedIds }
|
||||
val merged = if (preservedLocal.isEmpty()) {
|
||||
loaded
|
||||
} else {
|
||||
// Merge by timestamp so voice traces interleave with the
|
||||
// reloaded server messages in chronological order. The voice
|
||||
// trace IDs carry `System.currentTimeMillis()` in their suffix
|
||||
// (see appendLocalVoiceIntentTrace), so ChatMessage.timestamp
|
||||
// is the source of truth here.
|
||||
(loaded + preservedVoiceTraces).sortedBy { it.timestamp }
|
||||
// Merge by timestamp so preserved orphans interleave with the
|
||||
// reloaded server messages in chronological order. Voice trace IDs
|
||||
// carry `System.currentTimeMillis()` in their suffix (see
|
||||
// appendLocalVoiceIntentTrace); other orphans keep their live
|
||||
// ChatMessage.timestamp — the source of truth either way.
|
||||
(loaded + preservedLocal).sortedBy { it.timestamp }
|
||||
}
|
||||
|
||||
_messages.value = if (merged.size > MAX_MESSAGES) merged.takeLast(MAX_MESSAGES) else merged
|
||||
// === END PHASE3-voice-intents-chathistory ===
|
||||
|
||||
// Now that the reloaded messages are in state, fire callbacks so the
|
||||
// ViewModel can insert LOADING/FAILED attachments via mutateMessage.
|
||||
@@ -863,6 +1027,105 @@ class ChatHandler {
|
||||
}
|
||||
}
|
||||
|
||||
/** One adoptable server row during id reconciliation. `taken` enforces consume-once. */
|
||||
private class ReconcileSlot(
|
||||
val serverId: String,
|
||||
val role: MessageRole,
|
||||
val key: String,
|
||||
) {
|
||||
var taken: Boolean = false
|
||||
}
|
||||
|
||||
/**
|
||||
* Map optimistic (client-UUID) live message ids → their server ids so the
|
||||
* id-keyed delta-merge in [loadMessageHistory] updates rows in place.
|
||||
*
|
||||
* Strategy: match each still-unreconciled, non-[ChatMessage.clientOnly] live
|
||||
* row to an unclaimed server row by (role, marker-stripped content),
|
||||
* consume-once in document order, and adopt the server id. Rows whose id is
|
||||
* already a server id (SSE assistant, reconciled mid-turn) are skipped so we
|
||||
* never double-swap; clientOnly orphans have no server row and are never
|
||||
* mapped; a live row that matches no slot is left alone (graceful fallback —
|
||||
* the content-keyed attachment fallback and drop-and-reinsert still apply, so
|
||||
* a count divergence / truncation / compaction never forces a wrong map).
|
||||
*
|
||||
* Returns oldLiveId → serverId for the rows that reconciled (empty when there
|
||||
* is nothing to adopt).
|
||||
*/
|
||||
private fun reconcileLiveIdsToServer(
|
||||
items: List<MessageItem>,
|
||||
serverItemIds: Set<String>,
|
||||
): Map<String, String> {
|
||||
val live = _messages.value
|
||||
if (live.isEmpty()) return emptyMap()
|
||||
val liveIds = live.mapTo(HashSet()) { it.id }
|
||||
|
||||
// Adoptable slots: rendered server rows (not tool, not a hidden steering
|
||||
// marker) whose id no live row already carries.
|
||||
val slots = items.mapNotNull { item ->
|
||||
val serverId = item.id ?: return@mapNotNull null
|
||||
if (serverId in liveIds) return@mapNotNull null
|
||||
val role = renderedRoleOf(item) ?: return@mapNotNull null
|
||||
ReconcileSlot(serverId, role, reconcileKey(item.contentText))
|
||||
}
|
||||
if (slots.isEmpty()) return emptyMap()
|
||||
|
||||
val remap = HashMap<String, String>()
|
||||
for (msg in live) {
|
||||
if (msg.clientOnly) continue // no server row to adopt
|
||||
if (msg.id in serverItemIds) continue // already a server id
|
||||
val key = reconcileKey(msg.content)
|
||||
val slot = slots.firstOrNull { !it.taken && it.role == msg.role && it.key == key }
|
||||
?: continue
|
||||
slot.taken = true
|
||||
remap[msg.id] = slot.serverId
|
||||
}
|
||||
return remap
|
||||
}
|
||||
|
||||
/**
|
||||
* The rendered role for a server message item, or null for rows that never
|
||||
* become a visible bubble (tool results, unknown roles, and hidden
|
||||
* `[System:` steering markers). Mirrors the role/skip logic in
|
||||
* [loadMessageHistory] so reconciliation only adopts ids onto rows that
|
||||
* actually render.
|
||||
*/
|
||||
private fun renderedRoleOf(item: MessageItem): MessageRole? = when (item.role) {
|
||||
"user" -> MessageRole.USER
|
||||
"assistant" -> MessageRole.ASSISTANT
|
||||
"system" ->
|
||||
if (!showSystemMarkers &&
|
||||
item.contentText?.trimStart()?.startsWith("[System:") == true
|
||||
) {
|
||||
null
|
||||
} else {
|
||||
MessageRole.SYSTEM
|
||||
}
|
||||
else -> null
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize content for reconciliation matching: strip `MEDIA:`/`CARD:` marker
|
||||
* lines (raw server content still carries them; live content already had them
|
||||
* stripped during streaming) and collapse to trimmed, non-blank lines joined
|
||||
* by newlines. Lets a marker-bearing assistant turn match its live row while
|
||||
* staying byte-stable for plain user/assistant text.
|
||||
*/
|
||||
private fun reconcileKey(content: String?): String {
|
||||
val text = content.orEmpty()
|
||||
if (text.isEmpty()) return ""
|
||||
val sb = StringBuilder()
|
||||
for (line in text.lines()) {
|
||||
val t = line.trim()
|
||||
if (t.isEmpty()) continue
|
||||
if (mediaRelayRegex.containsMatchIn(t) || mediaBarePathRegex.containsMatchIn(t)) continue
|
||||
if (cardMarkerRegex.containsMatchIn(t)) continue
|
||||
if (sb.isNotEmpty()) sb.append('\n')
|
||||
sb.append(t)
|
||||
}
|
||||
return sb.toString()
|
||||
}
|
||||
|
||||
/**
|
||||
* Marker hit collected during [loadMessageHistory] for post-assignment dispatch.
|
||||
*/
|
||||
@@ -1003,17 +1266,19 @@ class ChatHandler {
|
||||
*/
|
||||
fun updateSessions(items: List<SessionItem>) {
|
||||
val mapped = items.map { item ->
|
||||
// If > 1e12, already in milliseconds; otherwise convert from seconds
|
||||
val ts = item.startedAt ?: 0.0
|
||||
val timestampMs = if (ts > 1e12) ts.toLong() else (ts * 1000).toLong()
|
||||
val startedAtMs = timestampToMillis(item.startedAt)
|
||||
val lastActivityAtMs = timestampToMillis(item.resolvedLastActivity)
|
||||
val activityAtMs = firstPositive(lastActivityAtMs, startedAtMs)
|
||||
ChatSession(
|
||||
sessionId = item.id,
|
||||
title = item.title,
|
||||
model = item.model,
|
||||
messageCount = item.messageCount ?: 0,
|
||||
updatedAt = timestampMs
|
||||
updatedAt = activityAtMs,
|
||||
startedAt = startedAtMs,
|
||||
lastActivityAt = lastActivityAtMs,
|
||||
)
|
||||
}
|
||||
}.sortedByDescending { it.activityTimestamp }
|
||||
// Preserve the active session's optimistic row when the server list
|
||||
// doesn't include it yet: a freshly created chat has 0 messages and the
|
||||
// drawer's `min_messages=1` query filters it out until its first turn
|
||||
@@ -2073,12 +2338,57 @@ class ChatHandler {
|
||||
// Note: do NOT set _isStreaming to false — the run is still active
|
||||
}
|
||||
|
||||
/**
|
||||
* Stamp a "Stopped" badge on a message whose turn the user cancelled, so
|
||||
* the bubble carries a persistent status (not just a transient toast).
|
||||
* No-op if already present. Call before [onStreamComplete] on cancel.
|
||||
*/
|
||||
fun markStopped(messageId: String) {
|
||||
_messages.update { messages ->
|
||||
messages.map { msg ->
|
||||
if (msg.id == messageId && "Stopped" !in msg.badges) {
|
||||
msg.copy(badges = msg.badges + "Stopped")
|
||||
} else {
|
||||
msg
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Stamp an "Error" badge on a message whose turn ended in a server error
|
||||
* (e.g. a gateway ❌ lifecycle status), so a failed turn doesn't read as a
|
||||
* normal answer. No-op if already present.
|
||||
*
|
||||
* Also marks the bubble [ChatMessage.clientOnly] = true: the only caller is
|
||||
* the gateway ❌ terminal-error path, which fires on an in-flight turn the
|
||||
* server never persists. That makes this assistant bubble a client-only
|
||||
* orphan, so the reload must preserve it (the "reply appears then vanishes"
|
||||
* regression). If the same id later turns up in the server transcript (a
|
||||
* turn that errored *after* persisting), the reload reconciles it as a
|
||||
* normal server-backed message and the Error badge rides along via the
|
||||
* priorById carry — the clientOnly flag only gates the not-in-transcript
|
||||
* orphan case.
|
||||
*/
|
||||
fun markError(messageId: String) {
|
||||
_messages.update { messages ->
|
||||
messages.map { msg ->
|
||||
if (msg.id == messageId && "Error" !in msg.badges) {
|
||||
msg.copy(badges = msg.badges + "Error", clientOnly = true)
|
||||
} else {
|
||||
msg
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The entire agent run is complete (run.completed / done).
|
||||
* Marks the stream as finished and finalizes all messages.
|
||||
*/
|
||||
fun onStreamComplete(messageId: String) {
|
||||
_isStreaming.value = false
|
||||
_turnStatus.value = null
|
||||
insideThinkingBlock = false
|
||||
|
||||
// Flush any remaining annotation text that didn't end with a newline
|
||||
@@ -2239,7 +2549,7 @@ class ChatHandler {
|
||||
*
|
||||
* The label parameter is the short human-readable action name
|
||||
* ("Send SMS", "Open App", "Call", etc). Error-code branches mirror the
|
||||
* `error_code` strings [com.hermesandroid.relay.network.handlers.BridgeCommandHandler]
|
||||
* `error_code` strings [com.hermesandroid.relay.network.relay.BridgeCommandHandler]
|
||||
* emits on destructive-verb rejections.
|
||||
*/
|
||||
internal fun formatPhoneActionResult(
|
||||
@@ -1,14 +1,14 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import android.content.Context
|
||||
import com.hermesandroid.relay.data.Profile
|
||||
import com.hermesandroid.relay.network.models.MessageItem
|
||||
import com.hermesandroid.relay.network.models.MessageListResponse
|
||||
import com.hermesandroid.relay.network.models.SessionItem
|
||||
import com.hermesandroid.relay.network.models.SessionListResponse
|
||||
import com.hermesandroid.relay.auth.KeystoreTokenStore
|
||||
import com.hermesandroid.relay.auth.LegacyEncryptedPrefsTokenStore
|
||||
import com.hermesandroid.relay.network.upstream.models.MessageItem
|
||||
import com.hermesandroid.relay.network.upstream.models.MessageListResponse
|
||||
import com.hermesandroid.relay.network.upstream.models.SessionItem
|
||||
import com.hermesandroid.relay.network.upstream.models.SessionListResponse
|
||||
import com.hermesandroid.relay.auth.SecureStoreCache
|
||||
import com.hermesandroid.relay.auth.SessionTokenStore
|
||||
import com.hermesandroid.relay.auth.buildRawTokenStore
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.Serializable
|
||||
@@ -76,6 +76,11 @@ data class DashboardWsTicket(
|
||||
val ttlSeconds: Int? = null,
|
||||
)
|
||||
|
||||
data class DashboardChatDisplaySettings(
|
||||
val showReasoning: Boolean? = null,
|
||||
val toolDisplay: String? = null,
|
||||
)
|
||||
|
||||
/**
|
||||
* Native client for the Hermes dashboard/admin server (:9119).
|
||||
*
|
||||
@@ -167,6 +172,9 @@ class DashboardApiClient(
|
||||
|
||||
// --- Models (dashboard parity with hermes-desktop Settings → Model) ---
|
||||
|
||||
suspend fun getChatDisplaySettings(): Result<DashboardChatDisplaySettings> =
|
||||
getJsonObject("/api/config").mapCatching { root -> parseChatDisplaySettings(root) }
|
||||
|
||||
/** Full provider/model universe — REST twin of the TUI's `model.options` RPC. */
|
||||
suspend fun getModelOptions(): Result<JsonObject> = getJsonObject("/api/model/options")
|
||||
|
||||
@@ -401,9 +409,11 @@ class DashboardApiClient(
|
||||
* [profile] null/blank → the launch (default) profile's DB (param omitted). The
|
||||
* returned ids are the same stored-session ids the gateway `session.resume`
|
||||
* reads, so list-here / resume-on-gateway stays consistent. `min_messages=1`
|
||||
* drops empty draft rows; `order=recent` keeps live conversations on top.
|
||||
* drops empty draft rows where supported; `order=recent` requests activity
|
||||
* ordering where the host honors it. Android still sorts by decoded
|
||||
* `last_active` locally because older hosts return started-time order.
|
||||
*/
|
||||
suspend fun listSessions(profile: String? = null, limit: Int = 50): Result<List<SessionItem>> =
|
||||
suspend fun listSessions(profile: String? = null, limit: Int = 200): Result<List<SessionItem>> =
|
||||
withContext(Dispatchers.IO) {
|
||||
val query = buildList {
|
||||
add("limit=${limit.coerceIn(1, 200)}")
|
||||
@@ -510,15 +520,23 @@ class DashboardApiClient(
|
||||
* an auth-gated 401/403 also proves the route is registered.
|
||||
*/
|
||||
suspend fun audioRoutesPresent(): Boolean = withContext(Dispatchers.IO) {
|
||||
val request = Request.Builder()
|
||||
.url("$baseUrl/api/audio/transcribe")
|
||||
.head()
|
||||
.build()
|
||||
try {
|
||||
okHttpClient.newCall(request).execute().use { it.code != 404 }
|
||||
} catch (_: Exception) {
|
||||
false
|
||||
// Route exists if HEAD returns anything but a clean 404:
|
||||
// - 405 Method Not Allowed: path registered, POST-only (FastAPI/Starlette)
|
||||
// - 401/403: registered but auth-gated
|
||||
// - 2xx: handled
|
||||
// A reverse proxy fronting the dashboard can rewrite a 405 into a 404,
|
||||
// which would read as absent. To cut that false-negative, probe BOTH
|
||||
// audio routes and treat the surface as present if EITHER answers
|
||||
// non-404 (they ship together upstream, so one reachable implies both).
|
||||
fun probe(path: String): Boolean {
|
||||
val request = Request.Builder().url("$baseUrl$path").head().build()
|
||||
return try {
|
||||
okHttpClient.newCall(request).execute().use { it.code != 404 }
|
||||
} catch (_: Exception) {
|
||||
false
|
||||
}
|
||||
}
|
||||
probe("/api/audio/transcribe") || probe("/api/audio/speak")
|
||||
}
|
||||
|
||||
suspend fun requestWsTicket(): Result<DashboardWsTicket> = withContext(Dispatchers.IO) {
|
||||
@@ -745,6 +763,28 @@ class DashboardApiClient(
|
||||
private fun isPasswordProvider(name: String): Boolean =
|
||||
name.equals("basic", ignoreCase = true) ||
|
||||
name.equals("password", ignoreCase = true)
|
||||
|
||||
fun parseChatDisplaySettings(root: JsonObject): DashboardChatDisplaySettings {
|
||||
val config = root["config"] as? JsonObject
|
||||
val display = (config?.get("display") as? JsonObject)
|
||||
?: (root["display"] as? JsonObject)
|
||||
return DashboardChatDisplaySettings(
|
||||
showReasoning = display.booleanField("show_reasoning"),
|
||||
toolDisplay = normalizeToolDisplay(
|
||||
display.stringField("tool_progress")
|
||||
?: display.stringField("tool_display")
|
||||
?: display.stringField("toolProgress"),
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
private fun normalizeToolDisplay(value: String?): String? =
|
||||
when (value?.trim()?.lowercase()) {
|
||||
"off", "none", "false", "0", "hidden", "hide" -> "off"
|
||||
"compact", "minimal", "summary", "brief" -> "compact"
|
||||
"all", "detailed", "detail", "full", "true", "1", "on", "show" -> "detailed"
|
||||
else -> null
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -776,22 +816,56 @@ class InMemoryDashboardCookieStore : DashboardCookieStore {
|
||||
class EncryptedDashboardCookieStore(
|
||||
context: Context,
|
||||
connectionId: String,
|
||||
/**
|
||||
* The connection's TOKEN-store file key. When non-null the dashboard cookies
|
||||
* ride that already-built keyset (so there is NO second keyset build on cold
|
||||
* start), and any cookies in this connection's old stand-alone
|
||||
* `hermes_dashboard_<id>` file are migrated across once. Null preserves the
|
||||
* original stand-alone-file behavior for callers that can't resolve the key.
|
||||
*/
|
||||
tokenStoreKey: String? = null,
|
||||
private val json: Json = Json { ignoreUnknownKeys = true },
|
||||
) : DashboardCookieStore {
|
||||
private val serializer = ListSerializer(StoredDashboardCookie.serializer())
|
||||
private val appContext = context.applicationContext
|
||||
private val prefsName = prefsName(connectionId)
|
||||
private val standaloneCookiePrefsName = prefsName(connectionId)
|
||||
// Unify onto the connection's token file when we know it; else stand alone.
|
||||
// (Explicit type + distinct name avoids a type-inference cycle with the
|
||||
// companion `prefsName(connectionId)` function above.)
|
||||
private val storePrefsName: String = tokenStoreKey ?: standaloneCookiePrefsName
|
||||
private val unified = tokenStoreKey != null && tokenStoreKey != standaloneCookiePrefsName
|
||||
|
||||
// DEFERRED on purpose. Building the Keystore-backed prefs takes 1-4s
|
||||
// on StrongBox devices and serializes through a process-GLOBAL Tink
|
||||
// lock (AndroidKeysetManager.Builder.build) — eager construction here
|
||||
// froze the main thread for ~11s at app start when several stores were
|
||||
// built concurrently (frozen-sphere incident, 2026-06-11). Construction
|
||||
// is now free on any thread; the expensive build happens on the first
|
||||
// actual cookie access, which is always an OkHttp/IO thread.
|
||||
// DEFERRED on purpose. Building the Keystore-backed prefs takes 1-4s on
|
||||
// StrongBox devices and serializes through a process-GLOBAL Tink lock
|
||||
// (AndroidKeysetManager.Builder.build) — eager construction here froze the
|
||||
// main thread for ~11s at app start (frozen-sphere incident, 2026-06-11).
|
||||
// Construction is free on any thread; the expensive build happens on the
|
||||
// first actual cookie access, always an OkHttp/IO thread. Going through
|
||||
// SecureStoreCache means that build is SHARED with the connection's token
|
||||
// store — so when unified there is NO second keyset build at all.
|
||||
private val store: SessionTokenStore by lazy {
|
||||
KeystoreTokenStore.tryCreate(appContext, prefsName)
|
||||
?: LegacyEncryptedPrefsTokenStore(appContext, prefsName)
|
||||
val s = SecureStoreCache.getOrBuild(storePrefsName) {
|
||||
buildRawTokenStore(appContext, storePrefsName)
|
||||
}
|
||||
if (unified) migrateCookiesFromStandaloneFile(s)
|
||||
s
|
||||
}
|
||||
|
||||
/**
|
||||
* One-shot copy of this connection's cookies from the old stand-alone
|
||||
* `hermes_dashboard_<id>` file into the unified token file, marker-gated so
|
||||
* the old file's keyset is built at most once ever. On failure (corrupt old
|
||||
* file) the user simply re-signs-in to Manage — cookies are re-obtainable,
|
||||
* unlike the relay session token.
|
||||
*/
|
||||
private fun migrateCookiesFromStandaloneFile(target: SessionTokenStore) {
|
||||
if (target.contains(KEY_COOKIES_MIGRATED)) return
|
||||
runCatching {
|
||||
val old = buildRawTokenStore(appContext, standaloneCookiePrefsName)
|
||||
old.getString(KEY_COOKIES)?.let { target.putString(KEY_COOKIES, it) }
|
||||
old.clearAll()
|
||||
}
|
||||
target.putString(KEY_COOKIES_MIGRATED, "1")
|
||||
}
|
||||
|
||||
override fun load(): List<StoredDashboardCookie> {
|
||||
@@ -810,6 +884,7 @@ class EncryptedDashboardCookieStore(
|
||||
|
||||
companion object {
|
||||
private const val KEY_COOKIES = "dashboard_cookies_json"
|
||||
private const val KEY_COOKIES_MIGRATED = "dashboard_cookies_migrated"
|
||||
|
||||
fun prefsName(connectionId: String): String =
|
||||
"hermes_dashboard_${connectionId.take(8)}"
|
||||
@@ -832,8 +907,11 @@ class DashboardCookieJar(
|
||||
|
||||
override fun loadForRequest(url: HttpUrl): List<Cookie> {
|
||||
val now = clockMillis()
|
||||
val stored = store.load().filterNot { it.isExpired(now) }
|
||||
if (stored.size != store.load().size) {
|
||||
// Load once (each load() is a decrypt + JSON decode); prune expired
|
||||
// entries back to disk only when something actually expired.
|
||||
val all = store.load()
|
||||
val stored = all.filterNot { it.isExpired(now) }
|
||||
if (stored.size != all.size) {
|
||||
store.save(stored)
|
||||
}
|
||||
return stored.mapNotNull { it.toCookie() }
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import android.util.Log
|
||||
import com.hermesandroid.relay.util.AppForegroundTracker
|
||||
@@ -188,6 +188,68 @@ class GatewayChatClient(
|
||||
private val _connectionState = MutableStateFlow(GatewayConnectionState.Idle)
|
||||
val connectionState: StateFlow<GatewayConnectionState> = _connectionState.asStateFlow()
|
||||
|
||||
/**
|
||||
* Active personality the gateway is applying, as a config value ("none" when
|
||||
* the overlay is cleared, otherwise the personality name). Tracks the
|
||||
* upstream `display.personality` the way the desktop/TUI do: updated from the
|
||||
* [setPersonality] / [getPersonality] round-trips AND from connection-level
|
||||
* `session.info` events, so a change made via `/personality`, the desktop, or
|
||||
* the TUI reflects in the app. Null until first observed.
|
||||
*/
|
||||
private val _serverPersonality = MutableStateFlow<String?>(null)
|
||||
val serverPersonality: StateFlow<String?> = _serverPersonality.asStateFlow()
|
||||
|
||||
/**
|
||||
* Active model / provider the gateway reports for our session, tracked off
|
||||
* `session.info` the same way as [serverPersonality]. Lets a `/model` switch
|
||||
* made on the desktop/TUI (or our own dispatch) reflect in the app's model
|
||||
* pill without an app reload. Null until first observed; only ever set to a
|
||||
* non-blank value.
|
||||
*/
|
||||
private val _serverModel = MutableStateFlow<String?>(null)
|
||||
val serverModel: StateFlow<String?> = _serverModel.asStateFlow()
|
||||
|
||||
private val _serverProvider = MutableStateFlow<String?>(null)
|
||||
val serverProvider: StateFlow<String?> = _serverProvider.asStateFlow()
|
||||
|
||||
/**
|
||||
* Active reasoning EFFORT from `session.info` (string; "" when reasoning is
|
||||
* disabled). The reasoning DISPLAY mode is NOT on session.info — it stays a
|
||||
* `config.get reasoning` concern ([getReasoningSettings]). Only ever set to a
|
||||
* non-blank value so a disabled-reasoning "" never clobbers the chip.
|
||||
*/
|
||||
private val _serverReasoningEffort = MutableStateFlow<String?>(null)
|
||||
val serverReasoningEffort: StateFlow<String?> = _serverReasoningEffort.asStateFlow()
|
||||
|
||||
/**
|
||||
* Server-reported credential warning (upstream `session.info.credential_warning`)
|
||||
* — present ONLY when the active provider's key is missing/invalid, absent
|
||||
* (→ null here) when healthy. Cleared on absence so it self-resolves when the
|
||||
* key is fixed.
|
||||
*/
|
||||
private val _serverCredentialWarning = MutableStateFlow<String?>(null)
|
||||
val serverCredentialWarning: StateFlow<String?> = _serverCredentialWarning.asStateFlow()
|
||||
|
||||
/**
|
||||
* Effective approval-bypass (YOLO) + fast-mode state from `session.info`
|
||||
* (`yolo`/`fast` booleans). YOLO has NO `config.get` upstream — session.info
|
||||
* is the only read. Null until first observed.
|
||||
*/
|
||||
private val _serverYolo = MutableStateFlow<Boolean?>(null)
|
||||
val serverYolo: StateFlow<Boolean?> = _serverYolo.asStateFlow()
|
||||
|
||||
private val _serverFast = MutableStateFlow<Boolean?>(null)
|
||||
val serverFast: StateFlow<Boolean?> = _serverFast.asStateFlow()
|
||||
|
||||
/**
|
||||
* Context-window usage `(used, max)` from `session.info`'s `usage` block
|
||||
* (upstream `_get_usage`). `session.info` is emitted on session resume, so
|
||||
* this lets the context bar paint immediately on resume instead of waiting
|
||||
* for the first turn's usage event. Null until observed / when omitted.
|
||||
*/
|
||||
private val _serverContext = MutableStateFlow<Pair<Int, Int>?>(null)
|
||||
val serverContext: StateFlow<Pair<Int, Int>?> = _serverContext.asStateFlow()
|
||||
|
||||
/** Serializes connect / session-establish so concurrent sends share one socket. */
|
||||
private val connectMutex = Mutex()
|
||||
|
||||
@@ -223,6 +285,24 @@ class GatewayChatClient(
|
||||
private fun currentSessionProfile(): String? =
|
||||
sessionProfileProvider().takeIf { !it.isNullOrBlank() }
|
||||
|
||||
/**
|
||||
* Supplies the explicit in-chat overrides to bind onto each fresh
|
||||
* `session.create` (upstream honors `model`/`provider`/`reasoning_effort`/
|
||||
* `fast` → the new session's per-session overrides). Pulled live so it
|
||||
* always reflects the current picker + safety/speed controls; null (or all
|
||||
* fields null) = no explicit override, so the new session inherits the
|
||||
* profile / server default. Wired by ChatViewModel. A live session keeps its
|
||||
* agent config, so this only affects session creation — mid-session switches
|
||||
* go through [setModel]/[setReasoning]/[setFast] (`config.set`).
|
||||
*/
|
||||
@Volatile
|
||||
var sessionModelProvider: () -> GatewaySessionModel? = { null }
|
||||
|
||||
private fun currentSessionModel(): GatewaySessionModel? =
|
||||
sessionModelProvider()?.takeIf {
|
||||
!it.model.isNullOrBlank() || !it.reasoningEffort.isNullOrBlank() || it.fast != null
|
||||
}
|
||||
|
||||
@Volatile
|
||||
private var activeTurn: GatewayTurn? = null
|
||||
|
||||
@@ -418,16 +498,31 @@ class GatewayChatClient(
|
||||
* send.
|
||||
*/
|
||||
fun prewarm(storedSessionId: String?) {
|
||||
scope.launch {
|
||||
try {
|
||||
connectMutex.withLock {
|
||||
ensureConnected()
|
||||
if (storedSessionId != null) resumeForPrewarm(storedSessionId)
|
||||
}
|
||||
} catch (e: Exception) {
|
||||
Log.d(TAG, "Gateway prewarm skipped: ${e.message}")
|
||||
scope.launch { prewarmAwait(storedSessionId) }
|
||||
}
|
||||
|
||||
/**
|
||||
* Suspending [prewarm]: establishes the socket and (when [storedSessionId]
|
||||
* is non-null) resumes the existing session, returning only once that work
|
||||
* has settled. Returns true when a live session is available afterwards.
|
||||
*
|
||||
* An in-chat model/effort/fast switch MUST await this before its
|
||||
* `config.set`. Otherwise the switch races the fire-and-forget [prewarm]
|
||||
* and runs with `liveSessionId == null`, which upstream applies as a GLOBAL
|
||||
* config write instead of a per-session one — so the pick never lands on
|
||||
* the session the next turn actually uses (a fresh chat pre-creates a
|
||||
* session, so this path is the common case, not the edge case).
|
||||
*/
|
||||
suspend fun prewarmAwait(storedSessionId: String?): Boolean {
|
||||
try {
|
||||
connectMutex.withLock {
|
||||
ensureConnected()
|
||||
if (storedSessionId != null) resumeForPrewarm(storedSessionId)
|
||||
}
|
||||
} catch (e: Exception) {
|
||||
Log.d(TAG, "Gateway prewarm skipped: ${e.message}")
|
||||
}
|
||||
return liveSessionId != null
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -565,6 +660,57 @@ class GatewayChatClient(
|
||||
},
|
||||
)
|
||||
|
||||
/**
|
||||
* Read the active personality (`config.get {key:"personality"}`). Returns the
|
||||
* upstream config value — `"none"` when the overlay is cleared, otherwise the
|
||||
* personality name. Connects on demand. Used to seed [serverPersonality] when
|
||||
* a gateway connection comes up so the app reflects whatever the server
|
||||
* (config / desktop / TUI) currently has active.
|
||||
*/
|
||||
suspend fun getPersonality(): Result<String> {
|
||||
if (webSocket == null || readySignal?.isCompleted != true) {
|
||||
try {
|
||||
connectMutex.withLock { ensureConnected() }
|
||||
} catch (e: Exception) {
|
||||
return Result.failure(e)
|
||||
}
|
||||
}
|
||||
return rpc("config.get", buildJsonObject { put("key", "personality") })
|
||||
.map { result ->
|
||||
(result.stringField("value") ?: "none").ifBlank { "none" }
|
||||
.also { _serverPersonality.value = it }
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Set the personality the way the desktop + TUI do (`config.set
|
||||
* {key:"personality"}`). The gateway persists `display.personality` +
|
||||
* `agent.system_prompt` to the active profile's config AND applies the
|
||||
* overlay live to the current session (no history reset). Pass `"none"`
|
||||
* (or `"default"`/`"neutral"`) to clear the overlay. Returns the resolved
|
||||
* active value (`"none"` or the name); also updates [serverPersonality]
|
||||
* directly so observers don't have to wait on the `session.info` echo (which
|
||||
* only fires when a live session exists).
|
||||
*/
|
||||
suspend fun setPersonality(value: String): Result<String> {
|
||||
if (webSocket == null || readySignal?.isCompleted != true) {
|
||||
try {
|
||||
connectMutex.withLock { ensureConnected() }
|
||||
} catch (e: Exception) {
|
||||
return Result.failure(e)
|
||||
}
|
||||
}
|
||||
val params = buildJsonObject {
|
||||
put("key", "personality")
|
||||
put("value", value)
|
||||
liveSessionId?.let { put("session_id", it) }
|
||||
}
|
||||
return rpc("config.set", params).map { result ->
|
||||
(result.stringField("value") ?: value).ifBlank { "none" }
|
||||
.also { _serverPersonality.value = it }
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch the curated provider/model list (`model.options`) — the same RPC
|
||||
* the upstream desktop + TUI model picker uses (grok / kimi / gpt-5.5 …,
|
||||
@@ -592,6 +738,11 @@ class GatewayChatClient(
|
||||
.mapNotNull { (it as? JsonPrimitive)?.contentOrNull },
|
||||
isCurrent = (obj["is_current"] as? JsonPrimitive)?.booleanOrNull ?: false,
|
||||
warning = obj.stringField("warning"),
|
||||
authenticated = (obj["authenticated"] as? JsonPrimitive)?.booleanOrNull ?: true,
|
||||
unavailableModels = (obj["unavailable_models"] as? JsonArray).orEmpty()
|
||||
.mapNotNull { (it as? JsonPrimitive)?.contentOrNull },
|
||||
freeTier = (obj["free_tier"] as? JsonPrimitive)?.booleanOrNull ?: false,
|
||||
totalModels = (obj["total_models"] as? JsonPrimitive)?.contentOrNull?.toIntOrNull() ?: 0,
|
||||
)
|
||||
}
|
||||
GatewayModelOptions(
|
||||
@@ -621,6 +772,96 @@ class GatewayChatClient(
|
||||
},
|
||||
)
|
||||
|
||||
/** Fetch the session/global reasoning effort and display mode. */
|
||||
suspend fun getReasoningSettings(): Result<GatewayReasoningSettings> {
|
||||
if (webSocket == null || readySignal?.isCompleted != true) {
|
||||
try {
|
||||
connectMutex.withLock { ensureConnected() }
|
||||
} catch (e: Exception) {
|
||||
return Result.failure(e)
|
||||
}
|
||||
}
|
||||
return rpc(
|
||||
"config.get",
|
||||
buildJsonObject {
|
||||
put("key", "reasoning")
|
||||
liveSessionId?.let { put("session_id", it) }
|
||||
},
|
||||
).map { result ->
|
||||
GatewayReasoningSettings(
|
||||
effort = result.stringField("value")?.takeIf { it.isNotBlank() } ?: "medium",
|
||||
display = result.stringField("display")?.takeIf { it.isNotBlank() },
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Switch the active reasoning effort through the same `config.set` path
|
||||
* the desktop/TUI `/reasoning` command uses. Values are upstream-defined:
|
||||
* none, minimal, low, medium, high, xhigh.
|
||||
*/
|
||||
suspend fun setReasoning(value: String): Result<JsonObject> =
|
||||
rpc(
|
||||
"config.set",
|
||||
buildJsonObject {
|
||||
put("key", "reasoning")
|
||||
put("value", value)
|
||||
liveSessionId?.let { put("session_id", it) }
|
||||
},
|
||||
)
|
||||
|
||||
/**
|
||||
* Toggle per-session approval bypass (YOLO) via `config.set {key:"yolo"}` —
|
||||
* the same session-scoped flag the desktop's setSessionYolo and the TUI's
|
||||
* Shift+Tab use (`value` "1"/"0", `scope` "session" = ephemeral, never writes
|
||||
* config.yaml). Requires a live session for the per-session flag. Updates
|
||||
* [serverYolo] from the echo so observers don't wait on `session.info`.
|
||||
* Returns the resolved enabled state. There is deliberately NO `getYolo()` —
|
||||
* upstream has no `config.get yolo`; session.info is the only read.
|
||||
*/
|
||||
suspend fun setYolo(enabled: Boolean, scope: String = "session"): Result<Boolean> {
|
||||
if (webSocket == null || readySignal?.isCompleted != true) {
|
||||
try {
|
||||
connectMutex.withLock { ensureConnected() }
|
||||
} catch (e: Exception) {
|
||||
return Result.failure(e)
|
||||
}
|
||||
}
|
||||
val params = buildJsonObject {
|
||||
put("key", "yolo")
|
||||
put("value", if (enabled) "1" else "0")
|
||||
put("scope", scope)
|
||||
liveSessionId?.let { put("session_id", it) }
|
||||
}
|
||||
return rpc("config.set", params).map { result ->
|
||||
(result.stringField("value") == "1").also { _serverYolo.value = it }
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Toggle fast mode (priority service tier) via `config.set {key:"fast"}` —
|
||||
* desktop parity (`value` "fast"/"normal", session-scoped). Capability-gated
|
||||
* upstream: enabling fails (error 4002) when the current model has no fast
|
||||
* tier. Updates [serverFast]; returns the resolved enabled state.
|
||||
*/
|
||||
suspend fun setFast(enabled: Boolean): Result<Boolean> {
|
||||
if (webSocket == null || readySignal?.isCompleted != true) {
|
||||
try {
|
||||
connectMutex.withLock { ensureConnected() }
|
||||
} catch (e: Exception) {
|
||||
return Result.failure(e)
|
||||
}
|
||||
}
|
||||
val params = buildJsonObject {
|
||||
put("key", "fast")
|
||||
put("value", if (enabled) "fast" else "normal")
|
||||
liveSessionId?.let { put("session_id", it) }
|
||||
}
|
||||
return rpc("config.set", params).map { result ->
|
||||
(result.stringField("value") == "fast").also { _serverFast.value = it }
|
||||
}
|
||||
}
|
||||
|
||||
fun shutdown() {
|
||||
activeTurn?.cancel()
|
||||
activeTurn = null
|
||||
@@ -719,10 +960,52 @@ class GatewayChatClient(
|
||||
currentSessionProfile()?.let { put("profile", it) }
|
||||
},
|
||||
)
|
||||
val live = resumed.getOrNull()?.stringField("session_id")
|
||||
val result = resumed.getOrNull()
|
||||
val live = result?.stringField("session_id")
|
||||
if (live != null) {
|
||||
liveSessionId = live
|
||||
storedSessionId = storedId
|
||||
// Paint the session's real model/provider/effort/etc NOW from the
|
||||
// resume result's embedded `info` (same shape session.info carries),
|
||||
// so a reopened session shows its ACTUAL model immediately instead of
|
||||
// a misleading default until the first turn's async session.info.
|
||||
(result["info"] as? JsonObject)?.let { applySessionInfo(it) }
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Apply connection-level session info (model / provider / reasoning effort /
|
||||
* personality / yolo / fast / context usage) into the `_server*` state flows.
|
||||
* Shared by the `session.info` event handler and the `session.resume` RPC
|
||||
* result — the resume response embeds the same `info` object, so reopening a
|
||||
* session can paint its real model up front rather than waiting for a turn.
|
||||
*/
|
||||
private fun applySessionInfo(info: JsonObject) {
|
||||
if (info.containsKey("personality")) {
|
||||
_serverPersonality.value =
|
||||
(info.stringField("personality") ?: "").ifBlank { "none" }
|
||||
}
|
||||
info.stringField("model")?.takeIf { it.isNotBlank() }?.let { _serverModel.value = it }
|
||||
info.stringField("provider")?.takeIf { it.isNotBlank() }?.let { _serverProvider.value = it }
|
||||
// reasoning effort: ignore "" (reasoning disabled) so it can't clobber
|
||||
// the chip; display mode is config.get-only, not here.
|
||||
info.stringField("reasoning_effort")?.takeIf { it.isNotBlank() }
|
||||
?.let { _serverReasoningEffort.value = it }
|
||||
// credential_warning: present only when the provider key is missing/
|
||||
// invalid. ABSENT means healthy — clear to null so it self-resolves.
|
||||
_serverCredentialWarning.value =
|
||||
info.stringField("credential_warning")?.takeIf { it.isNotBlank() }
|
||||
(info["yolo"] as? JsonPrimitive)?.booleanOrNull?.let { _serverYolo.value = it }
|
||||
(info["fast"] as? JsonPrimitive)?.booleanOrNull?.let { _serverFast.value = it }
|
||||
// Context usage: require used > 0 — a COLD resume resets counters and
|
||||
// reports 0 until the first turn rebuilds the prompt; painting 0 would
|
||||
// mislead on a session that actually has history.
|
||||
(info["usage"] as? JsonObject)?.let { usage ->
|
||||
val used = (usage["context_used"] as? JsonPrimitive)?.intOrNull
|
||||
val max = (usage["context_max"] as? JsonPrimitive)?.intOrNull
|
||||
if (used != null && used > 0 && max != null && max > 0) {
|
||||
_serverContext.value = used to max
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -745,10 +1028,12 @@ class GatewayChatClient(
|
||||
currentSessionProfile()?.let { put("profile", it) }
|
||||
},
|
||||
)
|
||||
val live = resumed.getOrNull()?.stringField("session_id")
|
||||
val result = resumed.getOrNull()
|
||||
val live = result?.stringField("session_id")
|
||||
if (live != null) {
|
||||
liveSessionId = live
|
||||
storedSessionId = requestedStoredId
|
||||
(result["info"] as? JsonObject)?.let { applySessionInfo(it) }
|
||||
return
|
||||
}
|
||||
Log.w(
|
||||
@@ -764,6 +1049,23 @@ class GatewayChatClient(
|
||||
put("cols", DEFAULT_COLS)
|
||||
if (!newSessionTitle.isNullOrBlank()) put("title", newSessionTitle)
|
||||
currentSessionProfile()?.let { put("profile", it) }
|
||||
// Bind the in-chat overrides to the new session as its
|
||||
// per-session overrides. Upstream tui_gateway session.create
|
||||
// reads `model`/`provider` (→ model_override), `reasoning_effort`
|
||||
// (→ create_reasoning_override) and `fast` (→ priority service
|
||||
// tier) — verified server.py:4175-4191. Without this a fresh
|
||||
// chat ignores the picker/safety controls and builds the agent
|
||||
// from the global default, and worse, setting effort/fast before
|
||||
// the first message runs a SESSIONLESS config.set that upstream
|
||||
// applies as a GLOBAL config write. A live session keeps its own
|
||||
// config — this is create-only; mid-session switches use
|
||||
// config.set (setModel/setReasoning/setFast).
|
||||
currentSessionModel()?.let { sm ->
|
||||
sm.model?.takeIf { it.isNotBlank() }?.let { put("model", it) }
|
||||
sm.provider?.takeIf { it.isNotBlank() }?.let { put("provider", it) }
|
||||
sm.reasoningEffort?.takeIf { it.isNotBlank() }?.let { put("reasoning_effort", it) }
|
||||
sm.fast?.let { put("fast", it) }
|
||||
}
|
||||
},
|
||||
).getOrElse { e ->
|
||||
throw GatewayPreflightException("session.create failed: ${e.message}")
|
||||
@@ -864,6 +1166,20 @@ class GatewayChatClient(
|
||||
return
|
||||
}
|
||||
|
||||
// `session.info` is connection-level (personality / model / context
|
||||
// usage), emitted on a config change even with no turn in flight. Capture
|
||||
// the active personality here — for our own session only — so a
|
||||
// `/personality`, desktop, or TUI change keeps the app in sync. Falls
|
||||
// through to the turn dispatch below so an in-flight turn still sees it.
|
||||
if (type == "session.info" &&
|
||||
(eventSessionId == null || liveSessionId == null || eventSessionId == liveSessionId)
|
||||
) {
|
||||
// Connection-level session info (model / provider / effort / persona /
|
||||
// yolo / fast / usage) — shared with the session.resume result via
|
||||
// applySessionInfo so both paths stay in lockstep.
|
||||
payload?.let { applySessionInfo(it) }
|
||||
}
|
||||
|
||||
val turn = activeTurn ?: return
|
||||
// Foreign-session events (another client's chat on the same gateway) are not ours.
|
||||
if (eventSessionId != null && liveSessionId != null && eventSessionId != liveSessionId) {
|
||||
@@ -1225,6 +1541,13 @@ class GatewayChatClient(
|
||||
onToolGenerating = { v -> callbackDispatcher { callbacks.onToolGenerating(v) } },
|
||||
onSubagentEvent = { v -> callbackDispatcher { callbacks.onSubagentEvent(v) } },
|
||||
onInteractionRequest = { v -> callbackDispatcher { callbacks.onInteractionRequest(v) } },
|
||||
// MUST be wrapped like every other member: GatewayTurnCallbacks gives
|
||||
// onStatusUpdate a default no-op, so omitting it here silently swallows
|
||||
// EVERY gateway status line — the ❌ terminal-error lifecycle update
|
||||
// included. Without it markError never fires, the turn isn't badged
|
||||
// "Error", and onComplete's history reload wipes the error bubble (the
|
||||
// "reply appears then vanishes" bug).
|
||||
onStatusUpdate = { kind, text -> callbackDispatcher { callbacks.onStatusUpdate(kind, text) } },
|
||||
)
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import com.hermesandroid.relay.network.models.UsageInfo
|
||||
import com.hermesandroid.relay.network.upstream.models.UsageInfo
|
||||
import kotlinx.serialization.json.JsonArray
|
||||
import kotlinx.serialization.json.JsonObject
|
||||
import kotlinx.serialization.json.JsonPrimitive
|
||||
@@ -217,8 +217,15 @@ class GatewayEventMapper(private val callbacks: GatewayTurnCallbacks) {
|
||||
),
|
||||
)
|
||||
|
||||
// Known-but-unrendered (notification.show, status.update, …) and
|
||||
// unknown types alike: ignore.
|
||||
"status.update" -> {
|
||||
val text = payload.string("text")
|
||||
if (!text.isNullOrBlank()) {
|
||||
callbacks.onStatusUpdate(payload.string("kind"), text)
|
||||
}
|
||||
}
|
||||
|
||||
// Known-but-unrendered (notification.show, …) and unknown types
|
||||
// alike: ignore.
|
||||
else -> Unit
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import android.annotation.SuppressLint
|
||||
import android.app.NotificationChannel
|
||||
@@ -1,6 +1,6 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import com.hermesandroid.relay.network.models.UsageInfo
|
||||
import com.hermesandroid.relay.network.upstream.models.UsageInfo
|
||||
|
||||
/**
|
||||
* Shared types for the Gateway chat transport — upstream hermes-agent's
|
||||
@@ -146,6 +146,14 @@ data class GatewayModelProvider(
|
||||
val models: List<String>,
|
||||
val isCurrent: Boolean,
|
||||
val warning: String?,
|
||||
// Picker hints from upstream `model.options` (build_models_payload,
|
||||
// picker_hints=True). Default to "usable" so older servers that omit them
|
||||
// don't gray everything out.
|
||||
val authenticated: Boolean = true,
|
||||
/** Paid models the current account can't pick (free-tier / no credits). */
|
||||
val unavailableModels: List<String> = emptyList(),
|
||||
val freeTier: Boolean = false,
|
||||
val totalModels: Int = 0,
|
||||
)
|
||||
|
||||
/** Result of the gateway `model.options` RPC. */
|
||||
@@ -155,6 +163,39 @@ data class GatewayModelOptions(
|
||||
val currentProvider: String,
|
||||
)
|
||||
|
||||
/**
|
||||
* The explicit in-chat overrides to bind onto a gateway `session.create` as the
|
||||
* new session's PER-SESSION overrides. Matches the upstream desktop client,
|
||||
* whose `session.create` carries `model`/`provider`/`reasoning_effort`/`fast`
|
||||
* (tui_gateway honors them → `session_model_override` / `create_reasoning_override`
|
||||
* / `create_service_tier_override`; verified `tui_gateway/server.py:4175-4191`).
|
||||
* Supplied live by ChatViewModel from the picker + safety/speed controls.
|
||||
*
|
||||
* Every field is nullable = "no explicit override for this new chat", so the
|
||||
* fresh session inherits the profile / server default rather than the picker
|
||||
* (or a stale local value) silently clobbering it. Crucially this keeps these
|
||||
* picks OFF the sessionless `config.set` path, which upstream applies as GLOBAL
|
||||
* writes (and `yolo` even leaks to other sessions via `os.environ`).
|
||||
*
|
||||
* [model] is the model id (e.g. `grok-4.3`); [provider] is the authenticated
|
||||
* provider slug (e.g. `xai`). [reasoningEffort] is the upstream effort string
|
||||
* (`low`/`medium`/`high`/…). [fast] pins the priority service tier when true.
|
||||
* Note `yolo` is intentionally absent — upstream `session.create` does NOT
|
||||
* accept it as a per-session override, so it is applied post-create instead.
|
||||
*/
|
||||
data class GatewaySessionModel(
|
||||
val model: String?,
|
||||
val provider: String?,
|
||||
val reasoningEffort: String? = null,
|
||||
val fast: Boolean? = null,
|
||||
)
|
||||
|
||||
/** Result of the gateway `config.get {key:"reasoning"}` RPC. */
|
||||
data class GatewayReasoningSettings(
|
||||
val effort: String,
|
||||
val display: String?,
|
||||
)
|
||||
|
||||
/**
|
||||
* Callback set for one gateway turn. Shapes intentionally mirror the SSE
|
||||
* callback lambdas in ChatViewModel.startStream() so the gateway branch can
|
||||
@@ -190,4 +231,10 @@ class GatewayTurnCallbacks(
|
||||
* cancelled.
|
||||
*/
|
||||
val onInteractionRequest: (GatewayAsk) -> Unit,
|
||||
/**
|
||||
* Gateway `status.update` lifecycle line — model fallback, retries, and
|
||||
* errors (often emoji-prefixed: 🔄 fallback, ⏳ retry, ❌ error). Default
|
||||
* no-op so non-gateway/legacy constructors don't need to provide it.
|
||||
*/
|
||||
val onStatusUpdate: (kind: String?, text: String) -> Unit = { _, _ -> },
|
||||
)
|
||||
@@ -1,21 +1,21 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import android.os.Handler
|
||||
import android.os.Looper
|
||||
import android.util.Log
|
||||
import com.hermesandroid.relay.data.AgentDisplay
|
||||
import com.hermesandroid.relay.data.AppAnalytics
|
||||
import com.hermesandroid.relay.network.models.CreateSessionRequest
|
||||
import com.hermesandroid.relay.network.models.HermesSseEvent
|
||||
import com.hermesandroid.relay.network.models.MessageItem
|
||||
import com.hermesandroid.relay.network.models.MessageListResponse
|
||||
import com.hermesandroid.relay.network.models.RenameSessionRequest
|
||||
import com.hermesandroid.relay.network.models.SessionItem
|
||||
import com.hermesandroid.relay.network.models.SessionListResponse
|
||||
import com.hermesandroid.relay.network.models.SessionResponse
|
||||
import com.hermesandroid.relay.network.models.SkillInfo
|
||||
import com.hermesandroid.relay.network.models.SkillListResponse
|
||||
import com.hermesandroid.relay.network.models.UsageInfo
|
||||
import com.hermesandroid.relay.network.upstream.models.CreateSessionRequest
|
||||
import com.hermesandroid.relay.network.upstream.models.HermesSseEvent
|
||||
import com.hermesandroid.relay.network.upstream.models.MessageItem
|
||||
import com.hermesandroid.relay.network.upstream.models.MessageListResponse
|
||||
import com.hermesandroid.relay.network.upstream.models.RenameSessionRequest
|
||||
import com.hermesandroid.relay.network.upstream.models.SessionItem
|
||||
import com.hermesandroid.relay.network.upstream.models.SessionListResponse
|
||||
import com.hermesandroid.relay.network.upstream.models.SessionResponse
|
||||
import com.hermesandroid.relay.network.upstream.models.SkillInfo
|
||||
import com.hermesandroid.relay.network.upstream.models.SkillListResponse
|
||||
import com.hermesandroid.relay.network.upstream.models.UsageInfo
|
||||
import com.hermesandroid.relay.util.TurnLatencyTracer
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.withContext
|
||||
@@ -288,7 +288,7 @@ class HermesApiClient(
|
||||
|
||||
// --- Session CRUD ---
|
||||
|
||||
suspend fun listSessionsResult(limit: Int = 50): Result<List<SessionItem>> = withContext(Dispatchers.IO) {
|
||||
suspend fun listSessionsResult(limit: Int = 200): Result<List<SessionItem>> = withContext(Dispatchers.IO) {
|
||||
try {
|
||||
val request = authRequest("$baseUrl/api/sessions?limit=$limit").get().build()
|
||||
client.newCall(request).execute().use { response ->
|
||||
@@ -308,7 +308,7 @@ class HermesApiClient(
|
||||
}
|
||||
}
|
||||
|
||||
suspend fun listSessions(limit: Int = 50): List<SessionItem> =
|
||||
suspend fun listSessions(limit: Int = 200): List<SessionItem> =
|
||||
listSessionsResult(limit).getOrElse { emptyList() }
|
||||
|
||||
suspend fun createSessionResult(
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import com.hermesandroid.relay.data.AgentDisplay
|
||||
import com.hermesandroid.relay.data.Attachment
|
||||
@@ -1,10 +1,10 @@
|
||||
package com.hermesandroid.relay.network
|
||||
package com.hermesandroid.relay.network.upstream
|
||||
|
||||
import android.content.Context
|
||||
import com.hermesandroid.relay.data.VoiceAudioRoute
|
||||
import com.hermesandroid.relay.network.shared.VoiceAudioClient
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.withContext
|
||||
import kotlinx.serialization.encodeToString
|
||||
import kotlinx.serialization.json.Json
|
||||
import kotlinx.serialization.json.JsonObject
|
||||
import kotlinx.serialization.json.JsonPrimitive
|
||||
@@ -21,102 +21,12 @@ import java.io.IOException
|
||||
import java.util.Base64
|
||||
import java.util.concurrent.TimeUnit
|
||||
|
||||
interface VoiceAudioClient {
|
||||
val route: VoiceAudioRoute
|
||||
suspend fun transcribe(audioFile: File): Result<String>
|
||||
suspend fun synthesize(text: String): Result<File>
|
||||
}
|
||||
|
||||
class RelayVoiceAudioClientAdapter(
|
||||
private val relayVoiceClient: RelayVoiceClient,
|
||||
) : VoiceAudioClient {
|
||||
override val route: VoiceAudioRoute = VoiceAudioRoute.Relay
|
||||
|
||||
override suspend fun transcribe(audioFile: File): Result<String> =
|
||||
relayVoiceClient.transcribe(audioFile)
|
||||
|
||||
override suspend fun synthesize(text: String): Result<File> =
|
||||
relayVoiceClient.synthesize(text)
|
||||
}
|
||||
|
||||
/**
|
||||
* Routes each STT/TTS call to the Standard (dashboard) or Relay voice client.
|
||||
*
|
||||
* Auto preference order is **Relay first, then Standard**: a paired Relay is
|
||||
* the purpose-built mobile facade — profile-aware voice config, no dashboard
|
||||
* sign-in dependency — so users who installed the plugin keep the richer
|
||||
* path. Standard is the zero-plugin route for vanilla Hermes installs and is
|
||||
* used whenever Relay isn't configured/paired (or fails mid-call). Power
|
||||
* users can force either route in Voice Settings.
|
||||
*/
|
||||
class AutoVoiceAudioClient(
|
||||
private val standardClient: VoiceAudioClient,
|
||||
private val relayClient: VoiceAudioClient,
|
||||
private val routeProvider: () -> VoiceAudioRoute,
|
||||
private val standardReadyProvider: () -> Boolean,
|
||||
private val relayReadyProvider: () -> Boolean,
|
||||
) : VoiceAudioClient {
|
||||
override val route: VoiceAudioRoute
|
||||
get() = routeProvider()
|
||||
|
||||
override suspend fun transcribe(audioFile: File): Result<String> =
|
||||
runWithSelectedRoute { it.transcribe(audioFile) }
|
||||
|
||||
override suspend fun synthesize(text: String): Result<File> =
|
||||
runWithSelectedRoute { it.synthesize(text) }
|
||||
|
||||
private suspend fun <T> runWithSelectedRoute(
|
||||
block: suspend (VoiceAudioClient) -> Result<T>,
|
||||
): Result<T> {
|
||||
return when (routeProvider()) {
|
||||
VoiceAudioRoute.Standard -> {
|
||||
if (!standardReadyProvider()) {
|
||||
Result.failure(
|
||||
IllegalStateException(
|
||||
"Standard Hermes voice is not available — check dashboard sign-in in Manage",
|
||||
),
|
||||
)
|
||||
} else {
|
||||
block(standardClient)
|
||||
}
|
||||
}
|
||||
VoiceAudioRoute.Relay -> {
|
||||
if (!relayReadyProvider()) {
|
||||
Result.failure(IllegalStateException("Relay voice is not available"))
|
||||
} else {
|
||||
block(relayClient)
|
||||
}
|
||||
}
|
||||
VoiceAudioRoute.Auto -> runAuto(block)
|
||||
}
|
||||
}
|
||||
|
||||
private suspend fun <T> runAuto(
|
||||
block: suspend (VoiceAudioClient) -> Result<T>,
|
||||
): Result<T> {
|
||||
var relayFailure: Result<T>? = null
|
||||
if (relayReadyProvider()) {
|
||||
val result = block(relayClient)
|
||||
if (result.isSuccess || !standardReadyProvider()) return result
|
||||
relayFailure = result
|
||||
}
|
||||
if (standardReadyProvider()) {
|
||||
val result = block(standardClient)
|
||||
if (result.isSuccess) return result
|
||||
return relayFailure ?: result
|
||||
}
|
||||
return relayFailure ?: Result.failure(
|
||||
IllegalStateException("Voice needs a reachable Hermes dashboard or Relay voice route"),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Standard (no-plugin) voice client — speaks the upstream **dashboard web
|
||||
* server** contract that hermes-desktop's voice mode uses:
|
||||
*
|
||||
* POST {dashboard}/api/audio/transcribe {data_url, mime_type} → {ok, transcript}
|
||||
* POST {dashboard}/api/audio/speak {text} → {ok, data_url, mime_type}
|
||||
* POST {dashboard}/api/audio/transcribe {data_url, mime_type} to {ok, transcript}
|
||||
* POST {dashboard}/api/audio/speak {text} to {ok, data_url, mime_type}
|
||||
*
|
||||
* These routes live on `hermes_cli/web_server.py` (:9119 by convention), NOT
|
||||
* on the API server (:8642) — current upstream api_server advertises
|
||||
@@ -124,13 +34,19 @@ class AutoVoiceAudioClient(
|
||||
* cookie session (gated_auth_middleware), so [okHttpClient] must carry the
|
||||
* same per-connection cookie jar the Manage tab signs in with; an API bearer
|
||||
* header is meaningless on this surface. Revisit when upstream PR #8199
|
||||
* lands the `/v1/audio` routes on the API server (docs/upstream-contributions.md §6).
|
||||
* lands the `/v1/audio` routes on the API server (docs/upstream-contributions.md section 6).
|
||||
* (No glob spellings in block comments — Kotlin block comments nest.)
|
||||
*/
|
||||
class StandardHermesVoiceClient(
|
||||
private val context: Context,
|
||||
private val okHttpClient: OkHttpClient,
|
||||
private val dashboardUrlProvider: () -> String?,
|
||||
// Active chat profile name (null = default/launch). Sent DEFENSIVELY on
|
||||
// /api/audio/speak: upstream `TTSSpeakRequest` is text-only and Pydantic
|
||||
// ignores extra fields, so this is harmless today and forward-compatible if
|
||||
// upstream ever adds profile-aware TTS. Until then, standard voice remains
|
||||
// the host's global TTS (see VoiceViewModel's standard-voice profile notice).
|
||||
private val profileProvider: () -> String? = { null },
|
||||
private val json: Json = Json {
|
||||
ignoreUnknownKeys = true
|
||||
isLenient = true
|
||||
@@ -150,6 +66,14 @@ class StandardHermesVoiceClient(
|
||||
if (!audioFile.exists() || audioFile.length() == 0L) {
|
||||
return@withContext Result.failure(IOException("Audio file missing or empty: ${audioFile.name}"))
|
||||
}
|
||||
// Upstream caps decoded transcription audio at 25 MB (web_server.py
|
||||
// _MAX_TRANSCRIPTION_UPLOAD_BYTES → HTTP 413). The decoded size equals
|
||||
// the file size, so guard here to avoid a wasted ~33 MB base64 upload.
|
||||
if (audioFile.length() > MAX_TRANSCRIBE_BYTES) {
|
||||
return@withContext Result.failure(
|
||||
IOException("Recording too long for Hermes - try a shorter utterance"),
|
||||
)
|
||||
}
|
||||
|
||||
val dataUrl = buildAudioDataUrl(audioFile)
|
||||
val payload = buildJsonObject {
|
||||
@@ -181,7 +105,12 @@ class StandardHermesVoiceClient(
|
||||
return@withContext Result.failure(IllegalArgumentException("Cannot synthesize blank text"))
|
||||
}
|
||||
|
||||
val payload = buildJsonObject { put("text", cleanText) }
|
||||
val payload = buildJsonObject {
|
||||
put("text", cleanText)
|
||||
// Defensive only — upstream /api/audio/speak ignores it (text-only
|
||||
// TTSSpeakRequest). Omitted for the default profile.
|
||||
profileProvider()?.trim()?.takeIf { it.isNotBlank() }?.let { put("profile", it) }
|
||||
}
|
||||
val request = Request.Builder()
|
||||
.url("$baseUrl/api/audio/speak")
|
||||
.post(json.encodeToString(JsonObject.serializer(), payload).toRequestBody(JSON_MEDIA))
|
||||
@@ -241,8 +170,10 @@ class StandardHermesVoiceClient(
|
||||
val body = runCatching { response.body.string() }.getOrDefault("")
|
||||
val detail = body.takeIf { it.isNotBlank() } ?: response.message
|
||||
val message = when (response.code) {
|
||||
400 -> "$operation rejected that input - ${detail.ifBlank { "bad request" }}"
|
||||
401, 403 -> "$operation needs dashboard sign-in - open Manage to sign in"
|
||||
404 -> "$operation unavailable on this Hermes build - update hermes-agent or use Relay"
|
||||
413 -> "Recording too long for Hermes - try a shorter utterance"
|
||||
in 500..599 -> "$operation failed - server error HTTP ${response.code}"
|
||||
else -> "$operation failed - HTTP ${response.code}: $detail"
|
||||
}
|
||||
@@ -292,5 +223,9 @@ class StandardHermesVoiceClient(
|
||||
|
||||
private companion object {
|
||||
val JSON_MEDIA = "application/json".toMediaType()
|
||||
|
||||
// Matches upstream _MAX_TRANSCRIPTION_UPLOAD_BYTES (web_server.py): the
|
||||
// dashboard rejects decoded transcription audio above 25 MB with 413.
|
||||
const val MAX_TRANSCRIBE_BYTES = 25L * 1024 * 1024
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
package com.hermesandroid.relay.network.models
|
||||
package com.hermesandroid.relay.network.upstream.models
|
||||
|
||||
import kotlinx.serialization.ExperimentalSerializationApi
|
||||
import kotlinx.serialization.KSerializer
|
||||
@@ -16,6 +16,7 @@ import kotlinx.serialization.json.JsonObject
|
||||
import kotlinx.serialization.json.JsonPrimitive
|
||||
import kotlinx.serialization.json.jsonArray
|
||||
import kotlinx.serialization.json.jsonPrimitive
|
||||
import java.time.Instant
|
||||
|
||||
/**
|
||||
* Models for the Hermes /api/sessions REST API.
|
||||
@@ -75,6 +76,43 @@ object FlexibleIdNonNullSerializer : KSerializer<String> {
|
||||
}
|
||||
}
|
||||
|
||||
/** Timestamp serializer for Hermes session metadata.
|
||||
*
|
||||
* Upstream currently returns epoch seconds for `started_at` / `last_active`;
|
||||
* some documented surfaces use ISO strings for update-style fields. Decode
|
||||
* both into epoch seconds so callers can convert once at the UI boundary.
|
||||
*/
|
||||
@OptIn(ExperimentalSerializationApi::class)
|
||||
object FlexibleTimestampSerializer : KSerializer<Double?> {
|
||||
override val descriptor = PrimitiveSerialDescriptor("FlexibleTimestamp", PrimitiveKind.DOUBLE)
|
||||
|
||||
override fun deserialize(decoder: Decoder): Double? {
|
||||
return try {
|
||||
val jsonDecoder = decoder as? JsonDecoder
|
||||
?: return decoder.decodeDouble()
|
||||
val element = jsonDecoder.decodeJsonElement()
|
||||
when (element) {
|
||||
is JsonNull -> null
|
||||
is JsonPrimitive -> parseTimestamp(element.content)
|
||||
else -> null
|
||||
}
|
||||
} catch (_: Exception) {
|
||||
null
|
||||
}
|
||||
}
|
||||
|
||||
override fun serialize(encoder: Encoder, value: Double?) {
|
||||
if (value != null) encoder.encodeDouble(value) else encoder.encodeNull()
|
||||
}
|
||||
|
||||
private fun parseTimestamp(raw: String): Double? {
|
||||
val trimmed = raw.trim()
|
||||
if (trimmed.isBlank()) return null
|
||||
trimmed.toDoubleOrNull()?.let { return it }
|
||||
return runCatching { Instant.parse(trimmed).toEpochMilli() / 1000.0 }.getOrNull()
|
||||
}
|
||||
}
|
||||
|
||||
// --- Session CRUD responses ---
|
||||
|
||||
@Serializable
|
||||
@@ -102,13 +140,32 @@ data class SessionItem(
|
||||
val title: String? = null,
|
||||
val model: String? = null,
|
||||
val source: String? = null,
|
||||
@SerialName("started_at") val startedAt: Double? = null,
|
||||
@SerialName("ended_at") val endedAt: Double? = null,
|
||||
@SerialName("started_at")
|
||||
@Serializable(with = FlexibleTimestampSerializer::class)
|
||||
val startedAt: Double? = null,
|
||||
@SerialName("ended_at")
|
||||
@Serializable(with = FlexibleTimestampSerializer::class)
|
||||
val endedAt: Double? = null,
|
||||
@SerialName("last_active")
|
||||
@Serializable(with = FlexibleTimestampSerializer::class)
|
||||
val lastActive: Double? = null,
|
||||
@SerialName("last_activity")
|
||||
@Serializable(with = FlexibleTimestampSerializer::class)
|
||||
val lastActivity: Double? = null,
|
||||
@SerialName("last_activity_at")
|
||||
@Serializable(with = FlexibleTimestampSerializer::class)
|
||||
val lastActivityAt: Double? = null,
|
||||
@SerialName("updated_at")
|
||||
@Serializable(with = FlexibleTimestampSerializer::class)
|
||||
val updatedAt: Double? = null,
|
||||
@SerialName("message_count") val messageCount: Int? = null,
|
||||
@SerialName("tool_call_count") val toolCallCount: Int? = null,
|
||||
@SerialName("input_tokens") val inputTokens: Int? = null,
|
||||
@SerialName("output_tokens") val outputTokens: Int? = null
|
||||
)
|
||||
) {
|
||||
val resolvedLastActivity: Double?
|
||||
get() = lastActive ?: lastActivity ?: lastActivityAt ?: updatedAt
|
||||
}
|
||||
|
||||
@Serializable
|
||||
data class CreateSessionRequest(
|
||||