diff --git a/.github/RELEASING.md b/.github/RELEASING.md
index 7840adcd200..e46742293e9 100644
--- a/.github/RELEASING.md
+++ b/.github/RELEASING.md
@@ -10,6 +10,14 @@ formulae, and container images.
The verification job runs the Go tests and `govulncheck` before any publishing
job starts. The vulnerability scan is fail-closed by default.
+CI and release jobs require Go `~1.26.9`: at least 1.26.9, while accepting
+newer 1.26 patches. This security floor is independent of the development
+minimum in `server/go.mod`. A plain `1.26.x` can resolve to an older patch
+while the `setup-go` version manifest catches up with an official Go release.
+When raising the security floor, update all Go workflows and the pinned
+Go builder image in `Dockerfile` together so shipped binaries also receive
+the fixes.
+
## Emergency vulnerability-scan bypass
Use the bypass only when `govulncheck` itself or its live vulnerability database
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 7b405b41446..b7c09227b02 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -72,7 +72,7 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- go-version: "1.26.x"
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
@@ -348,9 +348,9 @@ jobs:
id: go
uses: actions/setup-go@v5
with:
- # Deliberately independent from the minimum patch in server/go.mod:
- # CI follows the newest 1.26 patch available to setup-go.
- go-version: "1.26.x"
+ # Keep the security floor even when the setup-go manifest lags;
+ # continue accepting newer patches within Go 1.26.
+ go-version: "~1.26.9"
check-latest: true
cache: false
@@ -450,7 +450,7 @@ jobs:
id: go
uses: actions/setup-go@v5
with:
- go-version: "1.26.x"
+ go-version: "~1.26.9"
check-latest: true
cache: false
@@ -488,9 +488,9 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- # Deliberately independent from the minimum patch in server/go.mod:
- # CI follows the newest 1.26 patch available to setup-go.
- go-version: "1.26.x"
+ # Keep the security floor even when the setup-go manifest lags;
+ # continue accepting newer patches within Go 1.26.
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
@@ -527,9 +527,9 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- # Deliberately independent from the minimum patch in server/go.mod:
- # CI follows the newest 1.26 patch available to setup-go.
- go-version: "1.26.x"
+ # Keep the security floor even when the setup-go manifest lags;
+ # continue accepting newer patches within Go 1.26.
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
@@ -788,7 +788,7 @@ jobs:
- uses: actions/checkout@v6
- uses: actions/setup-go@v5
with:
- go-version: "1.26.x"
+ go-version: "~1.26.9"
cache-dependency-path: server/go.sum
- name: Verify Cursor background lifecycle and watchdog races
shell: bash
diff --git a/.github/workflows/desktop-smoke.yml b/.github/workflows/desktop-smoke.yml
index 6573ea6cb11..4a66647886a 100644
--- a/.github/workflows/desktop-smoke.yml
+++ b/.github/workflows/desktop-smoke.yml
@@ -30,9 +30,9 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- # Deliberately independent from the minimum patch in server/go.mod:
- # CI follows the newest 1.26 patch available to setup-go.
- go-version: "1.26.x"
+ # Keep the security floor even when the setup-go manifest lags;
+ # continue accepting newer patches within Go 1.26.
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
diff --git a/.github/workflows/openclaw-config-smoke.yml b/.github/workflows/openclaw-config-smoke.yml
index 867ecf06c83..56fa1a9c555 100644
--- a/.github/workflows/openclaw-config-smoke.yml
+++ b/.github/workflows/openclaw-config-smoke.yml
@@ -35,7 +35,7 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- go-version: "1.26.x"
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml
index 448dc861b2c..c46ed315fa2 100644
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -54,9 +54,9 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- # Deliberately independent from the minimum patch in server/go.mod:
- # releases follow the newest 1.26 patch available to setup-go.
- go-version: "1.26.x"
+ # Keep the security floor even when the setup-go manifest lags;
+ # continue accepting newer patches within Go 1.26.
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
@@ -98,9 +98,9 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- # Deliberately independent from the minimum patch in server/go.mod:
- # releases follow the newest 1.26 patch available to setup-go.
- go-version: "1.26.x"
+ # Keep the security floor even when the setup-go manifest lags;
+ # continue accepting newer patches within Go 1.26.
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
@@ -424,9 +424,9 @@ jobs:
- name: Setup Go
uses: actions/setup-go@v5
with:
- # Deliberately independent from the minimum patch in server/go.mod:
- # releases follow the newest 1.26 patch available to setup-go.
- go-version: "1.26.x"
+ # Keep the security floor even when the setup-go manifest lags;
+ # continue accepting newer patches within Go 1.26.
+ go-version: "~1.26.9"
check-latest: true
cache-dependency-path: server/go.sum
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index be618070c06..90624e67722 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -44,7 +44,7 @@ This keeps Docker simple while still isolating schema and data.
- Node.js `22`
- `pnpm` `10.28.2`
-- Go `1.26.6`
+- Go `1.26.9`
- Docker
## Important Rules
diff --git a/Dockerfile b/Dockerfile
index 8b0335a26f0..6710b85c0cf 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -1,5 +1,5 @@
# --- Build stage ---
-FROM golang:1.26-alpine AS builder
+FROM golang:1.26.9-alpine AS builder
RUN apk add --no-cache git
diff --git a/README.md b/README.md
index 7aa8c9d37dc..4fd378aa9e5 100644
--- a/README.md
+++ b/README.md
@@ -246,7 +246,7 @@ The docs are also available in [简体中文](https://multica.ai/docs/zh), [日
Contributors: start with the [Contributing Guide](CONTRIBUTING.md).
-**Prerequisites:** [Node.js](https://nodejs.org/) 22, [pnpm](https://pnpm.io/) 10.28.2, [Go](https://go.dev/) 1.26.6, [Docker](https://www.docker.com/)
+**Prerequisites:** [Node.js](https://nodejs.org/) 22, [pnpm](https://pnpm.io/) 10.28.2, [Go](https://go.dev/) 1.26.9, [Docker](https://www.docker.com/)
```bash
make dev
diff --git a/README.zh.md b/README.zh.md
index e3df62aba8c..5317472c33b 100644
--- a/README.zh.md
+++ b/README.zh.md
@@ -238,7 +238,7 @@ Multica 不自带模型。它驱动的是你本来就装好、登录好的那些
想参与贡献,先看[贡献指南](CONTRIBUTING.md)。
-**环境要求:**[Node.js](https://nodejs.org/) 22、[pnpm](https://pnpm.io/) 10.28.2、[Go](https://go.dev/) 1.26.6、[Docker](https://www.docker.com/)
+**环境要求:**[Node.js](https://nodejs.org/) 22、[pnpm](https://pnpm.io/) 10.28.2、[Go](https://go.dev/) 1.26.9、[Docker](https://www.docker.com/)
```bash
make dev
diff --git a/SELF_HOSTING_ADVANCED.md b/SELF_HOSTING_ADVANCED.md
index 97db85239f0..4a59f3d242c 100644
--- a/SELF_HOSTING_ADVANCED.md
+++ b/SELF_HOSTING_ADVANCED.md
@@ -428,7 +428,7 @@ If you are upgrading from a binary that pre-dates MUL-2957 (or the auto-hook fai
If you prefer to build and run services manually:
-**Prerequisites:** Go 1.26.6, Node.js 22, pnpm 10.28.2, PostgreSQL 17 — a stock install is enough, see [Database Setup](#database-setup).
+**Prerequisites:** Go 1.26.9, Node.js 22, pnpm 10.28.2, PostgreSQL 17 — a stock install is enough, see [Database Setup](#database-setup).
```bash
# Start your PostgreSQL (or use: docker compose up -d postgres)
diff --git a/apps/desktop/src/main/index.ts b/apps/desktop/src/main/index.ts
index a2f7bbc7b99..7a1c13372dd 100644
--- a/apps/desktop/src/main/index.ts
+++ b/apps/desktop/src/main/index.ts
@@ -312,7 +312,8 @@ function createWindow(): BrowserWindow {
minWidth: 900,
minHeight: 600,
titleBarStyle: "hiddenInset",
- trafficLightPosition: { x: 16, y: 17 },
+ // Center the native controls in the main window's 40px tab toolbar.
+ trafficLightPosition: { x: 16, y: 13 },
show: false,
autoHideMenuBar: true,
// Windows/Linux pick up the window/taskbar icon from this option.
diff --git a/apps/desktop/src/renderer/src/components/desktop-layout.tsx b/apps/desktop/src/renderer/src/components/desktop-layout.tsx
index 70ec88f5476..abf73ff7f46 100644
--- a/apps/desktop/src/renderer/src/components/desktop-layout.tsx
+++ b/apps/desktop/src/renderer/src/components/desktop-layout.tsx
@@ -37,7 +37,7 @@ import { TabContent } from "./tab-content";
import { WindowOverlay } from "./window-overlay";
import { WindowToolbar, WINDOW_TOOLBAR_CLEARANCE } from "./window-toolbar";
-const TOP_BAR_HEIGHT_CLASS = "h-12";
+const TOP_BAR_HEIGHT_CLASS = "h-10";
const toolbarMotion = {
type: "spring",
stiffness: 420,
diff --git a/apps/desktop/src/renderer/src/components/tab-bar.tsx b/apps/desktop/src/renderer/src/components/tab-bar.tsx
index e366fc231aa..8cd04478d2c 100644
--- a/apps/desktop/src/renderer/src/components/tab-bar.tsx
+++ b/apps/desktop/src/renderer/src/components/tab-bar.tsx
@@ -307,7 +307,10 @@ function SortableTabItem({
title={tab.pinned ? `${title} (pinned)` : undefined}
style={{ WebkitAppRegion: "no-drag" } as React.CSSProperties}
className={cn(
- "group relative flex size-full min-w-0 items-center gap-1.5 px-2.5 text-caption transition-colors",
+ // The 36px frame contains a centered 32px content row and a 4px
+ // bottom connector. Keep the connector clickable without letting it
+ // shift the icon, title, or hover actions below the toolbar center.
+ "group relative flex size-full min-w-0 items-center gap-1.5 px-2.5 pb-1 text-caption transition-colors",
"select-none cursor-default",
isActive
? "font-medium text-foreground"
@@ -392,7 +395,7 @@ function SortableTabItem({
) : (
)}
{showSeparator && (
@@ -401,7 +404,7 @@ function SortableTabItem({
// arrives rather than lingering 2px off its rounded edge.
)}
@@ -451,7 +454,7 @@ function SortableTabItem({
{showAddedHighlight && (
@@ -693,7 +696,7 @@ export function TabBar() {
unpinnedCount > 0 && (
)}
diff --git a/apps/desktop/src/renderer/src/components/window-toolbar.tsx b/apps/desktop/src/renderer/src/components/window-toolbar.tsx
index a6d04797e21..d08682021fe 100644
--- a/apps/desktop/src/renderer/src/components/window-toolbar.tsx
+++ b/apps/desktop/src/renderer/src/components/window-toolbar.tsx
@@ -246,7 +246,7 @@ export function WindowToolbar() {
` | プロジェクトのステータスを変更 | |
| | `resource list/add/update/remove` | プロジェクトの追加リソースを管理 | `--type`、`--url`、`--local-path`、`--daemon-id`、`--execution-mode`(ローカルディレクトリの `in_place` / `worktree`) |
| `label` | `list/get/create/update/delete` | ワークスペースのラベルを管理 | `list`: `--resource-type`(`issue` または `skill`)、`--full-id`;`create`: `--name`、`--color`、`--resource-type`(`issue` または `skill`)、`--description` |
-| `property` | `list/get/create/update/archive/unarchive` | ワークスペースのカスタムプロパティを管理 | `create`: `--name`、`--type`(`text`、`number`、`select`、`multi_select`、`date`、`checkbox`、`url`、`actor`、`multi_actor`)、`--option`(繰り返し可、select 系のみ);`list`: `--include-archived`;タイプは作成後に変更できません |
+| `property` | `list/get/create/update/archive/unarchive` | ワークスペースのカスタムプロパティを管理 | `create`: `--name`、`--type`(`text`、`number`、`select`、`multi_select`、`date`、`checkbox`、`url`、`actor`、`multi_actor`、`multi_text`、`multi_url`)、`--option`(繰り返し可、select 系のみ);`list`: `--include-archived`;タイプは作成後に変更できません |
| `agent` | `list/get/create/update/archive/restore` | エージェントを管理 | `--name`、`--runtime-id`(`create` では必須)、`--instructions`、`--conversation-starters`、`--model`、`--thinking-level`、`--mcp-config`、`--permission-mode`、`--max-concurrent-tasks` |
| | `copy ` | 新しいエージェントとして複製。元のエージェントは変更されません | `--name`(デフォルトは元の名前 + ` (copy)`)、`--runtime-id`(別のランタイムへ複製する場合は `--model` の同時指定が必須)、`--no-skills`;`custom_env`、`mcp_config`、`runtime_config` などの機密設定は複製されないため、`create` と同じフラグで再指定してください |
| | `tasks ` | エージェントの実行を表示 | |
diff --git a/apps/docs/content/docs/cli.ko.mdx b/apps/docs/content/docs/cli.ko.mdx
index dee41d0fe06..872d47884b3 100644
--- a/apps/docs/content/docs/cli.ko.mdx
+++ b/apps/docs/content/docs/cli.ko.mdx
@@ -250,7 +250,7 @@ CLI 설정 파일에는 사용자를 대신해 Multica에 접근할 수 있는 t
| | `status ` | 프로젝트 상태 변경 | |
| | `resource list/add/update/remove` | 프로젝트 추가 리소스 관리 | `--type`, `--url`, `--local-path`, `--daemon-id`, `--execution-mode`(로컬 디렉터리의 `in_place` / `worktree`) |
| `label` | `list/get/create/update/delete` | 워크스페이스 라벨 관리 | `list`: `--resource-type`(`issue` 또는 `skill`), `--full-id`; `create`: `--name`, `--color`, `--resource-type`(`issue` 또는 `skill`), `--description` |
-| `property` | `list/get/create/update/archive/unarchive` | 워크스페이스 사용자 지정 속성 관리 | `create`: `--name`, `--type`(`text`, `number`, `select`, `multi_select`, `date`, `checkbox`, `url`, `actor`, `multi_actor`), `--option`(반복 가능, select 유형만). `list`: `--include-archived`. 유형은 생성 후 변경 불가 |
+| `property` | `list/get/create/update/archive/unarchive` | 워크스페이스 사용자 지정 속성 관리 | `create`: `--name`, `--type`(`text`, `number`, `select`, `multi_select`, `date`, `checkbox`, `url`, `actor`, `multi_actor`, `multi_text`, `multi_url`), `--option`(반복 가능, select 유형만). `list`: `--include-archived`. 유형은 생성 후 변경 불가 |
| `agent` | `list/get/create/update/archive/restore` | 에이전트 관리 | `--name`, `--runtime-id`(`create`에서 필수), `--instructions`, `--conversation-starters`, `--model`, `--thinking-level`, `--mcp-config`, `--permission-mode`, `--max-concurrent-tasks` |
| | `copy ` | 원래 에이전트에 영향 없이 새 에이전트로 복사 | `--name`(기본값은 원래 이름 + ` (copy)`), `--runtime-id`(다른 런타임으로 복사할 때 `--model`도 필수), `--no-skills`. `custom_env`, `mcp_config`, `runtime_config` 같은 기밀 설정은 복사되지 않으며 `create`와 같은 flag로 다시 제공해야 함 |
| | `tasks ` | 에이전트 실행 확인 | |
diff --git a/apps/docs/content/docs/cli.mdx b/apps/docs/content/docs/cli.mdx
index fae681ab278..a5af8453374 100644
--- a/apps/docs/content/docs/cli.mdx
+++ b/apps/docs/content/docs/cli.mdx
@@ -273,7 +273,7 @@ The tables below cover every current top-level command, grouped the way the CLI
| | `status ` | Change project status | |
| | `resource list/add/update/remove` | Manage project resources | `--type`, `--url`, `--local-path`, `--daemon-id`, `--execution-mode` (`in_place` / `worktree` for a local directory) |
| `label` | `list/get/create/update/delete` | Manage workspace labels | `list`: `--resource-type` (`issue` or `skill`), `--full-id`; `create`: `--name`, `--color`, `--resource-type` (`issue` or `skill`), `--description` |
-| `property` | `list/get/create/update/archive/unarchive` | Manage workspace custom properties | `create`: `--name`, `--type` (`text`, `number`, `select`, `multi_select`, `date`, `checkbox`, `url`, `actor`, `multi_actor`), `--option` (repeatable, select types only); `list`: `--include-archived`; the type cannot be changed after creation |
+| `property` | `list/get/create/update/archive/unarchive` | Manage workspace custom properties | `create`: `--name`, `--type` (`text`, `number`, `select`, `multi_select`, `date`, `checkbox`, `url`, `actor`, `multi_actor`, `multi_text`, `multi_url`), `--option` (repeatable, select types only); `list`: `--include-archived`; the type cannot be changed after creation |
| `agent` | `list/get/create/update/archive/restore` | Manage agents | `--name`, `--runtime-id` (required for `create`), `--instructions`, `--conversation-starters`, `--model`, `--thinking-level`, `--mcp-config`, `--permission-mode`, `--max-concurrent-tasks` |
| | `copy ` | Copy into a new agent; the original is untouched | `--name` (defaults to the original name plus ` (copy)`), `--runtime-id` (copying to another runtime also requires `--model`), `--no-skills`; secret configuration such as `custom_env`, `mcp_config`, and `runtime_config` is not copied — re-provide it with the same flags as `create` |
| | `tasks ` | View an agent's runs | |
diff --git a/apps/docs/content/docs/cli.zh.mdx b/apps/docs/content/docs/cli.zh.mdx
index 65eb81c567d..0aedabad6f7 100644
--- a/apps/docs/content/docs/cli.zh.mdx
+++ b/apps/docs/content/docs/cli.zh.mdx
@@ -250,7 +250,7 @@ CLI 配置文件包含可代表你访问 Multica 的令牌。不要提交到仓
| | `status ` | 修改项目状态 | |
| | `resource list/add/update/remove` | 管理项目附加资源 | `--type`、`--url`、`--local-path`、`--daemon-id`、`--execution-mode`(本地目录的 `in_place` / `worktree`) |
| `label` | `list/get/create/update/delete` | 管理工作区标签 | `list`:`--resource-type`(`issue` 或 `skill`)、`--full-id`;`create`:`--name`、`--color`、`--resource-type`(`issue` 或 `skill`)、`--description` |
-| `property` | `list/get/create/update/archive/unarchive` | 管理工作区自定义属性 | `create`:`--name`、`--type`(`text`、`number`、`select`、`multi_select`、`date`、`checkbox`、`url`、`actor`、`multi_actor`)、`--option`(可重复,仅 select 类型);`list`:`--include-archived`;类型创建后不可修改 |
+| `property` | `list/get/create/update/archive/unarchive` | 管理工作区自定义属性 | `create`:`--name`、`--type`(`text`、`number`、`select`、`multi_select`、`date`、`checkbox`、`url`、`actor`、`multi_actor`、`multi_text`、`multi_url`)、`--option`(可重复,仅 select 类型);`list`:`--include-archived`;类型创建后不可修改 |
| `agent` | `list/get/create/update/archive/restore` | 管理智能体 | `--name`、`--runtime-id`(`create` 必填)、`--instructions`、`--conversation-starters`、`--model`、`--thinking-level`、`--mcp-config`、`--permission-mode`、`--max-concurrent-tasks` |
| | `copy ` | 复制为新智能体,原智能体不受影响 | `--name`(默认为原名加 ` (copy)`)、`--runtime-id`(复制到其他运行时时必须同时指定 `--model`)、`--no-skills`;`custom_env`、`mcp_config`、`runtime_config` 等机密配置不会复制,需用与 `create` 相同的 flag 重新提供 |
| | `tasks ` | 查看智能体的运行 | |
diff --git a/apps/docs/content/docs/self-host-quickstart.fr.mdx b/apps/docs/content/docs/self-host-quickstart.fr.mdx
index 1c256542e3e..6de98c74c95 100644
--- a/apps/docs/content/docs/self-host-quickstart.fr.mdx
+++ b/apps/docs/content/docs/self-host-quickstart.fr.mdx
@@ -315,7 +315,7 @@ docker compose -f docker-compose.selfhost.yml up -d
La version que vous exécutez au final est déterminée par `docker compose pull`, qui demande à GHCR vers quoi pointe le tag à cet instant. Une copie locale en retard de plusieurs mois téléchargera quand même les images `latest` du jour ; à l'inverse, `git pull` seul ne change rien tant que vous n'avez pas téléchargé les images et recréé les conteneurs.
-**Si vous avez figé `MULTICA_IMAGE_TAG`, aucune des deux commandes ne met quoi que ce soit à niveau.** Les deux images sont résolues en `${MULTICA_IMAGE_TAG:-latest}` (`docker-compose.selfhost.yml:42`, `:125`), et `.env.example` fournit `MULTICA_IMAGE_TAG=latest`. Si votre `.env` fige une version précise, `pull` récupère simplement de nouveau ce même tag et vous restez sur l'ancienne version — sans erreur ni avertissement. Vérifiez avant de mettre à niveau :
+**Si vous avez figé `MULTICA_IMAGE_TAG`, aucune des deux commandes ne met quoi que ce soit à niveau.** Les deux images sont résolues en `${MULTICA_IMAGE_TAG:-latest}` (`docker-compose.selfhost.yml`, `services.backend.image` / `services.frontend.image`), et `.env.example` fournit `MULTICA_IMAGE_TAG=latest`. Si votre `.env` fige une version précise, `pull` récupère simplement de nouveau ce même tag et vous restez sur l'ancienne version — sans erreur ni avertissement. Vérifiez avant de mettre à niveau :
```bash
grep MULTICA_IMAGE_TAG .env
@@ -354,7 +354,7 @@ Les migrations s'exécutent automatiquement au démarrage du backend ; celles qu
### Vérifier avec `/readyz`, pas `/health`
-`/health` est une sonde de **vivacité** (liveness) — elle renvoie `{"status":"ok"}` tant que le processus tourne, y compris lorsque les migrations ont échoué. `/readyz` (`server/cmd/server/router.go:680` ; `/healthz` en est un alias) vérifie la base de données et l'ensemble des migrations appliquées : c'est donc elle qui détecte une mise à niveau ratée :
+`/health` est une sonde de **vivacité** (liveness) — elle renvoie `{"status":"ok"}` tant que le processus tourne, y compris lorsque les migrations ont échoué. `/readyz` (`readyHandler`, enregistré dans `server/cmd/server/router.go` ; `/healthz` en est un alias) vérifie la base de données et l'ensemble des migrations appliquées : c'est donc elle qui détecte une mise à niveau ratée :
```bash
curl -s localhost:8080/readyz
diff --git a/apps/docs/content/docs/self-host-quickstart.ja.mdx b/apps/docs/content/docs/self-host-quickstart.ja.mdx
index 9ff5cc4bd5d..32e5e0872fc 100644
--- a/apps/docs/content/docs/self-host-quickstart.ja.mdx
+++ b/apps/docs/content/docs/self-host-quickstart.ja.mdx
@@ -314,7 +314,7 @@ docker compose -f docker-compose.selfhost.yml up -d
実際に動くバージョンを決めるのは `docker compose pull` で、そのタグが今どのイメージを指しているかを GHCR に問い合わせます。したがって数か月更新していない checkout でも今日の `latest` イメージを取得できますし、逆に `git pull` だけではイメージを取得してコンテナを作り直すまで何も変わりません。
-**`MULTICA_IMAGE_TAG` を固定している場合、どちらのコマンドでもアップグレードされません。** 両方のイメージが `${MULTICA_IMAGE_TAG:-latest}` として解決され(`docker-compose.selfhost.yml:42`、`:125`)、`.env.example` は `MULTICA_IMAGE_TAG=latest` を同梱しています。`.env` で特定のリリースに固定していると、`pull` は同じタグを取り直すだけで、古いバージョンのままです — エラーも警告も出ません。アップグレード前に確認してください。
+**`MULTICA_IMAGE_TAG` を固定している場合、どちらのコマンドでもアップグレードされません。** 両方のイメージが `${MULTICA_IMAGE_TAG:-latest}` として解決され(`docker-compose.selfhost.yml`:`services.backend.image`、`services.frontend.image`)、`.env.example` は `MULTICA_IMAGE_TAG=latest` を同梱しています。`.env` で特定のリリースに固定していると、`pull` は同じタグを取り直すだけで、古いバージョンのままです — エラーも警告も出ません。アップグレード前に確認してください。
```bash
grep MULTICA_IMAGE_TAG .env
@@ -353,7 +353,7 @@ migration は backend の起動時に自動で実行されます。過去デー
### `/health` ではなく `/readyz` で検証する
-`/health` は **liveness** プローブで、プロセスが生きている限り `{"status":"ok"}` を返します — migration が失敗していても同じです。`/readyz`(`server/cmd/server/router.go:680`、`/healthz` はそのエイリアス)はデータベースと適用済み migration の集合をチェックするため、失敗したアップグレードを捕捉できるのはこちらです。
+`/health` は **liveness** プローブで、プロセスが生きている限り `{"status":"ok"}` を返します — migration が失敗していても同じです。`/readyz`(`server/cmd/server/router.go` で登録された `readyHandler`、`/healthz` はそのエイリアス)はデータベースと適用済み migration の集合をチェックするため、失敗したアップグレードを捕捉できるのはこちらです。
```bash
curl -s localhost:8080/readyz
diff --git a/apps/docs/content/docs/self-host-quickstart.ko.mdx b/apps/docs/content/docs/self-host-quickstart.ko.mdx
index 0e36f9fd745..1f7950dcdc1 100644
--- a/apps/docs/content/docs/self-host-quickstart.ko.mdx
+++ b/apps/docs/content/docs/self-host-quickstart.ko.mdx
@@ -314,7 +314,7 @@ docker compose -f docker-compose.selfhost.yml up -d
실제로 어떤 버전이 뜰지는 `docker compose pull`이 결정합니다. 이 명령이 GHCR에 해당 태그가 지금 어떤 이미지를 가리키는지 물어보기 때문입니다. 그래서 몇 달 묵은 checkout에서도 오늘자 `latest` 이미지를 받을 수 있고, 반대로 `git pull`만 해서는 이미지를 받아 컨테이너를 다시 만들기 전까지 아무것도 바뀌지 않습니다.
-**`MULTICA_IMAGE_TAG`를 고정해 뒀다면 두 방법 모두 업그레이드되지 않습니다.** 두 이미지 모두 `${MULTICA_IMAGE_TAG:-latest}`로 해석되며(`docker-compose.selfhost.yml:42`, `:125`), `.env.example`에는 `MULTICA_IMAGE_TAG=latest`가 들어 있습니다. `.env`에서 특정 릴리스로 고정해 뒀다면 `pull`은 같은 태그를 다시 받아올 뿐이라 예전 버전에 그대로 머무릅니다 — 오류도, 경고도 없습니다. 업그레이드 전에 확인하세요.
+**`MULTICA_IMAGE_TAG`를 고정해 뒀다면 두 방법 모두 업그레이드되지 않습니다.** 두 이미지 모두 `${MULTICA_IMAGE_TAG:-latest}`로 해석되며(`docker-compose.selfhost.yml`, `services.backend.image` / `services.frontend.image`), `.env.example`에는 `MULTICA_IMAGE_TAG=latest`가 들어 있습니다. `.env`에서 특정 릴리스로 고정해 뒀다면 `pull`은 같은 태그를 다시 받아올 뿐이라 예전 버전에 그대로 머무릅니다 — 오류도, 경고도 없습니다. 업그레이드 전에 확인하세요.
```bash
grep MULTICA_IMAGE_TAG .env
@@ -353,7 +353,7 @@ Migration은 backend 시작 시 자동으로 실행됩니다. migration 103처
### `/health` 말고 `/readyz`로 검증하세요
-`/health`는 **liveness** 프로브라서 프로세스가 살아 있기만 하면 `{"status":"ok"}`를 반환합니다 — migration이 실패했더라도 마찬가지입니다. `/readyz`(`server/cmd/server/router.go:680`, `/healthz`는 별칭)는 데이터베이스와 적용된 migration 집합을 확인하므로, 잘못된 업그레이드를 잡아내는 것은 이쪽입니다.
+`/health`는 **liveness** 프로브라서 프로세스가 살아 있기만 하면 `{"status":"ok"}`를 반환합니다 — migration이 실패했더라도 마찬가지입니다. `/readyz`(`server/cmd/server/router.go`에 등록된 `readyHandler`, `/healthz`는 별칭)는 데이터베이스와 적용된 migration 집합을 확인하므로, 잘못된 업그레이드를 잡아내는 것은 이쪽입니다.
```bash
curl -s localhost:8080/readyz
diff --git a/apps/docs/content/docs/self-host-quickstart.mdx b/apps/docs/content/docs/self-host-quickstart.mdx
index 0e4bcc38bab..136d5326986 100644
--- a/apps/docs/content/docs/self-host-quickstart.mdx
+++ b/apps/docs/content/docs/self-host-quickstart.mdx
@@ -315,7 +315,7 @@ docker compose -f docker-compose.selfhost.yml up -d
The version you end up running is resolved by `docker compose pull`, which asks GHCR what the tag points at right now. A checkout that is months out of date will still pull today's `latest` images; conversely, `git pull` alone changes nothing until you pull images and recreate the containers.
-**If you pinned `MULTICA_IMAGE_TAG`, neither command upgrades anything.** Both images resolve as `${MULTICA_IMAGE_TAG:-latest}` (`docker-compose.selfhost.yml:42`, `:125`), and `.env.example` ships `MULTICA_IMAGE_TAG=latest`. If your `.env` pins an exact release, `pull` just re-fetches that same tag and you stay on the old version — no error, no warning. Check before you upgrade:
+**If you pinned `MULTICA_IMAGE_TAG`, neither command upgrades anything.** Both images resolve as `${MULTICA_IMAGE_TAG:-latest}` (`docker-compose.selfhost.yml`, `services.backend.image` / `services.frontend.image`), and `.env.example` ships `MULTICA_IMAGE_TAG=latest`. If your `.env` pins an exact release, `pull` just re-fetches that same tag and you stay on the old version — no error, no warning. Check before you upgrade:
```bash
grep MULTICA_IMAGE_TAG .env
@@ -354,7 +354,7 @@ Migrations run automatically when the backend starts; migrations that backfill h
### Verify with `/readyz`, not `/health`
-`/health` is a **liveness** probe — it returns `{"status":"ok"}` as long as the process is up, including when migrations failed. `/readyz` (`server/cmd/server/router.go:680`; `/healthz` is an alias) checks the database and the applied migration set, so it is the one that catches a bad upgrade:
+`/health` is a **liveness** probe — it returns `{"status":"ok"}` as long as the process is up, including when migrations failed. `/readyz` (`readyHandler`, registered in `server/cmd/server/router.go`; `/healthz` is an alias) checks the database and the applied migration set, so it is the one that catches a bad upgrade:
```bash
curl -s localhost:8080/readyz
diff --git a/apps/docs/content/docs/self-host-quickstart.zh.mdx b/apps/docs/content/docs/self-host-quickstart.zh.mdx
index 1703e15bec2..64830e92faa 100644
--- a/apps/docs/content/docs/self-host-quickstart.zh.mdx
+++ b/apps/docs/content/docs/self-host-quickstart.zh.mdx
@@ -314,7 +314,7 @@ docker compose -f docker-compose.selfhost.yml up -d
真正决定你跑哪个版本的是 `docker compose pull`:它去 GHCR 查这个 tag 当前指向哪个镜像。所以一个几个月没更新的 checkout 照样能拉到今天的 `latest` 镜像;反过来,只跑 `git pull` 而不拉镜像、不重建容器,什么都不会变。
-**如果你在 `.env` 里钉死了 `MULTICA_IMAGE_TAG`,两种写法都升不上去。** 两个镜像都写成 `${MULTICA_IMAGE_TAG:-latest}`(`docker-compose.selfhost.yml:42`、`:125`),而 `.env.example` 里默认是 `MULTICA_IMAGE_TAG=latest`。一旦你把它改成了具体版本,`pull` 只是把同一个 tag 重拉一遍,版本原地不动——不报错,也没有任何警告。升级前先确认:
+**如果你在 `.env` 里钉死了 `MULTICA_IMAGE_TAG`,两种写法都升不上去。** 两个镜像都写成 `${MULTICA_IMAGE_TAG:-latest}`(`docker-compose.selfhost.yml`:`services.backend.image`、`services.frontend.image`),而 `.env.example` 里默认是 `MULTICA_IMAGE_TAG=latest`。一旦你把它改成了具体版本,`pull` 只是把同一个 tag 重拉一遍,版本原地不动——不报错,也没有任何警告。升级前先确认:
```bash
grep MULTICA_IMAGE_TAG .env
@@ -353,7 +353,7 @@ Migration 在 backend 启动时自动运行;需要回填历史数据的 migrat
### 用 `/readyz` 验证,不要用 `/health`
-`/health` 是 **liveness** 探针,只要进程还活着就返回 `{"status":"ok"}`,migration 失败了它照样是 ok。`/readyz`(`server/cmd/server/router.go:680`,`/healthz` 是它的别名)会真正检查数据库和已应用的 migration 集合,升级出问题能被它拦下来:
+`/health` 是 **liveness** 探针,只要进程还活着就返回 `{"status":"ok"}`,migration 失败了它照样是 ok。`/readyz`(`server/cmd/server/router.go` 中注册的 `readyHandler`,`/healthz` 是它的别名)会真正检查数据库和已应用的 migration 集合,升级出问题能被它拦下来:
```bash
curl -s localhost:8080/readyz
diff --git a/apps/web/package.json b/apps/web/package.json
index 313896e9d73..cadf237720f 100644
--- a/apps/web/package.json
+++ b/apps/web/package.json
@@ -22,7 +22,7 @@
"fumadocs-core": "^15.5.2",
"fumadocs-mdx": "^12.0.3",
"lucide-react": "catalog:",
- "next": "^16.3.4",
+ "next": "^16.3.8",
"react": "catalog:",
"react-dom": "catalog:",
"shadcn": "^4.1.0",
diff --git a/docs/lark-typing-cleanup.md b/docs/lark-typing-cleanup.md
new file mode 100644
index 00000000000..a9212c0e3f2
--- /dev/null
+++ b/docs/lark-typing-cleanup.md
@@ -0,0 +1,104 @@
+# Lark Typing reaction cleanup
+
+Before sending an Add request, the server records each input in
+`channel_typing_reaction`, including the source message, session, workspace,
+installation and encrypted credential snapshot. A failed registration prevents
+the Add. Cleanup records survive deletion of the original session or installation.
+
+Terminal processing enumerates every input in the shared ledger, including
+coalesced inputs that are not the reply delivery's trigger. Cleanup ownership is
+monotonic: an Add returning after another replica observes termination cannot
+restore an active state. Failed flushes persist an input cutoff and context
+revision before detached ingestion, so taskless settlement also works when it
+precedes Add registration. Later inputs and rolled-back transactions do not
+authorize cleanup of the wrong turn.
+
+Known reaction IDs are deleted precisely. Both sweeps require the reaction's
+operator type to be `app` and its operator ID to match the installation's App ID;
+other apps and humans are preserved. An empty sweep cannot acknowledge an
+unfinished Add before its retention deadline. Legacy unregistered reactions retain
+the delivery-anchor fallback, which cannot reconstruct every historical input.
+
+A taskless indicator expires two minutes after registration. This recovers the
+badge when a hard crash loses an in-memory debounce timer after Add succeeds.
+It does not settle the input, cancel a task, or autonomously restart a lost run;
+the channel engine can process that input on subsequent traffic. A long burst
+can therefore lose its oldest taskless indicators before flushing. Inputs with
+active tasks and newer inputs keep their own indicators, within the seven-day
+maximum lifetime below.
+
+## Retry and retention limits
+
+- The worker runs every 30 seconds, claims at most 100 rows using `SKIP LOCKED`,
+ and applies a 30-second to one-hour backoff. Its pass has a ten-second budget;
+ each remote operation has a two-second budget. It does not hold database locks
+ across remote calls.
+- Claims advance retry counters before remote execution. A claimed row that does
+ not fit the pass budget is deferred too; the counter measures claims, not HTTP
+ attempts. Backlog latency and throughput are not guaranteed.
+- All indicators have a seven-day maximum lifetime from registration, including
+ running tasks, uncertain Adds and permanent credential failures. No new remote
+ retries are claimed after that deadline. An in-flight call has at most a
+ two-second budget. A late Add result cannot resurrect an expired ledger row.
+- Confirmed cleanup erases its encrypted snapshot immediately. Expiry erases the
+ snapshot and records `abandoned_at`, leaving `cleaned_at` null: this is an
+ explicit failed terminal outcome, not a claim that the remote badge disappeared.
+ A deleted workspace retains credentials only within the same remaining window;
+ it does not restart the retention clock. Revoked credentials, permanently failed
+ APIs or arbitrarily late remote writes can leave a badge requiring manual removal.
+- Expiration and pruning each get an independent two-second database budget before
+ remote retries, every 30 seconds. Each handles at most 1,000 rows with
+ `SKIP LOCKED`. Non-secret success/failure tombstones are pruned seven days after
+ their terminal timestamp. Physical redaction requires a healthy running worker
+ and database: downtime, locks or maintenance backlog can delay it beyond the
+ logical deadline. Database backups follow the operator's separate retention
+ policy. Do not describe this as a guaranteed wall-clock erasure SLA.
+- Each workspace has at most 1,000 pending ledger slots. A unique partial index
+ arbitrates concurrent registration; a full quota or a slot collision skips the
+ cosmetic Add without blocking input processing. Migration abandons/redacts any
+ existing pending rows beyond that quota. Active tasks themselves are unaffected.
+ Operators should treat repeated quota skips as a capacity incident, not normal
+ successful processing. Increasing this cap requires an explicit capacity review.
+
+Expiry emits a warning with workspace ID and count; maintenance failures emit an
+error; quota/contention skips and failed remote withdrawals emit warnings without
+credentials. Alert on these events and on an increasing overdue-redaction count.
+The following read-only query reports the durable operational state without
+exposing snapshots:
+
+```sql
+SELECT workspace_id,
+ count(*) FILTER (WHERE cleaned_at IS NULL AND abandoned_at IS NULL) AS pending,
+ count(*) FILTER (WHERE cleaned_at IS NULL AND abandoned_at IS NULL
+ AND created_at <= now() - interval '7 days') AS overdue_redaction,
+ count(*) FILTER (WHERE abandoned_at IS NOT NULL) AS abandoned,
+ min(created_at) FILTER (WHERE cleaned_at IS NULL AND abandoned_at IS NULL) AS oldest_pending
+FROM channel_typing_reaction GROUP BY workspace_id;
+```
+
+Remote retry claims still advance the backoff before execution; a 100-row batch
+can exceed the ten-second pass budget. Attempt counters count claims, not HTTP
+calls, and retry throughput is not guaranteed. This cannot starve the independent
+credential/GC passes, but requires workload-based capacity validation before rollout.
+
+## Migrations and rollback
+
+Migration 565 adds the ledger and taskless settlement flag. Migrations 566–568
+build the three indexes concurrently and register invalid-index cleanup hooks.
+Migration 569 adds the retention outcome and quota slots; 570–572 build quota,
+expiry and abandoned-record indexes with the same interrupted-build recovery.
+Interrupted builds are dropped and rebuilt before being marked applied; valid
+indexes are preserved on retry. There are no new foreign keys or cascading deletes.
+
+Validate upgrades using the complete historical schema, retained input rows,
+repeated `up`, and interrupted-index recovery. Four migrations against a minimal
+table fixture alone do not establish full upgrade compatibility.
+
+Application rollback preserves the schema, ledger and encryption keys. An older
+server may not enforce the new quota/expiry policy. Keep a compatible cleanup
+worker running or perform an explicitly reviewed offline redaction before retiring
+it; image rollback alone does not preserve these new retention guarantees.
+Bare `migrate down` rolls back all
+applied migrations in the directory, not just these eight, and must not be used as
+an application rollback shortcut. Review pending cleanup records before any
+separately planned schema removal.
diff --git a/packages/core/issues/queries.test.ts b/packages/core/issues/queries.test.ts
index 92470aa575d..1f7ef40c10b 100644
--- a/packages/core/issues/queries.test.ts
+++ b/packages/core/issues/queries.test.ts
@@ -1,5 +1,5 @@
// @vitest-environment node
-import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
+import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from "vitest";
import { QueryClient, QueryObserver } from "@tanstack/react-query";
import { setApiInstance } from "../api";
@@ -8,21 +8,27 @@ import type {
Issue,
IssueTableRowsRequest,
IssueTableRowsResponse,
+ ListIssuesCache,
ListIssuesParams,
ListIssuesResponse,
} from "../types";
+import { pruneDeletedIssueFromListCaches } from "./delete-cache";
import {
CHILDREN_BY_PARENTS_CHUNK_SIZE,
+ ISSUE_PAGE_SIZE,
PROJECT_GANTT_MAX_ISSUES,
PROJECT_GANTT_PAGE_LIMIT,
childrenByParentsOptions,
childIssuesOptions,
+ flattenIssueBuckets,
issueIdentifierOptions,
issueKeys,
+ issueListOptions,
issueTableRowPageOptions,
projectGanttIssuesOptions,
sourceContextPreviewOptions,
} from "./queries";
+import { onIssueUpdated } from "./ws-updaters";
const WS_ID = "ws-1";
const PROJECT_ID = "project-1";
@@ -315,6 +321,84 @@ describe("issueTableRowPageOptions", () => {
});
});
+describe("issueListOptions", () => {
+ const todoIssue = makeIssue(1, { status: "todo", status_category: "unstarted" });
+ const startedIssue = makeIssue(2, { status: "in_progress", status_category: "started" });
+ const doneIssue = makeIssue(3, { status: "done", status_category: "done" });
+ // A custom status the server did not resolve to a category.
+ const customIssue = makeIssue(4, { status: "qa" });
+
+ let qc: QueryClient;
+ let listIssues: Mock<(params?: ListIssuesParams) => Promise>;
+
+ beforeEach(() => {
+ qc = new QueryClient({ defaultOptions: { queries: { retry: false } } });
+ // Like the real client, the fake ignores `status_category`: every call
+ // returns the same first page.
+ listIssues = vi.fn<(params?: ListIssuesParams) => Promise>().mockResolvedValue({
+ issues: [todoIssue, startedIssue, doneIssue, customIssue],
+ total: 120,
+ });
+ installFakeApi(listIssues);
+ });
+
+ afterEach(() => {
+ qc.clear();
+ });
+
+ async function fetchList() {
+ const options = issueListOptions(WS_ID);
+ await qc.fetchQuery(options);
+ return () => qc.getQueryData(options.queryKey)!;
+ }
+
+ it("sends one list request per fetch", async () => {
+ await fetchList();
+
+ expect(listIssues).toHaveBeenCalledTimes(1);
+ expect(listIssues).toHaveBeenCalledWith({ limit: ISSUE_PAGE_SIZE, offset: 0 });
+ });
+
+ it("puts each issue in its own category's bucket only", async () => {
+ const cache = (await fetchList())();
+
+ expect(cache.byStatus).toEqual({
+ unstarted: { issues: [todoIssue, customIssue], total: 2 },
+ started: { issues: [startedIssue], total: 1 },
+ done: { issues: [doneIssue], total: 1 },
+ closed: { issues: [], total: 0 },
+ });
+ expect(flattenIssueBuckets(cache).map((issue) => issue.id)).toEqual([
+ todoIssue.id,
+ customIssue.id,
+ startedIssue.id,
+ doneIssue.id,
+ ]);
+ });
+
+ it("drops a deleted issue from every bucket", async () => {
+ const readCache = await fetchList();
+
+ pruneDeletedIssueFromListCaches(qc, WS_ID, todoIssue.id);
+
+ expect(flattenIssueBuckets(readCache()).map((issue) => issue.id)).not.toContain(todoIssue.id);
+ });
+
+ it("keeps only the updated copy of an issue moved to done", async () => {
+ const readCache = await fetchList();
+
+ onIssueUpdated(
+ qc,
+ WS_ID,
+ { id: todoIssue.id, status: "done", status_category: "done" },
+ { statusChanged: true },
+ );
+
+ const copies = flattenIssueBuckets(readCache()).filter((issue) => issue.id === todoIssue.id);
+ expect(copies).toEqual([{ ...todoIssue, status: "done", status_category: "done" }]);
+ });
+});
+
describe("projectGanttIssuesOptions", () => {
let qc: QueryClient;
diff --git a/packages/core/issues/queries.ts b/packages/core/issues/queries.ts
index b7c968abd21..af3665a8f27 100644
--- a/packages/core/issues/queries.ts
+++ b/packages/core/issues/queries.ts
@@ -19,6 +19,7 @@ import type {
ListIssuesCache,
} from "../types";
import { ALL_STATUSES } from "./config";
+import { issueColumnCategory } from "./status-category";
export function issueTasksOptions(issueId: string) {
return queryOptions({
@@ -242,7 +243,7 @@ export type AssigneeGroupedIssuesFilter = Omit<
"group_by" | "limit" | "offset" | "group_assignee_type" | "group_assignee_id"
>;
-/** Page size per status column. */
+/** Size of the issue list's first page, all categories together. */
export const ISSUE_PAGE_SIZE = 50;
/**
@@ -267,17 +268,20 @@ export function flattenIssueBuckets(data: ListIssuesCache) {
return out;
}
+/**
+ * One request for the list's first page, grouped into the category buckets
+ * here. Every reader of this cache looks an issue up by id and none pages a
+ * bucket, so one shared window serves them all. Each bucket's `total` is its
+ * own row count. A custom status the server did not resolve falls back to
+ * `issueColumnCategory`'s bucket, so no row is dropped.
+ */
async function fetchFirstPages(filter: MyIssuesFilter = {}, sort?: IssueSortParam): Promise {
- const responses = await Promise.all(
- PAGINATED_CATEGORIES.map((category) =>
- api.listIssues({ status_category: category, limit: ISSUE_PAGE_SIZE, offset: 0, ...sort, ...filter }),
- ),
- );
+ const res = await api.listIssues({ limit: ISSUE_PAGE_SIZE, offset: 0, ...sort, ...filter });
const byStatus: ListIssuesCache["byStatus"] = {};
- PAGINATED_CATEGORIES.forEach((status: IssueStatusCategory, i: number) => {
- const res = responses[i]!;
- byStatus[status] = { issues: res.issues, total: res.total };
- });
+ for (const category of PAGINATED_CATEGORIES) {
+ const issues = res.issues.filter((issue) => issueColumnCategory(issue) === category);
+ byStatus[category] = { issues, total: issues.length };
+ }
return { byStatus };
}
@@ -361,7 +365,7 @@ export function issueTableFacetsOptions(
* `Issue[]` for consumers. Mutations and ws-updaters must use
* `setQueryData(...)` and preserve the byStatus shape.
*
- * Fetches the first page of each paginated status in parallel.
+ * Fetches the first page in one request and buckets it by category.
*/
export function issueListOptions(wsId: string, sort?: IssueSortParam) {
return queryOptions({
diff --git a/packages/core/search-index/engine.test.ts b/packages/core/search-index/engine.test.ts
index d793298e000..e9894708416 100644
--- a/packages/core/search-index/engine.test.ts
+++ b/packages/core/search-index/engine.test.ts
@@ -144,18 +144,27 @@ describe("SearchIndexEngine.searchIssues ranking", () => {
]);
});
- it("puts an exact identifier or bare number first, even without a text match", () => {
- const target = issue({ title: "no text overlap", number: 4242, identifier: "MUL-4242" });
+ it.each(["MUL", "V2", "A1"])("puts a %s identifier or bare number first, even without a text match", (prefix) => {
+ const target = issue({ title: "no text overlap", number: 4242, identifier: `${prefix}-4242` });
const textHit = issue({ title: "mentions 4242 in the title" });
const engine = engineWith([textHit, target]);
- const byIdentifier = engine.searchIssues({ q: "mul-4242" });
+ const byIdentifier = engine.searchIssues({ q: ` ${prefix.toLowerCase()}-4242 ` });
expect(byIdentifier.map((h) => h.id)).toEqual([target.id]);
expect(byIdentifier[0]!.matchSource).toBe("comment");
expect(engine.searchIssues({ q: "4242" }).map((h) => h.id)).toEqual([target.id, textHit.id]);
});
+ it("keeps a cancelled digit-prefix identifier ahead of text hits before limiting", () => {
+ const target = issue({ title: "no text overlap", number: 12, identifier: "V2-12", status: "cancelled" });
+ const textHit = issue({ title: "notes about V2-12" });
+ const engine = engineWith([textHit, target]);
+
+ expect(engine.searchIssues({ q: "V2-12", include_closed: true, limit: 1 }).map((h) => h.id)).toEqual([target.id]);
+ expect(engine.searchIssues({ q: "V2-12", limit: 1 }).map((h) => h.id)).toEqual([textHit.id]);
+ });
+
it("demotes cancelled issues unless the query targets them directly", () => {
const cancelled = issue({ title: "billing export", status: "cancelled" });
const live = issue({ title: "old billing export job", status: "done" });
diff --git a/packages/core/search-index/engine.ts b/packages/core/search-index/engine.ts
index 865a5f39487..5aedf5b143b 100644
--- a/packages/core/search-index/engine.ts
+++ b/packages/core/search-index/engine.ts
@@ -140,10 +140,10 @@ function isTerminal(issue: Pick)
return category === "done" || category === "closed";
}
-/** Server parseQueryNumber: "MUL-123" or a bare "123". */
+/** Server parseQueryNumber: "MUL-123", "V2-12", or a bare "123". */
function parseQueryNumber(query: string): number | null {
const q = query.trim();
- const match = /^[a-z]+-(\d+)$/i.exec(q) ?? /^(\d+)$/.exec(q);
+ const match = /^[a-z][a-z0-9]*-(\d+)$/i.exec(q) ?? /^(\d+)$/.exec(q);
if (!match) return null;
const n = Number.parseInt(match[1]!, 10);
return Number.isInteger(n) && n > 0 ? n : null;
diff --git a/packages/core/search/cancelled-rank.test.ts b/packages/core/search/cancelled-rank.test.ts
index f80d79c64cf..26fde8e5a6b 100644
--- a/packages/core/search/cancelled-rank.test.ts
+++ b/packages/core/search/cancelled-rank.test.ts
@@ -1,3 +1,4 @@
+// @vitest-environment node
import { describe, it, expect } from "vitest";
import type { SearchIssueResult, SearchProjectResult } from "../types/api";
import {
@@ -39,6 +40,20 @@ describe("parseSearchQueryNumber", () => {
expect(parseSearchQueryNumber(" 123 ")).toBe(123);
});
+ it.each([
+ ["V2-12", 12],
+ [" v2-12 ", 12],
+ ["A1-7", 7],
+ ["a1b2-42", 42],
+ ])("reads a digit-containing prefix in %s", (query, number) => {
+ expect(parseSearchQueryNumber(query)).toBe(number);
+ });
+
+ it.each(["12-3", "2FAS-1", "V2-0", "V2--12", "V2-12x", "V_2-12"])(
+ "does not read %s as an issue number",
+ (query) => expect(parseSearchQueryNumber(query)).toBeNull(),
+ );
+
it("returns null for anything that is not a target", () => {
expect(parseSearchQueryNumber("search")).toBeNull();
expect(parseSearchQueryNumber("MUL-")).toBeNull();
diff --git a/packages/core/search/cancelled-rank.ts b/packages/core/search/cancelled-rank.ts
index e5884e57958..4085202be51 100644
--- a/packages/core/search/cancelled-rank.ts
+++ b/packages/core/search/cancelled-rank.ts
@@ -24,9 +24,9 @@ import { issueBehavesAs } from "../issues/status-category";
/**
* Mirrors the server's identifier pattern (parseQueryNumber in
- * server/internal/handler/issue.go): "MUL-123" or a bare "123".
+ * server/internal/handler/issue.go): "MUL-123", "V2-12", or a bare "123".
*/
-const IDENTIFIER_NUMBER_RE = /^[a-z]+-(\d+)$/i;
+const IDENTIFIER_NUMBER_RE = /^[a-z][a-z0-9]*-(\d+)$/i;
/** Extracts the issue number a query targets, or null when it targets none. */
export function parseSearchQueryNumber(query: string): number | null {
diff --git a/packages/core/types/api.ts b/packages/core/types/api.ts
index 8fb02b9f992..50b07951bd7 100644
--- a/packages/core/types/api.ts
+++ b/packages/core/types/api.ts
@@ -148,7 +148,8 @@ export interface ListIssuesParams {
/**
* Filter by lifecycle category rather than by exact key, so one bucket holds
* all concrete and custom statuses in that phase. Task views use exact
- * status keys for their columns instead.
+ * status keys for their columns instead. The web client's
+ * `ApiClient.listIssues` does not send it.
*/
status_category?: IssueStatusCategory;
/** Multi-value form of `status_category`. OR within the field. */
@@ -512,16 +513,16 @@ export interface WorkingAgentSummary {
running_task_count: number;
}
-/** Per-status bucket in the paginated issue cache. `total` is the server count (all pages), not the length of `issues`. */
+/** Per-category bucket in the issue list cache. The list's fetch sets `total` to the bucket's row count. */
export interface IssueStatusBucket {
issues: Issue[];
total: number;
}
/**
- * Frontend cache shape for the issue list. Data is bucketed by status so
- * each column can paginate independently. Assembled from per-status
- * `api.listIssues` responses by the query functions in `issues/queries.ts`.
+ * Frontend cache shape for the issue list. Data is bucketed by status
+ * category. Assembled from one `api.listIssues` response by the query
+ * functions in `issues/queries.ts`.
*/
export interface ListIssuesCache {
/** Bucketed by status CATEGORY — see PAGINATED_CATEGORIES. (MUL-6243) */
diff --git a/packages/core/types/index.ts b/packages/core/types/index.ts
index 9e67ddad4f7..de65f707fc3 100644
--- a/packages/core/types/index.ts
+++ b/packages/core/types/index.ts
@@ -120,8 +120,8 @@ export type { InboxItem, InboxSeverity, InboxItemType, InboxWorkspaceUnread, Arc
export type { NotificationGroupKey, NotificationGroupValue, NotificationPreferences, NotificationPreferenceResponse } from "./notification-preference";
export type { Comment, CommentType, CommentAuthorType, CommentSupplementReceipt, CommentSupplementStatus, CommentTriggerPreview, CommentTriggerPreviewAgent, CommentTriggerSource, CommentTriggerOutcome, CommentTriggerStatus, Reaction } from "./comment";
export type { Label, LabelResourceType, CreateLabelRequest, UpdateLabelRequest, ListLabelsResponse, IssueLabelsResponse, ResourceLabelsResponse } from "./label";
-export type { IssueProperty, IssuePropertyType, ScalarIssuePropertyType, IssuePropertyOption, IssuePropertyConfig, IssuePropertyValue, IssuePropertyValues, CreatePropertyRequest, UpdatePropertyRequest, ListPropertiesResponse, IssuePropertiesResponse, IssuePropertyActorKind, IssuePropertyActorRef, PropertyFilterOp, PropertyOperatorFilter, PropertyFilterValue } from "./property";
-export { ISSUE_PROPERTY_TYPES, isKnownPropertyType, ISSUE_PROPERTY_ACTOR_KINDS, MAX_ISSUE_PROPERTY_ACTOR_VALUES, isActorPropertyType, isFilterablePropertyType, isScalarPropertyType, formatActorRef, parseActorRef, actorRefsFromValue, actorRefValuesFromValue, hasUnknownActorRef, isPropertyOperatorFilter, isKnownPropertyFilterOp, propertyFilterValueKey, PROPERTY_FILTER_OP_SYMBOLS, PROPERTY_FILTER_OPS_BY_TYPE } from "./property";
+export type { IssueProperty, IssuePropertyType, ScalarIssuePropertyType, ListIssuePropertyType, IssuePropertyOption, IssuePropertyConfig, IssuePropertyValue, IssuePropertyValues, CreatePropertyRequest, UpdatePropertyRequest, ListPropertiesResponse, IssuePropertiesResponse, IssuePropertyActorKind, IssuePropertyActorRef, PropertyFilterOp, PropertyOperatorFilter, PropertyFilterValue } from "./property";
+export { ISSUE_PROPERTY_TYPES, isKnownPropertyType, ISSUE_PROPERTY_ACTOR_KINDS, MAX_ISSUE_PROPERTY_ACTOR_VALUES, isActorPropertyType, isFilterablePropertyType, isScalarPropertyType, isListPropertyType, formatActorRef, parseActorRef, actorRefsFromValue, actorRefValuesFromValue, hasUnknownActorRef, isPropertyOperatorFilter, isKnownPropertyFilterOp, propertyFilterValueKey, PROPERTY_FILTER_OP_SYMBOLS, PROPERTY_FILTER_OPS_BY_TYPE } from "./property";
export type {
QuickAction,
QuickActionVisibility,
diff --git a/packages/core/types/property.test.ts b/packages/core/types/property.test.ts
index 0e9789b643c..4c0e27643ed 100644
--- a/packages/core/types/property.test.ts
+++ b/packages/core/types/property.test.ts
@@ -8,6 +8,7 @@ import {
isActorPropertyType,
isFilterablePropertyType,
isKnownPropertyType,
+ isListPropertyType,
isScalarPropertyType,
parseActorRef,
} from "./property";
@@ -48,6 +49,8 @@ describe("isFilterablePropertyType", () => {
"url",
"actor",
"multi_actor",
+ "multi_text",
+ "multi_url",
]) {
expect(isFilterablePropertyType(type)).toBe(true);
}
@@ -69,6 +72,16 @@ describe("isScalarPropertyType", () => {
});
});
+describe("isListPropertyType", () => {
+ it("covers both list types and nothing else", () => {
+ expect(isListPropertyType("multi_text")).toBe(true);
+ expect(isListPropertyType("multi_url")).toBe(true);
+ expect(isListPropertyType("text")).toBe(false);
+ expect(isListPropertyType("url")).toBe(false);
+ expect(isListPropertyType("multi_select")).toBe(false);
+ });
+});
+
describe("parseActorRef", () => {
it("parses member references", () => {
expect(parseActorRef(`member:${MEMBER}`)).toEqual({ kind: "member", id: MEMBER });
diff --git a/packages/core/types/property.ts b/packages/core/types/property.ts
index 7931706b090..22b7e7c4dd9 100644
--- a/packages/core/types/property.ts
+++ b/packages/core/types/property.ts
@@ -7,7 +7,8 @@
* Values are typed per definition: select stores an option id, multi_select
* an array of option ids (config order), date a "YYYY-MM-DD" string, checkbox
* a boolean, number a number, text/url strings, actor a "member:"
- * reference string, multi_actor an array of them (insertion order).
+ * reference string, multi_actor an array of them (insertion order), and
+ * multi_text/multi_url arrays of free-form strings (insertion order).
*/
export type IssuePropertyType =
| "text"
@@ -18,7 +19,9 @@ export type IssuePropertyType =
| "checkbox"
| "url"
| "actor"
- | "multi_actor";
+ | "multi_actor"
+ | "multi_text"
+ | "multi_url";
export const ISSUE_PROPERTY_TYPES: IssuePropertyType[] = [
"text",
@@ -30,6 +33,8 @@ export const ISSUE_PROPERTY_TYPES: IssuePropertyType[] = [
"url",
"actor",
"multi_actor",
+ "multi_text",
+ "multi_url",
];
export function isKnownPropertyType(type: string): type is IssuePropertyType {
@@ -74,7 +79,8 @@ export function isFilterablePropertyType(type: string): boolean {
type === "multi_select" ||
type === "checkbox" ||
isScalarPropertyType(type) ||
- isActorPropertyType(type)
+ isActorPropertyType(type) ||
+ isListPropertyType(type)
);
}
@@ -85,6 +91,16 @@ export function isScalarPropertyType(type: string): type is ScalarIssuePropertyT
return type === "text" || type === "url" || type === "number" || type === "date";
}
+/**
+ * Free-form list properties: multi_text / multi_url. Values are string arrays
+ * in insertion order; filtering matches any single element exactly.
+ */
+export type ListIssuePropertyType = Extract;
+
+export function isListPropertyType(type: string): type is ListIssuePropertyType {
+ return type === "multi_text" || type === "multi_url";
+}
+
export function formatActorRef(kind: IssuePropertyActorKind, id: string): string {
return `${kind}:${id}`;
}
@@ -267,6 +283,8 @@ export const PROPERTY_FILTER_OPS_BY_TYPE: Record {
expect(headings()).not.toContain("Cancelled");
});
- it("keeps a cancelled direct hit visible behind a full window of live candidates", async () => {
+ it.each([
+ ["MUL-99", "MUL-99"],
+ ["V2-99", "v2-99"],
+ ["V2-99", "99"],
+ ])("keeps cancelled %s visible behind a full window when searching %s", async (identifier, query) => {
// 20 non-cancelled cached candidates already fill every slot. Exempting
// the direct hit from the demotion would leave it at position 21 and the
// truncation would still delete it, so it has to be pinned to the front.
searchIssuesMock.mockResolvedValue({
issues: [
- { id: "i-hit", identifier: "MUL-99", title: "Abandoned plan", status: "cancelled" },
+ { id: "i-hit", identifier, title: "Abandoned plan", status: "cancelled" },
],
total: 1,
});
@@ -1322,15 +1326,15 @@ describe("MentionList cancelled demotion", () => {
status: "todo" as const,
}));
- render( );
+ render( );
await waitFor(() => {
- expect(screen.getByText("MUL-99")).toBeInTheDocument();
+ expect(screen.getByText(identifier)).toBeInTheDocument();
});
const labels = rowLabels();
expect(labels).toHaveLength(20);
- expect(labels[0]).toBe("MUL-99");
+ expect(labels[0]).toBe(identifier);
// A live candidate gave up the last slot, not the record the user typed.
expect(labels).not.toContain("MUL-219");
});
diff --git a/packages/views/issues/components/filter-chips-bar.tsx b/packages/views/issues/components/filter-chips-bar.tsx
index 191f6f10f9d..a396efd944d 100644
--- a/packages/views/issues/components/filter-chips-bar.tsx
+++ b/packages/views/issues/components/filter-chips-bar.tsx
@@ -23,7 +23,7 @@ import { projectListOptions } from "@multica/core/projects/queries";
import { PROJECT_STATUS_CONFIG } from "@multica/core/projects/config";
import { labelListOptions } from "@multica/core/labels/queries";
import { propertyListOptions } from "@multica/core/properties";
-import { isActorPropertyType, isScalarPropertyType, parseActorRef, propertyFilterValueKey, PROPERTY_FILTER_OP_SYMBOLS, type PropertyFilterValue } from "@multica/core/types";
+import { isActorPropertyType, isListPropertyType, isScalarPropertyType, parseActorRef, propertyFilterValueKey, PROPERTY_FILTER_OP_SYMBOLS, type PropertyFilterValue } from "@multica/core/types";
import {
type ActorFilterValue,
type FilterDimension,
@@ -531,8 +531,9 @@ function useFilterChips(
? t(($) => $.pickers.custom_property.true_label)
: t(($) => $.pickers.custom_property.false_label);
}
- // Scalar properties have no option list — the filter value IS the label.
- if (isScalarPropertyType(definition.type)) {
+ // Scalar and list properties have no option list — the filter value IS
+ // the label (for lists it is one matched element).
+ if (isScalarPropertyType(definition.type) || isListPropertyType(definition.type)) {
return member;
}
return definition.config.options?.find((o) => o.id === member)?.name;
diff --git a/packages/views/issues/components/issue-chip.tsx b/packages/views/issues/components/issue-chip.tsx
index 0427e71b77a..9b808778b6d 100644
--- a/packages/views/issues/components/issue-chip.tsx
+++ b/packages/views/issues/components/issue-chip.tsx
@@ -62,7 +62,8 @@ export function IssueChip({
const { data: issues = [] } = useQuery(issueListOptions(wsId));
const listIssue = issues.find((i) => i.id === issueId);
- // Fallback fetch for issues outside the first page of the list (e.g. Done).
+ // Fallback fetch for issues outside the list's first page (the first
+ // ISSUE_PAGE_SIZE issues across all categories).
const { data: detailIssue } = useQuery({
...issueDetailOptions(wsId, issueId),
enabled: !listIssue,
diff --git a/packages/views/issues/components/issues-header.tsx b/packages/views/issues/components/issues-header.tsx
index c080ac5cf18..3da45a9ebc6 100644
--- a/packages/views/issues/components/issues-header.tsx
+++ b/packages/views/issues/components/issues-header.tsx
@@ -76,7 +76,7 @@ import type {
ProjectStatus,
WorkingAgentSummary,
} from "@multica/core/types";
-import { formatActorRef, isActorPropertyType, isFilterablePropertyType, isScalarPropertyType, propertyFilterValueKey, PROPERTY_FILTER_OP_SYMBOLS, PROPERTY_FILTER_OPS_BY_TYPE, type PropertyFilterOp, type PropertyFilterValue } from "@multica/core/types";
+import { formatActorRef, isActorPropertyType, isFilterablePropertyType, isListPropertyType, isScalarPropertyType, propertyFilterValueKey, PROPERTY_FILTER_OP_SYMBOLS, PROPERTY_FILTER_OPS_BY_TYPE, type PropertyFilterOp, type PropertyFilterValue } from "@multica/core/types";
import { ProjectIcon } from "../../projects/components/project-icon";
import { useProjectStatusLabels } from "../../projects/components/labels";
import { ActorAvatar } from "../../common/actor-avatar";
@@ -785,10 +785,11 @@ function PropertyFilterOptions({
actorId: undefined as string | undefined,
};
// Scalar value state lives at the top level so the hooks stay unconditional
- // (Rules of Hooks): it is only rendered for text / number / date / url, but
- // must be declared regardless of which branch runs. The draft syncs to the
- // committed scalar member whenever that changes, so a filter cleared or
- // rewritten elsewhere cannot be written back from a stale input.
+ // (Rules of Hooks): it is only rendered for text / number / date / url and
+ // the list types (equality-only input), but must be declared regardless of
+ // which branch runs. The draft syncs to the committed scalar member whenever
+ // that changes, so a filter cleared or rewritten elsewhere cannot be written
+ // back from a stale input.
const committedMember = selected.find((member) => member !== NO_PROPERTY_VALUE);
const committedScalar =
typeof committedMember === "object" ? committedMember.value : (committedMember ?? "");
@@ -833,9 +834,9 @@ function PropertyFilterOptions({
noValueOption,
];
- if (isScalarPropertyType(property.type)) {
+ if (isScalarPropertyType(property.type) || isListPropertyType(property.type)) {
const placeholder =
- property.type === "url"
+ property.type === "url" || property.type === "multi_url"
? t(($) => $.pickers.custom_property.url_placeholder)
: property.type === "number"
? t(($) => $.pickers.custom_property.number_placeholder)
@@ -855,6 +856,12 @@ function PropertyFilterOptions({
if (op === "after") return t(($) => $.pickers.custom_property.op_after);
return PROPERTY_FILTER_OP_SYMBOLS[op] ?? op;
};
+ // List types are equality-only: their PROPERTY_FILTER_OPS_BY_TYPE entry is
+ // empty by design, and looking it up needs the scalar narrowing anyway
+ // (IssueProperty.type is a lenient string).
+ const scalarOps = isScalarPropertyType(property.type)
+ ? (PROPERTY_FILTER_OPS_BY_TYPE[property.type] ?? [])
+ : [];
const opButtons: { op: PropertyFilterOp | "is"; label: string }[] = [
{
op: "is",
@@ -863,7 +870,7 @@ function PropertyFilterOptions({
? "="
: t(($) => $.pickers.custom_property.op_is),
},
- ...(PROPERTY_FILTER_OPS_BY_TYPE[property.type] ?? []).map((op) => ({
+ ...scalarOps.map((op) => ({
op,
label: scalarOperatorLabel(op),
})),
diff --git a/packages/views/issues/components/pickers/custom-property-picker.list.test.tsx b/packages/views/issues/components/pickers/custom-property-picker.list.test.tsx
new file mode 100644
index 00000000000..7c9dbdb5172
--- /dev/null
+++ b/packages/views/issues/components/pickers/custom-property-picker.list.test.tsx
@@ -0,0 +1,179 @@
+import { act, fireEvent, render, screen, waitFor } from "@testing-library/react";
+import userEvent from "@testing-library/user-event";
+import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
+import { beforeEach, describe, expect, it, vi } from "vitest";
+import { api } from "@multica/core/api";
+import { I18nProvider } from "@multica/core/i18n/react";
+import type { Issue, IssueProperty } from "@multica/core/types";
+import enIssues from "../../../locales/en/issues.json";
+import { CustomPropertyValueEditor, CustomPropertyValueInput } from "./custom-property-picker";
+
+vi.mock("@multica/core/api", () => ({
+ api: { setIssueProperty: vi.fn(), unsetIssueProperty: vi.fn() },
+}));
+vi.mock("@multica/core/hooks", () => ({ useWorkspaceId: () => "ws-1" }));
+
+const property: IssueProperty = {
+ id: "p-1", workspace_id: "ws-1", name: "Related links", type: "multi_url",
+ config: {}, position: 1, archived: false,
+ created_at: "2026-01-01T00:00:00Z", updated_at: "2026-01-01T00:00:00Z",
+};
+const issue: Issue = {
+ id: "issue-1", workspace_id: "ws-1", number: 1, identifier: "MUL-1",
+ title: "Related links", description: null, status: "todo", priority: "none",
+ assignee_type: null, assignee_id: null, creator_type: "member", creator_id: "member-1",
+ parent_issue_id: null, project_id: null, position: 1, stage: null,
+ start_date: null, due_date: null, labels: [], metadata: {},
+ properties: { [property.id]: ["https://existing.example"] },
+ created_at: "2026-01-01T00:00:00Z", updated_at: "2026-01-01T00:00:00Z",
+};
+
+function renderEditor() {
+ const client = new QueryClient({ defaultOptions: { mutations: { retry: false } } });
+ const editor = (currentIssue: Issue, currentProperty = property) => (
+
+
+
+
+
+ );
+ const view = render(editor(issue));
+ return {
+ ...view,
+ showIssue: (next: Issue, nextProperty = property) => view.rerender(editor(next, nextProperty)),
+ };
+}
+
+describe("list property editing", () => {
+ beforeEach(() => vi.resetAllMocks());
+
+ it("explains a missing URL scheme and keeps the draft editable", async () => {
+ const user = userEvent.setup();
+ renderEditor();
+ const input = await screen.findByPlaceholderText("https://…");
+ await user.type(input, "example.com{Enter}");
+
+ expect(await screen.findByRole("alert")).toHaveTextContent("http:// or https://");
+ expect(input).toHaveValue("example.com");
+ expect(input).toHaveAttribute("aria-invalid", "true");
+ expect(api.setIssueProperty).not.toHaveBeenCalled();
+
+ await user.clear(input);
+ await user.type(input, "https://example.com");
+ expect(screen.queryByRole("alert")).not.toBeInTheDocument();
+ });
+
+ it("retains a server-rejected URL and clears it only after a successful retry", async () => {
+ const user = userEvent.setup();
+ vi.mocked(api.setIssueProperty).mockRejectedValueOnce(new Error("URL must have a host"));
+ renderEditor();
+ const input = await screen.findByPlaceholderText("https://…");
+ await user.type(input, "https://{Enter}");
+
+ expect(await screen.findByRole("alert")).toHaveTextContent("URL must have a host");
+ expect(input).toHaveValue("https://");
+ expect(api.setIssueProperty).toHaveBeenCalledWith(issue.id, property.id, [
+ "https://existing.example", "https://",
+ ]);
+
+ vi.mocked(api.setIssueProperty).mockResolvedValueOnce({
+ properties: { [property.id]: ["https://existing.example", "https://example.com"] },
+ issue_revision: 2,
+ });
+ await user.type(input, "example.com{Enter}");
+ await waitFor(() => expect(input).toHaveValue(""));
+ expect(screen.queryByRole("alert")).not.toBeInTheDocument();
+ });
+
+ it("keeps the draft but clears the old error when reopened", async () => {
+ const user = userEvent.setup();
+ renderEditor();
+ await user.type(await screen.findByRole("textbox", { name: property.name }), "example.com{Enter}");
+ expect(await screen.findByRole("alert")).toBeInTheDocument();
+
+ await user.keyboard("{Escape}");
+ await user.click(screen.getByRole("button", { name: "https://existing.example" }));
+
+ const input = await screen.findByRole("textbox", { name: property.name });
+ expect(input).toHaveValue("example.com");
+ expect(screen.queryByRole("alert")).not.toBeInTheDocument();
+ expect(input).toHaveAttribute("aria-invalid", "false");
+ expect(input).not.toHaveAttribute("aria-describedby");
+ });
+
+ it.each(["issue", "property"])("resets the draft and error when switching %s in the same editor", async (target) => {
+ const user = userEvent.setup();
+ const { showIssue } = renderEditor();
+ await user.type(await screen.findByRole("textbox", { name: property.name }), "example.com{Enter}");
+ expect(await screen.findByRole("alert")).toBeInTheDocument();
+
+ const nextProperty = target === "property" ? { ...property, id: "p-2" } : property;
+ showIssue({
+ ...issue,
+ id: target === "issue" ? "issue-2" : issue.id,
+ properties: { [nextProperty.id]: ["https://next.example"] },
+ }, nextProperty);
+
+ expect(await screen.findByRole("textbox", { name: property.name })).toHaveValue("");
+ expect(screen.queryByRole("alert")).not.toBeInTheDocument();
+ expect(screen.getByRole("button", { name: "Remove https://next.example" })).toBeEnabled();
+ });
+
+ it("does not carry an in-flight failure into another issue's draft", async () => {
+ const user = userEvent.setup();
+ let fail!: (error: Error) => void;
+ vi.mocked(api.setIssueProperty).mockImplementationOnce(() => new Promise((_resolve, reject) => {
+ fail = reject;
+ }));
+ const { showIssue } = renderEditor();
+ await user.type(await screen.findByRole("textbox", { name: property.name }), "https://old.example{Enter}");
+ await waitFor(() => expect(api.setIssueProperty).toHaveBeenCalledTimes(1));
+
+ showIssue({ ...issue, id: "issue-2" });
+ const input = await screen.findByRole("textbox", { name: property.name });
+ expect(input).toHaveValue("");
+ expect(input).not.toHaveAttribute("readonly");
+ await user.type(input, "https://new.example");
+
+ await act(async () => fail(new Error("Previous issue write failed")));
+ expect(input).toHaveValue("https://new.example");
+ expect(screen.queryByRole("alert")).not.toBeInTheDocument();
+ });
+
+ it("keeps the draft and prevents overlapping add/remove requests while saving", async () => {
+ const user = userEvent.setup();
+ let finish!: () => void;
+ vi.mocked(api.setIssueProperty).mockImplementationOnce(() => new Promise((resolve) => {
+ finish = () => resolve({ properties: {}, issue_revision: 2 });
+ }));
+ renderEditor();
+ const input = await screen.findByPlaceholderText("https://…");
+ await user.type(input, "https://example.com{Enter}");
+ await waitFor(() => expect(api.setIssueProperty).toHaveBeenCalledTimes(1));
+
+ expect(input).toHaveValue("https://example.com");
+ expect(screen.getByRole("button", { name: "Remove https://existing.example" })).toBeDisabled();
+ fireEvent.submit(input.closest("form")!);
+ expect(api.setIssueProperty).toHaveBeenCalledTimes(1);
+
+ await act(async () => finish());
+ await waitFor(() => expect(input).toHaveValue(""));
+ });
+
+ it("still appends and clears a synchronous issue-creation draft", async () => {
+ const user = userEvent.setup();
+ const onChange = vi.fn();
+ render(
+
+
+ ,
+ );
+ const input = await screen.findByPlaceholderText("Enter value…");
+ await user.type(input, "Smith, John{Enter}");
+ expect(onChange).toHaveBeenCalledWith(["alpha", "Smith, John"]);
+ await waitFor(() => expect(input).toHaveValue(""));
+ });
+});
diff --git a/packages/views/issues/components/pickers/custom-property-picker.test.ts b/packages/views/issues/components/pickers/custom-property-picker.test.ts
index c224848b19c..60ac76468f1 100644
--- a/packages/views/issues/components/pickers/custom-property-picker.test.ts
+++ b/packages/views/issues/components/pickers/custom-property-picker.test.ts
@@ -49,4 +49,9 @@ describe("isCustomPropertyReadOnly", () => {
expect(isCustomPropertyReadOnly(multi, [MEMBER, AGENT])).toBe(false);
expect(isCustomPropertyReadOnly(multi, [AGENT])).toBe(false);
});
+
+ it("keeps multi_text and multi_url editable", () => {
+ expect(isCustomPropertyReadOnly(property({ type: "multi_text" }), ["alpha"])).toBe(false);
+ expect(isCustomPropertyReadOnly(property({ type: "multi_url" }), undefined)).toBe(false);
+ });
});
diff --git a/packages/views/issues/components/pickers/custom-property-picker.tsx b/packages/views/issues/components/pickers/custom-property-picker.tsx
index 17a5236912d..792a2b0194a 100644
--- a/packages/views/issues/components/pickers/custom-property-picker.tsx
+++ b/packages/views/issues/components/pickers/custom-property-picker.tsx
@@ -1,10 +1,10 @@
"use client";
-import { useEffect, useState } from "react";
-import { CalendarDays, Check, ExternalLink } from "lucide-react";
+import { useEffect, useId, useRef, useState } from "react";
+import { CalendarDays, Check, ExternalLink, X } from "lucide-react";
import { toast } from "sonner";
import type { Issue, IssueProperty, IssuePropertyValue } from "@multica/core/types";
-import { hasUnknownActorRef } from "@multica/core/types";
+import { hasUnknownActorRef, isListPropertyType } from "@multica/core/types";
import {
useSetIssueProperty,
useUnsetIssueProperty,
@@ -36,6 +36,8 @@ const EDITABLE_PROPERTY_TYPES = [
"url",
"actor",
"multi_actor",
+ "multi_text",
+ "multi_url",
];
/**
@@ -71,6 +73,7 @@ export function isCustomPropertyReadOnly(
* actor → member picker (commits and closes)
* multi_actor → member picker with toggling items (stays open)
* text/number/url → popover with an input, Enter commits
+ * multi_text/multi_url → popover with removable rows + an appending input
*
* Archived definitions render read-only: the popover only offers Clear
* (the server rejects new values on archived properties but always allows
@@ -97,20 +100,28 @@ export function CustomPropertyValueEditor({
return (
{
+ // List editors await the write and show failures beside the draft.
+ if (isListPropertyType(property.type) && !isCustomPropertyReadOnly(property, value)) {
+ const variables = { issueId: issue.id, propertyId: property.id };
+ return (next === undefined
+ ? unsetProperty.mutateAsync(variables)
+ : setProperty.mutateAsync({ ...variables, value: next })
+ ).then(() => {});
+ }
if (next === undefined) {
- unsetProperty.mutate(
+ return unsetProperty.mutate(
{ issueId: issue.id, propertyId: property.id },
{ onError },
);
- return;
}
- setProperty.mutate(
+ return setProperty.mutate(
{ issueId: issue.id, propertyId: property.id, value: next },
{ onError },
);
@@ -122,7 +133,7 @@ export function CustomPropertyValueEditor({
/**
* Mutation-free custom-property editor. Create flows use this while an issue
* still exists only as a draft; issue detail wraps it above with the normal
- * optimistic mutations.
+ * optimistic mutations. List editors await onChange before clearing the draft.
*/
export function CustomPropertyValueInput({
property,
@@ -136,7 +147,7 @@ export function CustomPropertyValueInput({
}: {
property: IssueProperty;
value: IssuePropertyValue | undefined;
- onChange: (value: IssuePropertyValue | undefined) => void;
+ onChange: (value: IssuePropertyValue | undefined) => void | Promise;
defaultOpen?: boolean;
open?: boolean;
onOpenChange?: (open: boolean) => void;
@@ -274,6 +285,20 @@ export function CustomPropertyValueInput({
emptyRow={emptyRow}
/>
);
+ case "multi_text":
+ case "multi_url":
+ return (
+
+ );
case "date": {
const date = typeof value === "string" ? dateOnlyToLocalDate(value) : undefined;
return (
@@ -459,6 +484,153 @@ function TextishPropertyEditor({
);
}
+/**
+ * List editor for multi_text / multi_url: current entries as removable rows
+ * above an input that appends on Enter. The popover stays open across
+ * additions (multi-select interaction); each add/remove commits the whole
+ * array, and removing the last entry clears the property.
+ */
+function ListPropertyEditor({
+ property,
+ value,
+ open,
+ onOpenChange,
+ onCommit,
+ onClear,
+ trigger,
+ triggerRender,
+}: {
+ property: IssueProperty;
+ value: IssuePropertyValue | undefined;
+ open: boolean;
+ onOpenChange: (v: boolean) => void;
+ onCommit: (next: IssuePropertyValue) => void | Promise;
+ onClear: () => void | Promise;
+ trigger?: React.ReactNode;
+ triggerRender?: React.ReactElement>;
+}) {
+ const { t } = useT("issues");
+ const [draft, setDraft] = useState("");
+ const [error, setError] = useState(null);
+ const [saving, setSaving] = useState(false);
+ const savingRef = useRef(false);
+ const errorId = useId();
+
+ useEffect(() => {
+ if (open) setError(null);
+ }, [open]);
+
+ const items = Array.isArray(value) ? value : [];
+ const placeholder =
+ property.type === "multi_url"
+ ? t(($) => $.pickers.custom_property.url_placeholder)
+ : t(($) => $.pickers.custom_property.value_placeholder);
+
+ const save = async (next: string[], clearDraft = false) => {
+ if (savingRef.current) return;
+ savingRef.current = true;
+ setSaving(true);
+ setError(null);
+ try {
+ if (next.length === 0) await onClear();
+ else await onCommit(next);
+ if (clearDraft) setDraft("");
+ } catch (error) {
+ setError(error instanceof Error ? error.message : String(error));
+ } finally {
+ savingRef.current = false;
+ setSaving(false);
+ }
+ };
+
+ const add = () => {
+ if (savingRef.current) return;
+ const trimmed = draft.trim();
+ if (!trimmed) return;
+ if (property.type === "multi_url" && !/^https?:\/\//i.test(trimmed)) {
+ setError(t(($) => $.pickers.custom_property.url_scheme_required));
+ return;
+ }
+ if (items.includes(trimmed)) {
+ setDraft("");
+ setError(null);
+ return;
+ }
+ void save([...items, trimmed], true);
+ };
+
+ return (
+
+
+ {trigger}
+
+
+ {items.length > 0 && (
+
+ {items.map((item) => (
+
+ {item}
+ {property.type === "multi_url" && (
+ $.pickers.custom_property.open_link)}
+ onClick={() => window.open(item, "_blank", "noopener,noreferrer")}
+ >
+
+
+ )}
+ $.pickers.custom_property.remove_item, { value: item })}
+ disabled={saving}
+ onClick={() => void save(items.filter((entry) => entry !== item))}
+ >
+
+
+
+ ))}
+
+ )}
+
+ {error && (
+
+ {error}
+
+ )}
+
+
+ );
+}
+
/**
* Read view of a custom property value, shared by row triggers everywhere
* (sidebar rows now; cards/filters later). Option ids resolve to named,
@@ -535,6 +707,32 @@ export function CustomPropertyValueDisplay({
}
/>
);
+ case "multi_text":
+ case "multi_url": {
+ const items = Array.isArray(value) ? value : [];
+ if (items.length === 0) {
+ return (
+
+ {t(($) => $.pickers.custom_property.empty)}
+
+ );
+ }
+ return (
+
+ {items.map((item) => (
+
+ {property.type === "multi_url" && (
+
+ )}
+ {item}
+
+ ))}
+
+ );
+ }
case "date":
return (
diff --git a/packages/views/issues/components/table-view.tsx b/packages/views/issues/components/table-view.tsx
index 3badfbc2297..fa51d5cd8b1 100644
--- a/packages/views/issues/components/table-view.tsx
+++ b/packages/views/issues/components/table-view.tsx
@@ -921,6 +921,9 @@ function propertyDisplayValue(
.map((option) => option.name)
.join(", ");
}
+ if (property.type === "multi_text" || property.type === "multi_url") {
+ return Array.isArray(value) ? value.join(", ") : String(value);
+ }
if (isActorPropertyType(property.type)) {
return actorRefsFromValue(value)
.map((ref) => (getActorName ? getActorName(ref.kind, ref.id) : formatActorRef(ref.kind, ref.id)))
@@ -1076,7 +1079,7 @@ function IssueTableHeaderCell({
const property = propertyId ? meta.propertyById.get(propertyId) : undefined;
const staticSort = propertyId
? property &&
- !["multi_select", "checkbox", "actor", "multi_actor"].includes(property.type)
+ !["multi_select", "checkbox", "actor", "multi_actor", "multi_text", "multi_url"].includes(property.type)
? (`property:${propertyId}` as SortField)
: undefined
: SORTABLE_COLUMNS[key as TableSystemColumnKey];
diff --git a/packages/views/issues/utils/filter.test.ts b/packages/views/issues/utils/filter.test.ts
index 8b15f2e8607..f32a03344e5 100644
--- a/packages/views/issues/utils/filter.test.ts
+++ b/packages/views/issues/utils/filter.test.ts
@@ -500,6 +500,17 @@ describe("property filters", () => {
expect(result.map((i) => i.id)).toEqual(["P2"]);
});
+ it("multi_text/multi_url match any single element exactly", () => {
+ const aliasId = "prop-aliases";
+ const withAlpha = makeIssue({ id: "L1", properties: { [aliasId]: ["beta", "alpha"] } });
+ const without = makeIssue({ id: "L2", properties: { [aliasId]: ["beta"] } });
+ const result = filterIssues([withAlpha, without], {
+ ...NO_FILTER,
+ propertyFilters: { [aliasId]: ["alpha"] },
+ });
+ expect(result.map((i) => i.id)).toEqual(["L1"]);
+ });
+
it("checkbox values match the true/false pseudo-options", () => {
const result = filterIssues([checked, unset], {
...NO_FILTER,
diff --git a/packages/views/locales/en/issues.json b/packages/views/locales/en/issues.json
index c3fee3c8dfa..b61648c3009 100644
--- a/packages/views/locales/en/issues.json
+++ b/packages/views/locales/en/issues.json
@@ -872,9 +872,11 @@
"false_label": "No",
"value_placeholder": "Enter value…",
"url_placeholder": "https://…",
+ "url_scheme_required": "Start the URL with http:// or https://.",
"number_placeholder": "0",
"actor_search_placeholder": "Search members…",
"open_link": "Open link",
+ "remove_item": "Remove {{value}}",
"archived_hint": "Archived property — value is read-only",
"unknown_value": "Unavailable",
"unknown_value_hint": "This value was set by a newer version of Multica and can't be edited here — update the app, or clear it"
diff --git a/packages/views/locales/en/settings.json b/packages/views/locales/en/settings.json
index 136349db310..4e37ee3db29 100644
--- a/packages/views/locales/en/settings.json
+++ b/packages/views/locales/en/settings.json
@@ -544,7 +544,9 @@
"checkbox": "Checkbox",
"url": "URL",
"actor": "Member",
- "multi_actor": "Members"
+ "multi_actor": "Members",
+ "multi_text": "Text list",
+ "multi_url": "URL list"
},
"actions": {
"open": "Actions for {{name}}",
diff --git a/packages/views/locales/fr/issues.json b/packages/views/locales/fr/issues.json
index 9033f26735d..3528dc16329 100644
--- a/packages/views/locales/fr/issues.json
+++ b/packages/views/locales/fr/issues.json
@@ -872,9 +872,11 @@
"false_label": "Non",
"value_placeholder": "Saisir une valeur…",
"url_placeholder": "https://…",
+ "url_scheme_required": "L’URL doit commencer par http:// ou https://.",
"number_placeholder": "0",
"actor_search_placeholder": "Rechercher des membres…",
"open_link": "Ouvrir le lien",
+ "remove_item": "Retirer {{value}}",
"archived_hint": "Propriété archivée — la valeur est en lecture seule",
"unknown_value": "Indisponible",
"unknown_value_hint": "Cette valeur a été définie par une version plus récente de Multica et ne peut pas être modifiée ici — mettez l'application à jour, ou effacez-la"
diff --git a/packages/views/locales/fr/settings.json b/packages/views/locales/fr/settings.json
index 7659b9c036c..36764c3f27f 100644
--- a/packages/views/locales/fr/settings.json
+++ b/packages/views/locales/fr/settings.json
@@ -544,7 +544,9 @@
"checkbox": "Case à cocher",
"url": "URL",
"actor": "Membre",
- "multi_actor": "Membres"
+ "multi_actor": "Membres",
+ "multi_text": "Liste de textes",
+ "multi_url": "Liste d'URL"
},
"actions": {
"open": "Actions pour {{name}}",
diff --git a/packages/views/locales/ja/issues.json b/packages/views/locales/ja/issues.json
index 54758ab5ee7..b7d546a85f4 100644
--- a/packages/views/locales/ja/issues.json
+++ b/packages/views/locales/ja/issues.json
@@ -833,9 +833,11 @@
"false_label": "いいえ",
"value_placeholder": "値を入力…",
"url_placeholder": "https://…",
+ "url_scheme_required": "URL は http:// または https:// で始めてください。",
"number_placeholder": "0",
"actor_search_placeholder": "メンバーを検索…",
"open_link": "リンクを開く",
+ "remove_item": "{{value}} を削除",
"archived_hint": "アーカイブ済みプロパティ — 値は読み取り専用です",
"unknown_value": "表示できません",
"unknown_value_hint": "この値は新しいバージョンの Multica で設定されており、ここでは編集できません。アプリを更新するか、値をクリアしてください"
diff --git a/packages/views/locales/ja/settings.json b/packages/views/locales/ja/settings.json
index 09f9af4ed48..26d33572c70 100644
--- a/packages/views/locales/ja/settings.json
+++ b/packages/views/locales/ja/settings.json
@@ -540,7 +540,9 @@
"checkbox": "チェックボックス",
"url": "URL",
"actor": "メンバー",
- "multi_actor": "メンバー(複数選択)"
+ "multi_actor": "メンバー(複数選択)",
+ "multi_text": "テキストリスト",
+ "multi_url": "URLリスト"
},
"actions": {
"open": "{{name}} の操作",
diff --git a/packages/views/locales/ko/issues.json b/packages/views/locales/ko/issues.json
index 651e2f75949..b737da945ff 100644
--- a/packages/views/locales/ko/issues.json
+++ b/packages/views/locales/ko/issues.json
@@ -833,9 +833,11 @@
"false_label": "아니요",
"value_placeholder": "값 입력…",
"url_placeholder": "https://…",
+ "url_scheme_required": "URL은 http:// 또는 https://로 시작해야 합니다.",
"number_placeholder": "0",
"actor_search_placeholder": "멤버 검색…",
"open_link": "링크 열기",
+ "remove_item": "{{value}} 제거",
"archived_hint": "보관된 속성 — 값은 읽기 전용입니다",
"unknown_value": "표시할 수 없음",
"unknown_value_hint": "이 값은 최신 버전의 Multica에서 설정되어 여기서는 편집할 수 없습니다. 앱을 업데이트하거나 값을 지우세요"
diff --git a/packages/views/locales/ko/settings.json b/packages/views/locales/ko/settings.json
index c9c848ce7ed..56dbe605291 100644
--- a/packages/views/locales/ko/settings.json
+++ b/packages/views/locales/ko/settings.json
@@ -540,7 +540,9 @@
"checkbox": "체크박스",
"url": "URL",
"actor": "멤버",
- "multi_actor": "멤버(다중 선택)"
+ "multi_actor": "멤버(다중 선택)",
+ "multi_text": "텍스트 목록",
+ "multi_url": "URL 목록"
},
"actions": {
"open": "{{name}} 작업",
diff --git a/packages/views/locales/zh-Hans/issues.json b/packages/views/locales/zh-Hans/issues.json
index cfc309dba76..61877ceb6fa 100644
--- a/packages/views/locales/zh-Hans/issues.json
+++ b/packages/views/locales/zh-Hans/issues.json
@@ -827,9 +827,11 @@
"false_label": "否",
"value_placeholder": "输入值…",
"url_placeholder": "https://…",
+ "url_scheme_required": "URL 必须以 http:// 或 https:// 开头。",
"number_placeholder": "0",
"actor_search_placeholder": "搜索成员…",
"open_link": "打开链接",
+ "remove_item": "移除 {{value}}",
"archived_hint": "已归档属性——值为只读",
"unknown_value": "无法显示",
"unknown_value_hint": "该值由更新版本的 Multica 设置,当前版本无法编辑——请更新应用,或清除该值",
diff --git a/packages/views/locales/zh-Hans/settings.json b/packages/views/locales/zh-Hans/settings.json
index f92c7a0caec..9076af09775 100644
--- a/packages/views/locales/zh-Hans/settings.json
+++ b/packages/views/locales/zh-Hans/settings.json
@@ -540,7 +540,9 @@
"checkbox": "勾选",
"url": "链接",
"actor": "成员",
- "multi_actor": "成员(多选)"
+ "multi_actor": "成员(多选)",
+ "multi_text": "文本列表",
+ "multi_url": "链接列表"
},
"actions": {
"open": "{{name}} 的操作",
diff --git a/packages/views/onboarding/templates/install-runtime-issue.ts b/packages/views/onboarding/templates/install-runtime-issue.ts
index bf10208188c..76ca57d87e5 100644
--- a/packages/views/onboarding/templates/install-runtime-issue.ts
+++ b/packages/views/onboarding/templates/install-runtime-issue.ts
@@ -89,18 +89,21 @@ const zh = `欢迎来到 Multica。
完整文档:https://multica.ai/docs/install-agent-runtime
-中文用户建议先装 Kimi CLI:
+中文用户建议先装 Kimi Code CLI(旧版 Kimi CLI 已停止维护)。
-1. 在 macOS / Linux 终端安装 Kimi CLI:
- curl -LsSf https://code.kimi.com/install.sh | bash
- Windows PowerShell:
- Invoke-RestMethod https://code.kimi.com/install.ps1 | Invoke-Expression
-2. 确认终端能找到 Kimi:
+Windows 用户请先安装 [Git for Windows](https://gitforwindows.org/),Kimi Code CLI 需要其中的 Git Bash。若 Git Bash 安装在非标准路径,请将 KIMI_SHELL_PATH 设为 bash.exe 的绝对路径。
+
+1. 安装 Kimi Code CLI:
+ Linux / macOS:
+ curl -fsSL https://code.kimi.com/kimi-code/install.sh | bash
+ Windows PowerShell:
+ irm https://code.kimi.com/kimi-code/install.ps1 | iex
+2. 重新打开终端,让安装时修改的 PATH 生效,再确认能找到 Kimi:
kimi --version
3. 在你想让 Kimi 工作的项目目录里启动一次:
kimi
-4. 首次启动后输入 /login,按提示完成 Kimi Code 或 API key 配置。
-5. 等 Multica 识别到它。运行中的守护进程每隔几分钟会重新检查一次新装的 CLI,通常不需要重启。
+4. 首次启动后输入 /login,按提示完成 Kimi Code 或 API key 配置。
+5. 等 Multica 识别到它。运行中的守护进程每隔几分钟会重新检查一次新装的 CLI,通常不需要重启。
想立刻生效:
multica daemon restart
桌面端请打开任意一个本机 runtime 并点 Restart。退出再打开 app 是不够的 —— 守护进程会继续在后台运行。
diff --git a/packages/views/settings/components/properties-tab.tsx b/packages/views/settings/components/properties-tab.tsx
index 9d0605a9c0b..247dd0456c0 100644
--- a/packages/views/settings/components/properties-tab.tsx
+++ b/packages/views/settings/components/properties-tab.tsx
@@ -353,6 +353,10 @@ export function PropertyTypeLabel({ type }: { type: string }) {
return <>{t(($) => $.properties.types.actor)}>;
case "multi_actor":
return <>{t(($) => $.properties.types.multi_actor)}>;
+ case "multi_text":
+ return <>{t(($) => $.properties.types.multi_text)}>;
+ case "multi_url":
+ return <>{t(($) => $.properties.types.multi_url)}>;
default:
// Forward compat: newer servers may ship types this build doesn't know.
return <>{type}>;
diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml
index bf8687d1f56..50aa313b780 100644
--- a/pnpm-lock.yaml
+++ b/pnpm-lock.yaml
@@ -651,16 +651,16 @@ importers:
version: 17.4.1
fumadocs-core:
specifier: ^15.5.2
- version: 15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3)
+ version: 15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3)
fumadocs-mdx:
specifier: ^12.0.3
- version: 12.0.3(fumadocs-core@15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3))(next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3)
+ version: 12.0.3(fumadocs-core@15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3))(next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3)
lucide-react:
specifier: 'catalog:'
version: 1.0.1(react@19.2.3)
next:
- specifier: ^16.3.4
- version: 16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3)
+ specifier: ^16.3.8
+ version: 16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3)
react:
specifier: 'catalog:'
version: 19.2.3
@@ -2823,8 +2823,8 @@ packages:
'@next/env@15.5.18':
resolution: {integrity: sha512-hAV85Ckd9QR6RvH04MEKwsfLTksvFpO47j9xwtoIuvuPnlwecpSi+uZTtm8HirVbtlI2Fnz//xpcSTjFdyJk+g==}
- '@next/env@16.3.4':
- resolution: {integrity: sha512-cjWZnUUa6jZq2kFaNe/ZyJdZonOZ/QoN0Zka2nz/FLOrfx14pQuM9c5RaSVkWMqgdt4ksgPAMWPyHSs/CyV48Q==}
+ '@next/env@16.3.8':
+ resolution: {integrity: sha512-Al9zqHVV7TJv0eFuOU4U7Lvv74PTih4Ch63sk2xCIpSTkE3udFnaOcnzP2lQVymiL7yS9Cj2iClUXlR3EQ5sEw==}
'@next/eslint-plugin-next@16.2.3':
resolution: {integrity: sha512-nE/b9mht28XJxjTwKs/yk7w4XTaU3t40UHVAky6cjiijdP/SEy3hGsnQMPxmXPTpC7W4/97okm6fngKnvCqVaA==}
@@ -2835,8 +2835,8 @@ packages:
cpu: [arm64]
os: [darwin]
- '@next/swc-darwin-arm64@16.3.4':
- resolution: {integrity: sha512-iBr3I5LZNk5/bgl5//iTgD2tcym14MX0Xo7fD//u9dYAEgGzza1y9oywluPtf74YnOswVdH1908aK9xVz7zQTw==}
+ '@next/swc-darwin-arm64@16.3.8':
+ resolution: {integrity: sha512-2JPRMh2nmQG5CiL7cXGL9AGwnPWJQ//cTtAUCT+w511QHk79SYz3LGv/pc5X643B/WEO0rvu3Yww0hqwt3kgeA==}
engines: {node: '>= 10'}
cpu: [arm64]
os: [darwin]
@@ -2847,8 +2847,8 @@ packages:
cpu: [x64]
os: [darwin]
- '@next/swc-darwin-x64@16.3.4':
- resolution: {integrity: sha512-2dpiSyl2Jw/NrBPaU2MAKGSa+2MR82pJIn4Sm5Rjr+gxAeuh0z158Su3Z2O8zn7UNNq+ej4bToed6RcRN/Lydg==}
+ '@next/swc-darwin-x64@16.3.8':
+ resolution: {integrity: sha512-GZtCCOBKJ4leVIT/Th0llWKhD1ca92lzbQiS5R5ON9QkoiFnilFsebDae1JU2a3HWoKMEmEZWGs1AGLavVM72Q==}
engines: {node: '>= 10'}
cpu: [x64]
os: [darwin]
@@ -2860,8 +2860,8 @@ packages:
os: [linux]
libc: [glibc]
- '@next/swc-linux-arm64-gnu@16.3.4':
- resolution: {integrity: sha512-+t+U8HZT+fApePCS5h89CSH3datz29MkzyfCn+6fpsZBG/oiEOhINcb9rtkv6sdpToLGFn2e6146NzaKCXkqrA==}
+ '@next/swc-linux-arm64-gnu@16.3.8':
+ resolution: {integrity: sha512-O659ygeQYqneJ1fBKMpFxIFqYkYswu8IAS1OCKK/4f3ZgJJm1dRz4fVJZRi/kLLWjnBKnebOePA4WNv+sV1pVA==}
engines: {node: '>= 10'}
cpu: [arm64]
os: [linux]
@@ -2874,8 +2874,8 @@ packages:
os: [linux]
libc: [musl]
- '@next/swc-linux-arm64-musl@16.3.4':
- resolution: {integrity: sha512-mx03GNs1ocQA5JQ4FxDMmIsNkdrZh8cuezKCrId28e5/gIPU/l7Kcy2+vmCCzdjnnmXJy+iOAu+7K0QppO6Urg==}
+ '@next/swc-linux-arm64-musl@16.3.8':
+ resolution: {integrity: sha512-dSjKSyWpzxoO1d3DIZZcP4XJcNaKeLmxQMFOiYl5vuBRMmweIqnAhty8tAmRsvTss779cK1FtYnDMj40e4TQlg==}
engines: {node: '>= 10'}
cpu: [arm64]
os: [linux]
@@ -2888,8 +2888,8 @@ packages:
os: [linux]
libc: [glibc]
- '@next/swc-linux-x64-gnu@16.3.4':
- resolution: {integrity: sha512-YIhGY6fSMfha52bnVxnzc9zaVBzJg+cqQTOD8tXIBSx4fuv0pVMxQTE0PaS59YhnMOiYiG09IMwxJAf/CFm/Dw==}
+ '@next/swc-linux-x64-gnu@16.3.8':
+ resolution: {integrity: sha512-lbqOuz3RPRcv+o9msNsJw5x4+Y1ZwPTs6vmL6DCf7i0fZfvng/F59wyeDwqHIvV0mK//RBy/jJkZ+nCKsSMXjQ==}
engines: {node: '>= 10'}
cpu: [x64]
os: [linux]
@@ -2902,8 +2902,8 @@ packages:
os: [linux]
libc: [musl]
- '@next/swc-linux-x64-musl@16.3.4':
- resolution: {integrity: sha512-+eaaX6axpDb0yF1GCpiERe6njplvdC+nks/fKfcHu3XPGRrald8P3/X7yv7QLdjA51knnxwl9pxdIJsg+w1L+Q==}
+ '@next/swc-linux-x64-musl@16.3.8':
+ resolution: {integrity: sha512-+316WswI8ScVgZeUd+1KGaXkHhaYQzCjvH/05TZSpJ8zBizb1a4G7DtO7F12jcBIqMOtsz9ji1t48fmKtzqsGA==}
engines: {node: '>= 10'}
cpu: [x64]
os: [linux]
@@ -2915,8 +2915,8 @@ packages:
cpu: [arm64]
os: [win32]
- '@next/swc-win32-arm64-msvc@16.3.4':
- resolution: {integrity: sha512-0jcXW7Xs/uzICrmgV3MhDYDeRy++1CqnpDIerlPIqYO4bhzB4WNbX/aRnQclustsAyTkFKB0z6rbcjmNg5tR8A==}
+ '@next/swc-win32-arm64-msvc@16.3.8':
+ resolution: {integrity: sha512-ji0gd4kMYUxO+1fJBIbiBVRCjzG/lloiyCccnlebvb1ZJ5qXCPZqYg4Jl1DrrixnWNMKylzgpmMWx0yNDYXlzw==}
engines: {node: '>= 10'}
cpu: [arm64]
os: [win32]
@@ -2927,8 +2927,8 @@ packages:
cpu: [x64]
os: [win32]
- '@next/swc-win32-x64-msvc@16.3.4':
- resolution: {integrity: sha512-vvBzwu1pYQCp92maZCFCIw/XgOTMR5tur9GjakwIo2cmwRTMKajRZZDS9+e4KsUZWKu1E007WUeAFXRRjZeuzw==}
+ '@next/swc-win32-x64-msvc@16.3.8':
+ resolution: {integrity: sha512-WcTlaKt/TWkh5kUjdJcUmB1XgZ+1c6fz4Y9fDHL73YNSdGaUWjceeWrrlwF0nv19iABYWC4iAq1oX1w4Bn0vfg==}
engines: {node: '>= 10'}
cpu: [x64]
os: [win32]
@@ -8802,8 +8802,8 @@ packages:
sass:
optional: true
- next@16.3.4:
- resolution: {integrity: sha512-/Ztf6CeRH+ejEXUrYtqI4gkS66eFIHuSwqi60RgcpWKodxFZx2/dqVCMKBwILfAHXQ+F1b1vAudgj3mnxqtoIA==}
+ next@16.3.8:
+ resolution: {integrity: sha512-U7QEZaTini6wKrb8A8hqLLqYQyCetegKjCpJOyxk642vWoMoU1x5PyZCJFvgYgiptA8xc5j/9xYlZFO7w9Sjmw==}
engines: {node: '>=20.9.0'}
hasBin: true
peerDependencies:
@@ -13178,7 +13178,7 @@ snapshots:
'@next/env@15.5.18': {}
- '@next/env@16.3.4': {}
+ '@next/env@16.3.8': {}
'@next/eslint-plugin-next@16.2.3':
dependencies:
@@ -13187,49 +13187,49 @@ snapshots:
'@next/swc-darwin-arm64@15.5.18':
optional: true
- '@next/swc-darwin-arm64@16.3.4':
+ '@next/swc-darwin-arm64@16.3.8':
optional: true
'@next/swc-darwin-x64@15.5.18':
optional: true
- '@next/swc-darwin-x64@16.3.4':
+ '@next/swc-darwin-x64@16.3.8':
optional: true
'@next/swc-linux-arm64-gnu@15.5.18':
optional: true
- '@next/swc-linux-arm64-gnu@16.3.4':
+ '@next/swc-linux-arm64-gnu@16.3.8':
optional: true
'@next/swc-linux-arm64-musl@15.5.18':
optional: true
- '@next/swc-linux-arm64-musl@16.3.4':
+ '@next/swc-linux-arm64-musl@16.3.8':
optional: true
'@next/swc-linux-x64-gnu@15.5.18':
optional: true
- '@next/swc-linux-x64-gnu@16.3.4':
+ '@next/swc-linux-x64-gnu@16.3.8':
optional: true
'@next/swc-linux-x64-musl@15.5.18':
optional: true
- '@next/swc-linux-x64-musl@16.3.4':
+ '@next/swc-linux-x64-musl@16.3.8':
optional: true
'@next/swc-win32-arm64-msvc@15.5.18':
optional: true
- '@next/swc-win32-arm64-msvc@16.3.4':
+ '@next/swc-win32-arm64-msvc@16.3.8':
optional: true
'@next/swc-win32-x64-msvc@15.5.18':
optional: true
- '@next/swc-win32-x64-msvc@16.3.4':
+ '@next/swc-win32-x64-msvc@16.3.8':
optional: true
'@noble/ciphers@1.3.0': {}
@@ -18292,7 +18292,7 @@ snapshots:
transitivePeerDependencies:
- supports-color
- fumadocs-core@15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3):
+ fumadocs-core@15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3):
dependencies:
'@formatjs/intl-localematcher': 0.6.2
'@orama/orama': 3.1.18
@@ -18315,7 +18315,7 @@ snapshots:
optionalDependencies:
'@types/react': 19.2.14
lucide-react: 1.0.1(react@19.2.3)
- next: 16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3)
+ next: 16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3)
react: 19.2.3
react-dom: 19.2.3(react@19.2.3)
react-router: 7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3)
@@ -18348,14 +18348,14 @@ snapshots:
transitivePeerDependencies:
- supports-color
- fumadocs-mdx@12.0.3(fumadocs-core@15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3))(next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3):
+ fumadocs-mdx@12.0.3(fumadocs-core@15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3))(next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3):
dependencies:
'@mdx-js/mdx': 3.1.1
'@standard-schema/spec': 1.1.0
chokidar: 4.0.3
esbuild: 0.25.12
estree-util-value-to-estree: 3.5.0
- fumadocs-core: 15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3)
+ fumadocs-core: 15.8.5(@types/react@19.2.14)(lucide-react@1.0.1(react@19.2.3))(next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react-dom@19.2.3(react@19.2.3))(react-router@7.14.0(react-dom@19.2.3(react@19.2.3))(react@19.2.3))(react@19.2.3)
js-yaml: 4.1.1
lru-cache: 11.2.7
mdast-util-to-markdown: 2.1.2
@@ -18368,7 +18368,7 @@ snapshots:
unist-util-visit: 5.1.0
zod: 4.3.6
optionalDependencies:
- next: 16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3)
+ next: 16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3)
react: 19.2.3
transitivePeerDependencies:
- supports-color
@@ -20453,7 +20453,7 @@ snapshots:
postcss: 8.5.26
react: 19.2.3
react-dom: 19.2.3(react@19.2.3)
- styled-jsx: 5.1.6(react@19.2.3)
+ styled-jsx: 5.1.6(@babel/core@7.29.0)(react@19.2.3)
optionalDependencies:
'@next/swc-darwin-arm64': 15.5.18
'@next/swc-darwin-x64': 15.5.18
@@ -20471,25 +20471,25 @@ snapshots:
- '@babel/core'
- babel-plugin-macros
- next@16.3.4(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3):
+ next@16.3.8(@babel/core@7.29.0)(@opentelemetry/api@1.9.1)(@playwright/test@1.58.2)(@types/node@25.5.0)(babel-plugin-react-compiler@1.0.0)(react-dom@19.2.3(react@19.2.3))(react@19.2.3):
dependencies:
- '@next/env': 16.3.4
+ '@next/env': 16.3.8
'@swc/helpers': 0.5.23
baseline-browser-mapping: 2.10.9
caniuse-lite: 1.0.30001780
postcss: 8.5.26
react: 19.2.3
react-dom: 19.2.3(react@19.2.3)
- styled-jsx: 5.1.6(react@19.2.3)
+ styled-jsx: 5.1.6(@babel/core@7.29.0)(react@19.2.3)
optionalDependencies:
- '@next/swc-darwin-arm64': 16.3.4
- '@next/swc-darwin-x64': 16.3.4
- '@next/swc-linux-arm64-gnu': 16.3.4
- '@next/swc-linux-arm64-musl': 16.3.4
- '@next/swc-linux-x64-gnu': 16.3.4
- '@next/swc-linux-x64-musl': 16.3.4
- '@next/swc-win32-arm64-msvc': 16.3.4
- '@next/swc-win32-x64-msvc': 16.3.4
+ '@next/swc-darwin-arm64': 16.3.8
+ '@next/swc-darwin-x64': 16.3.8
+ '@next/swc-linux-arm64-gnu': 16.3.8
+ '@next/swc-linux-arm64-musl': 16.3.8
+ '@next/swc-linux-x64-gnu': 16.3.8
+ '@next/swc-linux-x64-musl': 16.3.8
+ '@next/swc-win32-arm64-msvc': 16.3.8
+ '@next/swc-win32-x64-msvc': 16.3.8
'@opentelemetry/api': 1.9.1
'@playwright/test': 1.58.2
babel-plugin-react-compiler: 1.0.0
@@ -22542,10 +22542,12 @@ snapshots:
dependencies:
inline-style-parser: 0.2.7
- styled-jsx@5.1.6(react@19.2.3):
+ styled-jsx@5.1.6(@babel/core@7.29.0)(react@19.2.3):
dependencies:
client-only: 0.0.1
react: 19.2.3
+ optionalDependencies:
+ '@babel/core': 7.29.0
styleq@0.1.3: {}
diff --git a/scripts/dev.sh b/scripts/dev.sh
index 40e374bf863..312de856f96 100755
--- a/scripts/dev.sh
+++ b/scripts/dev.sh
@@ -13,7 +13,7 @@ command -v docker >/dev/null 2>&1 || missing+=("docker")
if [ ${#missing[@]} -gt 0 ]; then
echo "✗ Missing prerequisites: ${missing[*]}"
- echo " Please install: Node.js 22, pnpm 10.28.2, Go 1.26.6, Docker"
+ echo " Please install: Node.js 22, pnpm 10.28.2, Go 1.26.9, Docker"
exit 1
fi
diff --git a/server/cmd/migrate/main.go b/server/cmd/migrate/main.go
index 088c1f31dba..b754ad06d28 100644
--- a/server/cmd/migrate/main.go
+++ b/server/cmd/migrate/main.go
@@ -154,6 +154,12 @@ var concurrentIndexCleanups = map[string]string{
"556_wakeup_system_rule_index": "issue_wakeup_system_rule_idx",
"559_issue_child_event_id": "issue_child_event_id_idx",
"560_issue_child_event_pending": "issue_child_event_pending_idx",
+ "566_channel_typing_reaction_id_idx": "channel_typing_reaction_id_idx",
+ "567_channel_typing_reaction_retry_idx": "channel_typing_reaction_retry_idx",
+ "568_channel_typing_reaction_gc_idx": "channel_typing_reaction_gc_idx",
+ "570_channel_typing_quota_idx": "channel_typing_reaction_quota_idx",
+ "571_channel_typing_expiry_idx": "channel_typing_reaction_expiry_idx",
+ "572_channel_typing_abandoned_idx": "channel_typing_reaction_abandoned_idx",
"510_wakeup_id": "issue_wakeup_id_idx",
"511_wakeup_issue": "issue_wakeup_issue_idx",
"512_wakeup_due": "issue_wakeup_due_idx",
diff --git a/server/cmd/migrate/migrate_typing_cleanup_test.go b/server/cmd/migrate/migrate_typing_cleanup_test.go
new file mode 100644
index 00000000000..b7d5e5227a5
--- /dev/null
+++ b/server/cmd/migrate/migrate_typing_cleanup_test.go
@@ -0,0 +1,157 @@
+package main
+
+import (
+ "context"
+ "errors"
+ "fmt"
+ "math/rand/v2"
+ "os"
+ "path/filepath"
+ "testing"
+ "time"
+
+ "github.com/jackc/pgx/v5"
+ "github.com/jackc/pgx/v5/pgconn"
+)
+
+// FIXME(test-integration): Exercises PostgreSQL interrupted concurrent builds and
+// the production migration hooks; requires the isolated test database.
+func TestTypingCleanupRepairsInterruptedIndexes(t *testing.T) {
+ versions := []string{
+ "565_channel_typing_cleanup",
+ "566_channel_typing_reaction_id_idx",
+ "567_channel_typing_reaction_retry_idx",
+ "568_channel_typing_reaction_gc_idx",
+ "569_channel_typing_limits",
+ "570_channel_typing_quota_idx",
+ "571_channel_typing_expiry_idx",
+ "572_channel_typing_abandoned_idx",
+ }
+ files := make([]string, len(versions))
+ for i, version := range versions {
+ files[i] = filepath.Join("..", "..", "migrations", version+".up.sql")
+ }
+ for i, index := range []string{
+ "channel_typing_reaction_id_idx",
+ "channel_typing_reaction_retry_idx",
+ "channel_typing_reaction_gc_idx",
+ "channel_typing_reaction_quota_idx",
+ "channel_typing_reaction_expiry_idx",
+ "channel_typing_reaction_abandoned_idx",
+ } {
+ t.Run(index, func(t *testing.T) {
+ buildPosition := i + 1
+ if i >= 3 {
+ buildPosition++
+ }
+ ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
+ defer cancel()
+ admin := openTestPool(t)
+ schema := fmt.Sprintf("typing_index_%d_%d", time.Now().UnixNano(), rand.Uint32())
+ schemaIdent := pgx.Identifier{schema}.Sanitize()
+ if _, err := admin.Exec(ctx, "CREATE SCHEMA "+schemaIdent); err != nil {
+ t.Fatal(err)
+ }
+ t.Cleanup(func() {
+ cleanupCtx, cleanupCancel := context.WithTimeout(context.Background(), 10*time.Second)
+ defer cleanupCancel()
+ if _, err := admin.Exec(cleanupCtx, "DROP SCHEMA "+schemaIdent+" CASCADE"); err != nil {
+ t.Errorf("drop isolated schema: %v", err)
+ }
+ })
+ // Scope every connection so the real SQL and production hooks are
+ // exercised unchanged, including their unqualified relation names.
+ pool := openTestPoolWithSearchPath(t, schema)
+ if _, err := pool.Exec(ctx, "CREATE TABLE chat_message (id uuid NOT NULL)"); err != nil {
+ t.Fatal(err)
+ }
+ tableSQL, err := os.ReadFile(files[0])
+ if err != nil {
+ t.Fatal(err)
+ }
+ if _, err := pool.Exec(ctx, string(tableSQL)); err != nil {
+ t.Fatal(err)
+ }
+ limitsSQL, err := os.ReadFile(files[4])
+ if err != nil {
+ t.Fatal(err)
+ }
+ if _, err := pool.Exec(ctx, string(limitsSQL)); err != nil {
+ t.Fatal(err)
+ }
+ buildSQL, err := os.ReadFile(files[buildPosition])
+ if err != nil {
+ t.Fatal(err)
+ }
+ // Hold a writer lock so the actual concurrent build is interrupted
+ // after its catalog entry exists. No pg_catalog mutation or mock DDL.
+ blocker, err := pool.Begin(ctx)
+ if err != nil {
+ t.Fatal(err)
+ }
+ defer blocker.Rollback(context.Background())
+ if _, err := blocker.Exec(ctx, "LOCK TABLE channel_typing_reaction IN ROW EXCLUSIVE MODE"); err != nil {
+ t.Fatal(err)
+ }
+ builder, err := pool.Acquire(ctx)
+ if err != nil {
+ t.Fatal(err)
+ }
+ // Close instead of returning a connection with changed session state.
+ conn := builder.Hijack()
+ defer conn.Close(context.Background())
+ if _, err := conn.Exec(ctx, "SET statement_timeout = '2s'"); err != nil {
+ t.Fatal(err)
+ }
+ _, buildErr := conn.Exec(ctx, string(buildSQL))
+ var pgErr *pgconn.PgError
+ if !errors.As(buildErr, &pgErr) || pgErr.Code != "57014" {
+ t.Fatalf("want interrupted build (57014), got %v", buildErr)
+ }
+ if err := blocker.Rollback(ctx); err != nil {
+ t.Fatal(err)
+ }
+ assertIndexValidity(t, pool, schema, index, false)
+ indexOID := func() uint32 {
+ t.Helper()
+ var oid uint32
+ if err := pool.QueryRow(ctx, "SELECT to_regclass($1)::oid", index).Scan(&oid); err != nil {
+ t.Fatal(err)
+ }
+ return oid
+ }
+ invalidOID := indexOID()
+ opts := runOptions{
+ Direction: "up", Files: files,
+ SchemaMigrationsTable: schema + ".schema_migrations",
+ AdvisoryLockKey: int64(rand.Uint64()&0x7fffffffffffffff) | 1,
+ Hooks: preMigrationHooks,
+ }
+ if err := runMigrations(ctx, pool, opts); err != nil {
+ t.Fatal(err)
+ }
+ assertIndexReadyAndValid(t, pool, schema, index, true)
+ rebuiltOID := indexOID()
+ if rebuiltOID == invalidOID {
+ t.Fatal("INVALID index was not dropped and rebuilt")
+ }
+ var applied int
+ if err := pool.QueryRow(ctx, "SELECT count(*) FROM schema_migrations WHERE version = ANY($1)", versions).Scan(&applied); err != nil || applied != len(versions) {
+ t.Fatalf("want all migrations recorded, got %d: %v", applied, err)
+ }
+ // Cover a completed build whose migration stamp was interrupted:
+ // the hook must preserve the valid index on retry.
+ if _, err := pool.Exec(ctx, "DELETE FROM schema_migrations WHERE version = $1", versions[buildPosition]); err != nil {
+ t.Fatal(err)
+ }
+ if err := runMigrations(ctx, pool, opts); err != nil {
+ t.Fatal(err)
+ }
+ if indexOID() != rebuiltOID {
+ t.Fatal("retry replaced an already-valid index")
+ }
+ assertIndexReadyAndValid(t, pool, schema, index, true)
+ t.Log("interrupted INVALID index dropped and rebuilt; all migrations recorded; valid retry preserves index")
+ })
+ }
+}
diff --git a/server/cmd/multica/cmd_issue.go b/server/cmd/multica/cmd_issue.go
index 0b9dfee3b15..5af1550d161 100644
--- a/server/cmd/multica/cmd_issue.go
+++ b/server/cmd/multica/cmd_issue.go
@@ -255,7 +255,13 @@ var issueChildrenCmd = &cobra.Command{
var issueCreateCmd = &cobra.Command{
Use: "create",
Short: "Create a new issue",
- RunE: runIssueCreate,
+ Long: `Create a new issue. Use --property "Name=Value" to set a custom property
+atomically with creation (repeatable, one distinct property per flag).
+For multi_text / multi_url, use a JSON array of strings when entries contain
+commas or the value starts with "["; simple entries can use comma-separated values:
+ multica issue create --title "Review specs" --property 'Aliases=["Smith, John","[draft] spec"]'
+ multica issue create --title "Read docs" --property 'Related links=["https://en.wikipedia.org/wiki/Washington,_D.C."]'`,
+ RunE: runIssueCreate,
}
var issueUpdateCmd = &cobra.Command{
@@ -632,7 +638,7 @@ func init() {
issueCreateCmd.Flags().String("output", "json", "Output format: table or json")
issueCreateCmd.Flags().StringSlice("attachment", nil, "File path(s) to attach (can be specified multiple times). Each file is uploaded and its markdown reference is appended to the description, which is what makes it render on the issue page")
issueCreateCmd.Flags().StringSlice("attachment-id", nil, "Existing attachment UUID(s) to bind to the created issue (can be specified multiple times)")
- issueCreateCmd.Flags().StringArray("property", nil, `Set a custom property atomically with creation as "Name=Value" (repeatable, one distinct property per flag). Multi-value properties use comma-separated values inside one flag. Property and option/member names are case-insensitive; UUIDs are accepted. Filter-only __none__, >=, <=, and != forms are rejected.`)
+ issueCreateCmd.Flags().StringArray("property", nil, `Set a custom property atomically with creation as "Name=Value" (repeatable, one distinct property per flag). Multi-value properties use comma-separated values; multi_text/multi_url also accept JSON arrays (required for commas in entries or values starting with "["). Property and option/member names are case-insensitive; UUIDs are accepted. Filter-only __none__, >=, <=, and != forms are rejected.`)
// issue update
issueUpdateCmd.Flags().String("title", "", "New title")
diff --git a/server/cmd/multica/cmd_issue_property_filter_test.go b/server/cmd/multica/cmd_issue_property_filter_test.go
index fff043972aa..0aacf9fea6f 100644
--- a/server/cmd/multica/cmd_issue_property_filter_test.go
+++ b/server/cmd/multica/cmd_issue_property_filter_test.go
@@ -358,3 +358,47 @@ func TestRunIssueListFetchesCatalogOnce(t *testing.T) {
t.Fatalf("expected one /api/issues request, got %d", len(*queries))
}
}
+
+func TestEncodeIssuePropertyListValues(t *testing.T) {
+ // Both write paths — `issue property set --value` and
+ // `issue create --property` — go through encodeIssuePropertyValue, so this
+ // covers the list-value forms for both commands.
+ multiText := propertyDTO{ID: "p-text", Name: "Aliases", Type: "multi_text"}
+ multiURL := propertyDTO{ID: "p-url", Name: "Related docs", Type: "multi_url"}
+ encode := func(property propertyDTO, raw string) ([]byte, error) {
+ return encodeIssuePropertyValue(t.Context(), nil, &memberDirectory{}, property, raw)
+ }
+
+ cases := []struct {
+ name string
+ property propertyDTO
+ raw string
+ want string
+ wantErr string
+ }{
+ {"comma form keeps simple entries", multiText, "alpha,beta", `["alpha","beta"]`, ""},
+ {"comma form skips empty tokens", multiText, "alpha, ,beta", `["alpha","beta"]`, ""},
+ {"JSON form preserves a comma in text", multiText, `["Smith, John","Doe, Jane"]`, `["Smith, John","Doe, Jane"]`, ""},
+ {"JSON form preserves a comma in a URL", multiURL, `["https://en.wikipedia.org/wiki/Washington,_D.C."]`, `["https://en.wikipedia.org/wiki/Washington,_D.C."]`, ""},
+ {"JSON form mixes both list types", multiURL, `["https://a.example/x?q=1,2","https://b.example"]`, `["https://a.example/x?q=1,2","https://b.example"]`, ""},
+ {"malformed JSON array is rejected", multiText, `["Smith, John", 7]`, "", "JSON array of strings"},
+ {"empty value is rejected", multiText, "", "", "at least one entry"},
+ }
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ got, err := encode(tc.property, tc.raw)
+ if tc.wantErr != "" {
+ if err == nil || !strings.Contains(err.Error(), tc.wantErr) {
+ t.Fatalf("encode(%q) error = %v, want containing %q", tc.raw, err, tc.wantErr)
+ }
+ return
+ }
+ if err != nil {
+ t.Fatalf("encode(%q): %v", tc.raw, err)
+ }
+ if string(got) != tc.want {
+ t.Fatalf("encode(%q) = %s, want %s", tc.raw, got, tc.want)
+ }
+ })
+ }
+}
diff --git a/server/cmd/multica/cmd_property.go b/server/cmd/multica/cmd_property.go
index fb7fb3f8513..55645073195 100644
--- a/server/cmd/multica/cmd_property.go
+++ b/server/cmd/multica/cmd_property.go
@@ -24,7 +24,7 @@ import (
// multica property {list|get|create|update|archive|unarchive} — workspace
// custom property definitions, and multica issue property {list|set|unset} —
// typed values on a single issue. See server/internal/handler/property.go
-// for the validation contract (9 types, 20 active definitions/workspace,
+// for the validation contract (11 types, 20 active definitions/workspace,
// owner/admin-only definition management, agents rejected on definition
// writes).
//
@@ -77,14 +77,23 @@ var propertyCreateCmd = &cobra.Command{
Use: "create",
Short: "Create a property definition (workspace owner/admin only)",
Long: `Create a property definition. Types: text, number, select, multi_select,
-date, checkbox, url, actor, multi_actor. Select types take repeatable --option
-flags:
+date, checkbox, url, actor, multi_actor, multi_text, multi_url. Select types
+take repeatable --option flags:
multica property create --name Severity --type select \
--option "Critical:#ef4444" --option "Major:#f59e0b" --option "Minor:#6b7280"
The ":#rrggbb" color suffix is optional.
The actor types hold workspace members; they take no options:
- multica property create --name Reviewer --type actor`,
+ multica property create --name Reviewer --type actor
+
+multi_text / multi_url hold free-form lists without options:
+ multica property create --name "Related links" --type multi_url
+Set their values with issue property set --value or issue create --property.
+Use comma-separated values for simple entries, or a JSON array of strings.
+JSON is required for entries containing commas or input starting with "[":
+ multica issue property set --name "Related links" \
+ --value '["https://en.wikipedia.org/wiki/Washington,_D.C."]'
+ multica issue property set --name Aliases --value '["[draft] spec"]'`,
Args: exactArgs(0),
RunE: runPropertyCreate,
}
@@ -136,7 +145,14 @@ var issuePropertySetCmd = &cobra.Command{
checkbox --value true|false
number --value 3.5
date --value 2026-07-13
- text / url --value "any string"`,
+ text / url --value "any string"
+ multi_text --value "alpha,beta" (comma-separated strings)
+ multi_url --value "https://a.example,https://b.example"
+For multi_text / multi_url, entries containing commas or input starting with
+"[" must use a JSON array of strings instead:
+ multi_url --value '["https://en.wikipedia.org/wiki/Washington,_D.C."]'
+ multi_text --value '["Smith, John","Doe, Jane"]'
+ multi_text --value '["[draft] spec"]'`,
Args: exactArgs(1),
RunE: runIssuePropertySet,
}
@@ -161,7 +177,7 @@ func init() {
propertyGetCmd.Flags().String("output", "json", "Output format: table or json")
propertyCreateCmd.Flags().String("output", "table", "Output format: table or json")
propertyCreateCmd.Flags().String("name", "", "Property name (required)")
- propertyCreateCmd.Flags().String("type", "", "Property type: text, number, select, multi_select, date, checkbox, url, actor, multi_actor (required)")
+ propertyCreateCmd.Flags().String("type", "", "Property type: text, number, select, multi_select, date, checkbox, url, actor, multi_actor, multi_text, multi_url (required)")
propertyCreateCmd.Flags().String("description", "", "Property description")
propertyCreateCmd.Flags().String("icon", "", "Property icon key from the Web picker (for example, flag, tag, or shield)")
propertyCreateCmd.Flags().StringArray("option", nil, `Select option as "Name" or "Name:#rrggbb" (repeatable; select types only)`)
@@ -574,6 +590,29 @@ func encodeIssuePropertyValue(ctx context.Context, client *cli.APIClient, direct
return nil, fmt.Errorf("--value must list at least one member")
}
return json.Marshal(refs)
+ case "multi_text", "multi_url":
+ // JSON array form first: it round-trips every entry the API accepts,
+ // including ones that contain commas ("Smith, John",
+ // "https://en.wikipedia.org/wiki/Washington,_D.C."), which the comma
+ // form below would split. Comma form stays for simple entries.
+ if strings.HasPrefix(raw, "[") {
+ var items []string
+ if err := json.Unmarshal([]byte(raw), &items); err != nil {
+ return nil, fmt.Errorf("--value must be a JSON array of strings (or a comma-separated list for entries without commas)")
+ }
+ return json.Marshal(items)
+ }
+ parts := strings.Split(raw, ",")
+ items := make([]string, 0, len(parts))
+ for _, part := range parts {
+ if trimmed := strings.TrimSpace(part); trimmed != "" {
+ items = append(items, trimmed)
+ }
+ }
+ if len(items) == 0 {
+ return nil, fmt.Errorf("--value must list at least one entry")
+ }
+ return json.Marshal(items)
case "number":
if _, err := strconv.ParseFloat(raw, 64); err != nil {
return nil, fmt.Errorf("value %q is not a valid number", raw)
@@ -660,12 +699,15 @@ func actorPropertyName(actorNames map[string]string, ref string) string {
return ref
}
-// issuePropertyDisplayValues resolves each item of a multi_select or
-// multi_actor value to its display name. The result stays index-parallel
-// with the stored array (a non-string item renders as JSON rather than being
-// dropped) and is nil for every other type or a non-array value.
+// issuePropertyDisplayValues resolves each item of a multi-value property to
+// its display string: option ids to names (multi_select), actor references to
+// member names (multi_actor), elements as-is (multi_text / multi_url). The
+// result stays index-parallel with the stored array (a non-string item renders
+// as JSON rather than being dropped) and is nil for every other type or a
+// non-array value.
func issuePropertyDisplayValues(property propertyDTO, value any, actorNames map[string]string) []string {
- if property.Type != "multi_select" && property.Type != "multi_actor" {
+ if property.Type != "multi_select" && property.Type != "multi_actor" &&
+ property.Type != "multi_text" && property.Type != "multi_url" {
return nil
}
items, ok := value.([]any)
@@ -680,8 +722,12 @@ func issuePropertyDisplayValues(property propertyDTO, value any, actorNames map[
names = append(names, formatMetadataValue(item))
case property.Type == "multi_select":
names = append(names, propertyOptionName(property, s))
- default:
+ case property.Type == "multi_actor":
names = append(names, actorPropertyName(actorNames, s))
+ default:
+ // multi_text and multi_url are free-form. A string that happens to
+ // equal a member reference is still the text the user stored.
+ names = append(names, s)
}
}
return names
@@ -1106,7 +1152,8 @@ func resolvePropertyFilterValue(ctx context.Context, client *cli.APIClient, dire
return "", fmt.Errorf("--property %s: value %q is not a date in YYYY-MM-DD form", property.Name, trimmed)
}
return trimmed, nil
- case "url":
+ case "url", "multi_url":
+ // One --value token matches one element exactly.
if len(trimmed) > maxPropertyURLValueLen {
return "", fmt.Errorf("--property %s: value must be %d characters or fewer", property.Name, maxPropertyURLValueLen)
}
@@ -1114,7 +1161,7 @@ func resolvePropertyFilterValue(ctx context.Context, client *cli.APIClient, dire
return "", fmt.Errorf("--property %s: value %q is not an http(s) URL", property.Name, trimmed)
}
return trimmed, nil
- case "text":
+ case "text", "multi_text":
if utf8.RuneCountInString(raw) > maxPropertyTextValueLen {
return "", fmt.Errorf("--property %s: value must be %d characters or fewer", property.Name, maxPropertyTextValueLen)
}
diff --git a/server/cmd/multica/cmd_property_test.go b/server/cmd/multica/cmd_property_test.go
index d88a75e3088..74f9671aad0 100644
--- a/server/cmd/multica/cmd_property_test.go
+++ b/server/cmd/multica/cmd_property_test.go
@@ -57,6 +57,34 @@ func TestBuildIssuePropertyRowsMultiValueDisplays(t *testing.T) {
}
}
+// Free-form lists are stored and shown as the caller wrote them. Member-name
+// lookup belongs to actor properties only: a multi_text or multi_url entry
+// that happens to equal a member reference must not be replaced by that
+// member's name.
+func TestIssuePropertyDisplayValuesFreeFormListsIgnoreActorNames(t *testing.T) {
+ ref := "member:abababab-2222-4222-8222-222222222222"
+ actorNames := map[string]string{ref: "Ada"}
+
+ aliases := propertyDTO{ID: "p-aliases", Name: "Aliases", Type: "multi_text"}
+ gotText := issuePropertyDisplayValues(aliases, []any{ref, "plain"}, actorNames)
+ wantText := []string{ref, "plain"}
+ if !reflect.DeepEqual(gotText, wantText) {
+ t.Fatalf("multi_text display_values = %v, want %v", gotText, wantText)
+ }
+
+ links := propertyDTO{ID: "p-links", Name: "Links", Type: "multi_url"}
+ gotURL := issuePropertyDisplayValues(links, []any{ref}, actorNames)
+ if !reflect.DeepEqual(gotURL, []string{ref}) {
+ t.Fatalf("multi_url display_values = %v, want [%s]", gotURL, ref)
+ }
+
+ owners := propertyDTO{ID: testReviewerDefID, Name: "Owners", Type: "multi_actor"}
+ gotActors := issuePropertyDisplayValues(owners, []any{ref}, actorNames)
+ if !reflect.DeepEqual(gotActors, []string{"Ada"}) {
+ t.Fatalf("multi_actor display_values = %v, want [Ada]", gotActors)
+ }
+}
+
func TestFormatIssuePropertyValueMultiValueNotArray(t *testing.T) {
platforms := propertyDTO{ID: testPlatformsDefID, Name: "Platforms", Type: "multi_select"}
// A value that is not an array falls through to the raw rendering and
diff --git a/server/cmd/server/main.go b/server/cmd/server/main.go
index 6f6773cf898..a2432fe75d1 100644
--- a/server/cmd/server/main.go
+++ b/server/cmd/server/main.go
@@ -793,6 +793,10 @@ func main() {
go h.ChannelSupervisor.Run(sweepCtx)
}
+ if h.LarkTyping != nil {
+ go h.LarkTyping.Run(sweepCtx)
+ }
+
// Media intent-ledger reconciler (PR #5580): settles uploaded-but-unbound
// channel media objects. An independent worker so object-storage latency
// spikes cannot starve any other sweeper's cadence.
diff --git a/server/cmd/server/router.go b/server/cmd/server/router.go
index adca0985ddf..42b49d8b163 100644
--- a/server/cmd/server/router.go
+++ b/server/cmd/server/router.go
@@ -615,10 +615,11 @@ func NewRouterWithOptions(pool *pgxpool.Pool, hub *realtime.Hub, bus *events.Bus
patcher.Register(bus)
// Typing indicator: shows a "processing" reaction on the user's
- // message while the agent is working, then removes it before the
- // reply is sent. Best-effort; failures are logged only.
+ // message while the agent is working. Terminal cleanup has its own
+ // budget, with a durable retry worker for failed or missed cleanup.
typingIndicator := lark.NewTypingIndicatorManager(larkClient, installSvc, cs, slog.Default())
patcher.SetTypingIndicatorManager(typingIndicator)
+ h.LarkTyping = typingIndicator
// Inbound pipeline seams: lark_inbound_audit logger and the
// shared channel-agnostic chat-session service. They back the
diff --git a/server/go.mod b/server/go.mod
index cd664420bcf..f138abf5b20 100644
--- a/server/go.mod
+++ b/server/go.mod
@@ -1,6 +1,6 @@
module github.com/multica-ai/multica/server
-go 1.26.6
+go 1.26.9
require (
github.com/aws/aws-sdk-go-v2 v1.47.0
@@ -71,7 +71,7 @@ require (
github.com/tidwall/sjson v1.2.5 // indirect
go.uber.org/atomic v1.11.0 // indirect
golang.org/x/mod v0.41.0 // indirect
- golang.org/x/net v0.59.0 // indirect
+ golang.org/x/net v0.60.0 // indirect
golang.org/x/telemetry v0.0.0-20260910141331-15ceca2b0a1f // indirect
golang.org/x/text v0.42.0 // indirect
golang.org/x/tools v0.50.0 // indirect
diff --git a/server/go.sum b/server/go.sum
index 269da211a22..ed102b84b7e 100644
--- a/server/go.sum
+++ b/server/go.sum
@@ -170,8 +170,8 @@ go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c=
golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o=
-golang.org/x/net v0.59.0 h1:5zfYln+w5XCxwrnMMJPufRgNoXEaGxl0wo5GqPXyues=
-golang.org/x/net v0.59.0/go.mod h1:2DA/G1UfVbCpQPeWTmMPGY7Cs2PkBkwu743bVX5PIVg=
+golang.org/x/net v0.60.0 h1:79p50tfZlm0J9YfoDsSi639qSXNGVwEzOPLCxM2FsYU=
+golang.org/x/net v0.60.0/go.mod h1:2DA/G1UfVbCpQPeWTmMPGY7Cs2PkBkwu743bVX5PIVg=
golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk=
golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0=
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
diff --git a/server/internal/daemon/agents_probe.go b/server/internal/daemon/agents_probe.go
index 0551c32c1c8..928a38f6701 100644
--- a/server/internal/daemon/agents_probe.go
+++ b/server/internal/daemon/agents_probe.go
@@ -142,7 +142,7 @@ var probeAgentCLIs = func() map[string]AgentEntry {
// Codex Desktop bundles its CLI inside the macOS app instead of
// installing it onto PATH.
for _, p := range codexDesktopAppBundlePaths() {
- if _, err := os.Stat(p); err == nil {
+ if executableCandidate(p) {
return AgentEntry{
Path: p,
Command: cmd,
diff --git a/server/internal/daemon/config.go b/server/internal/daemon/config.go
index 5c80e659015..638df6a43fa 100644
--- a/server/internal/daemon/config.go
+++ b/server/internal/daemon/config.go
@@ -978,17 +978,21 @@ var defaultAgentCommandNames = append([]string{
// codexDesktopAppBundlePaths returns candidate macOS app-bundle locations for
// the bundled Codex CLI. OpenAI relocated the Desktop app from Codex.app to
-// ChatGPT.app (#5205). Candidates are ordered by install location first
-// (system /Applications before user ~/Applications); within each location the
-// new ChatGPT.app path is tried before the legacy Codex.app path, so updated
-// installs win while older installs still resolve.
+// ChatGPT.app (#5205), then nested the CLI under Resources/codex-cli/bin
+// (#8941). Candidates are ordered by install location first (system
+// /Applications before user ~/Applications). Within each location the current
+// ChatGPT.app nested path is tried before the flat Resources/codex path, which
+// is tried before the legacy Codex.app path, so the newest install wins while
+// older installs still resolve.
var codexDesktopAppBundlePaths = func() []string {
paths := []string{
+ "/Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex",
"/Applications/ChatGPT.app/Contents/Resources/codex",
"/Applications/Codex.app/Contents/Resources/codex",
}
if home, err := os.UserHomeDir(); err == nil {
paths = append(paths,
+ filepath.Join(home, "Applications", "ChatGPT.app", "Contents", "Resources", "codex-cli", "bin", "codex"),
filepath.Join(home, "Applications", "ChatGPT.app", "Contents", "Resources", "codex"),
filepath.Join(home, "Applications", "Codex.app", "Contents", "Resources", "codex"),
)
diff --git a/server/internal/daemon/config_test.go b/server/internal/daemon/config_test.go
index 98220efc469..5be99274e01 100644
--- a/server/internal/daemon/config_test.go
+++ b/server/internal/daemon/config_test.go
@@ -1435,35 +1435,137 @@ func TestLoadConfig_UsesChatGPTAppBundleCodexPath(t *testing.T) {
}
}
-func TestCodexDesktopAppBundlePaths_IncludesChatGPTAndLegacy(t *testing.T) {
- paths := codexDesktopAppBundlePaths()
- var hasChatGPT, hasLegacy bool
- for _, p := range paths {
- if strings.Contains(p, "ChatGPT.app") && strings.HasSuffix(filepath.ToSlash(p), "Contents/Resources/codex") {
- hasChatGPT = true
+// Regression for #8941: ChatGPT.app 26.924+ ships the CLI at
+// Resources/codex-cli/bin/codex. Discovery must use that path when the older
+// flat Resources/codex file is absent.
+func TestLoadConfig_UsesNestedChatGPTCodexCLIPath(t *testing.T) {
+ pathDir := t.TempDir()
+ nested := filepath.Join(pathDir, "ChatGPT.app", "Contents", "Resources", "codex-cli", "bin", "codex")
+ if err := os.MkdirAll(filepath.Dir(nested), 0o755); err != nil {
+ t.Fatalf("mkdir: %v", err)
+ }
+ if err := os.WriteFile(nested, []byte("#!/bin/sh\nexit 0\n"), 0o755); err != nil {
+ t.Fatalf("write fake CLI: %v", err)
+ }
+
+ oldBundlePaths := codexDesktopAppBundlePaths
+ codexDesktopAppBundlePaths = func() []string { return []string{nested} }
+ t.Cleanup(func() { codexDesktopAppBundlePaths = oldBundlePaths })
+
+ t.Setenv("PATH", t.TempDir())
+ t.Setenv("SHELL", filepath.Join(t.TempDir(), "fish"))
+ t.Setenv("MULTICA_DAEMON_ID", "11111111-1111-1111-1111-111111111111")
+ pinNonCodexAgentsToMissingPaths(t)
+
+ cfg, err := LoadConfig(Overrides{
+ ServerURL: "http://localhost:0",
+ WorkspacesRoot: t.TempDir(),
+ })
+ if err != nil {
+ t.Fatalf("LoadConfig: %v", err)
+ }
+ got, ok := cfg.Agents["codex"]
+ if !ok {
+ t.Fatalf("expected codex agent from nested ChatGPT.app path, got %#v", cfg.Agents)
+ }
+ if got.Path != nested {
+ t.Fatalf("codex path = %q, want nested path %q", got.Path, nested)
+ }
+}
+
+// A bundled CLI that exists but cannot be spawned must stay unregistered:
+// registering it would advertise a healthy runtime whose every task fails.
+func TestProbeAgentCLIsIgnoresNonExecutableCodexBundle(t *testing.T) {
+ if runtime.GOOS == "windows" {
+ t.Skip("the Codex Desktop app bundle fallback is macOS-only")
+ }
+
+ pathDir := t.TempDir()
+ nested := filepath.Join(pathDir, "ChatGPT.app", "Contents", "Resources", "codex-cli", "bin", "codex")
+ if err := os.MkdirAll(filepath.Dir(nested), 0o755); err != nil {
+ t.Fatalf("mkdir: %v", err)
+ }
+ if err := os.WriteFile(nested, []byte("#!/bin/sh\nexit 0\n"), 0o644); err != nil {
+ t.Fatalf("write non-executable fake CLI: %v", err)
+ }
+
+ oldBundlePaths := codexDesktopAppBundlePaths
+ codexDesktopAppBundlePaths = func() []string { return []string{nested} }
+ t.Cleanup(func() { codexDesktopAppBundlePaths = oldBundlePaths })
+
+ t.Setenv("PATH", t.TempDir())
+ t.Setenv("SHELL", filepath.Join(t.TempDir(), "fish"))
+ pinNonCodexAgentsToMissingPaths(t)
+
+ if _, found := probeAgentCLIs()["codex"]; found {
+ t.Fatal("codex was registered from a non-executable app bundle path")
+ }
+}
+
+// When both the nested CLI and the older flat binary exist, the nested path
+// is the current ChatGPT.app layout and must win.
+func TestLoadConfig_PrefersNestedChatGPTCodexCLIPath(t *testing.T) {
+ pathDir := t.TempDir()
+ nested := filepath.Join(pathDir, "ChatGPT.app", "Contents", "Resources", "codex-cli", "bin", "codex")
+ flat := filepath.Join(pathDir, "ChatGPT.app", "Contents", "Resources", "codex")
+ for _, p := range []string{nested, flat} {
+ if err := os.MkdirAll(filepath.Dir(p), 0o755); err != nil {
+ t.Fatalf("mkdir: %v", err)
}
- if strings.Contains(p, "Codex.app") && strings.HasSuffix(filepath.ToSlash(p), "Contents/Resources/codex") {
- hasLegacy = true
+ if err := os.WriteFile(p, []byte("#!/bin/sh\nexit 0\n"), 0o755); err != nil {
+ t.Fatalf("write fake CLI: %v", err)
}
}
- if !hasChatGPT {
- t.Fatalf("codexDesktopAppBundlePaths missing ChatGPT.app entry: %#v", paths)
+
+ oldBundlePaths := codexDesktopAppBundlePaths
+ codexDesktopAppBundlePaths = func() []string { return []string{nested, flat} }
+ t.Cleanup(func() { codexDesktopAppBundlePaths = oldBundlePaths })
+
+ t.Setenv("PATH", t.TempDir())
+ t.Setenv("SHELL", filepath.Join(t.TempDir(), "fish"))
+ t.Setenv("MULTICA_DAEMON_ID", "11111111-1111-1111-1111-111111111111")
+ pinNonCodexAgentsToMissingPaths(t)
+
+ cfg, err := LoadConfig(Overrides{
+ ServerURL: "http://localhost:0",
+ WorkspacesRoot: t.TempDir(),
+ })
+ if err != nil {
+ t.Fatalf("LoadConfig: %v", err)
+ }
+ got, ok := cfg.Agents["codex"]
+ if !ok {
+ t.Fatalf("expected codex agent, got %#v", cfg.Agents)
}
- if !hasLegacy {
- t.Fatalf("codexDesktopAppBundlePaths missing legacy Codex.app entry: %#v", paths)
+ if got.Path != nested {
+ t.Fatalf("codex path = %q, want nested path %q", got.Path, nested)
}
- // New path must be preferred (listed before legacy).
- chatgptIdx, legacyIdx := -1, -1
+}
+
+func TestCodexDesktopAppBundlePaths_IncludesChatGPTAndLegacy(t *testing.T) {
+ paths := codexDesktopAppBundlePaths()
+ const (
+ nestedSuffix = "ChatGPT.app/Contents/Resources/codex-cli/bin/codex"
+ flatSuffix = "ChatGPT.app/Contents/Resources/codex"
+ legacySuffix = "Codex.app/Contents/Resources/codex"
+ )
+ nestedIdx, flatIdx, legacyIdx := -1, -1, -1
for i, p := range paths {
- if chatgptIdx < 0 && strings.Contains(p, "ChatGPT.app") {
- chatgptIdx = i
- }
- if legacyIdx < 0 && strings.Contains(p, "Codex.app") {
+ slash := filepath.ToSlash(p)
+ switch {
+ case nestedIdx < 0 && strings.HasSuffix(slash, nestedSuffix):
+ nestedIdx = i
+ case flatIdx < 0 && strings.HasSuffix(slash, flatSuffix):
+ flatIdx = i
+ case legacyIdx < 0 && strings.HasSuffix(slash, legacySuffix):
legacyIdx = i
}
}
- if chatgptIdx < 0 || legacyIdx < 0 || chatgptIdx > legacyIdx {
- t.Fatalf("expected ChatGPT.app before Codex.app, got indices chat=%d legacy=%d paths=%#v", chatgptIdx, legacyIdx, paths)
+ if nestedIdx < 0 || flatIdx < 0 || legacyIdx < 0 {
+ t.Fatalf("codexDesktopAppBundlePaths missing nested, flat, or legacy entry: %#v", paths)
+ }
+ if nestedIdx > flatIdx || flatIdx > legacyIdx {
+ t.Fatalf("expected nested ChatGPT path before flat ChatGPT path before Codex.app, got nested=%d flat=%d legacy=%d paths=%#v", nestedIdx, flatIdx, legacyIdx, paths)
}
}
diff --git a/server/internal/daemon/daemon.go b/server/internal/daemon/daemon.go
index 85745dab5cd..be0e3537555 100644
--- a/server/internal/daemon/daemon.go
+++ b/server/internal/daemon/daemon.go
@@ -6543,18 +6543,12 @@ func providerNeedsInlineSystemPrompt(provider string) bool {
// changes, so binding it to workdir reuse discards healthy conversation history
// and forces the model to reconstruct it through `multica chat history`.
//
-// A matching workdir is not sufficient on its own. Hermes keys its sessions to
-// HERMES_HOME — the per-task overlay under envRoot — not to the cwd, and the
-// two keys come apart precisely in the local_directory flow: reuse is disabled
-// there (shouldReusePriorWorkdir), so every task builds a fresh overlay with an
-// empty state.db, while envWorkDir stays the user's own directory and therefore
-// still equals PriorWorkDir. The gate read "reused" and forwarded a session id
-// that could not possibly resolve, and Hermes answers an unresolvable resume by
-// silently starting over (GH #6806). sessionHomeReachable is the provider's own
-// answer to "can a prior session still be found here?" — for Hermes, whether
-// the conversation's session store got mounted (execenv.Environment
-// HermesSessionStore) — and false drops the resume with the same disclosure as
-// a workdir mismatch.
+// Hermes is also independent of cwd: its transcript lives in HERMES_HOME's
+// state.db. sessionHomeReachable checks whether the conversation-scoped store
+// mounted by execenv holds history, or whether the task-local home was reused.
+// Requiring the prior cwd as well would discard reachable history whenever a
+// local_directory task gets a new worktree (#9062). An empty or unavailable
+// store must still drop the resume, even when the cwd matches (#6806).
// sameExistingDir reports whether two paths name the same existing directory.
// False when either cannot be stat'd, which is the safe answer for cwd-keyed
// providers: an absent prior workdir means there is nothing to resume from.
@@ -6577,6 +6571,8 @@ func gateResumeToReachableSession(task *Task, taskCtx *execenv.TaskContextForEnv
var reachable bool
if providerUsesPiSessionFile(provider) {
reachable = piSessionResumable(task.PriorSessionID, refusesMissingSessionCwd)
+ } else if provider == "hermes" {
+ reachable = sessionHomeReachable
} else {
// Compare the directories, not the spelling. Reuse runs in the canonical
// path it validated and locked, which need not be character-identical to
diff --git a/server/internal/daemon/daemon_test.go b/server/internal/daemon/daemon_test.go
index 3bda712758e..d2c550f7faa 100644
--- a/server/internal/daemon/daemon_test.go
+++ b/server/internal/daemon/daemon_test.go
@@ -2021,6 +2021,106 @@ func TestGateResumeToReachableSession(t *testing.T) {
}
}
+// Hermes resumes from HERMES_HOME, independently of the task's cwd (#9062).
+func TestGateHermesResumeToSessionHome(t *testing.T) {
+ t.Parallel()
+ for _, tt := range []struct {
+ name string
+ sameWorkdir bool
+ history bool
+ envReused bool
+ noStore bool
+ want bool
+ }{
+ {name: "changed worktree with history", history: true, want: true},
+ {name: "in place with history", sameWorkdir: true, history: true, want: true},
+ {name: "changed worktree with unavailable history"},
+ {name: "in place with unavailable history", sameWorkdir: true},
+ {name: "empty store despite reused environment", sameWorkdir: true, envReused: true},
+ {name: "unmounted store in fresh environment", noStore: true},
+ {name: "task local history in reused environment", sameWorkdir: true, noStore: true, envReused: true, want: true},
+ } {
+ t.Run(tt.name, func(t *testing.T) {
+ priorDir, workDir := t.TempDir(), t.TempDir()
+ if tt.sameWorkdir {
+ workDir = priorDir
+ }
+ env := &execenv.Environment{HermesSessionStore: "conversation-store", HermesSessionHistoryPresent: tt.history}
+ if tt.noStore {
+ env.HermesSessionStore = ""
+ }
+ task := Task{PriorSessionID: "session-1", PriorWorkDir: priorDir}
+ taskCtx := execenv.TaskContextForEnv{PriorSessionResumed: true}
+ got := gateResumeToReachableSession(&task, &taskCtx, "hermes", workDir,
+ sessionHomeReachable("hermes", env, tt.envReused), false, slog.Default())
+ if got != tt.want || (task.PriorSessionID == "session-1") != tt.want || taskCtx.PriorSessionResumed != tt.want {
+ t.Fatalf("resume = %v, session = %q, resumed = %v; want %v", got, task.PriorSessionID, taskCtx.PriorSessionResumed, tt.want)
+ }
+ if task.PriorSessionResumeUnavailable != !tt.want || taskCtx.PriorSessionResumeUnavailable != !tt.want {
+ t.Fatal("session continuity notice does not match reachability")
+ }
+ })
+ }
+}
+
+func TestHermesPreparedSessionReachability(t *testing.T) {
+ // Keep profile/store resolution inside synthetic homes, never the user's.
+ t.Setenv("MULTICA_TASK_CONFIG_ROOT", "")
+ home := t.TempDir()
+ t.Setenv("HOME", home)
+ t.Setenv("USERPROFILE", home)
+ sourceHome, root := t.TempDir(), t.TempDir()
+ taskNumber := 0
+ prepare := func(t *testing.T, agentID, issueID string) *execenv.Environment {
+ t.Helper()
+ taskCtx := execenv.TaskContextForEnv{AgentID: agentID, IssueID: issueID,
+ AgentSkills: []execenv.SkillContextForEnv{{Name: "fixture", Content: "synthetic skill"}},
+ }
+ taskNumber++
+ env, err := execenv.Prepare(execenv.PrepareParams{
+ WorkspacesRoot: root, WorkspaceID: "workspace-1", TaskID: fmt.Sprintf("task-%012d", taskNumber),
+ Provider: "hermes", HermesSourceHome: sourceHome, Task: taskCtx,
+ HermesSessionStore: execenv.HermesSessionStorePath("", agentID, sourceHome, taskCtx),
+ }, slog.Default())
+ if err != nil {
+ t.Fatal(err)
+ }
+ t.Cleanup(func() { _ = env.Cleanup(true) })
+ return env
+ }
+ first := prepare(t, "agent-1", "issue-1")
+ if first.HermesSessionStore == "" {
+ if runtime.GOOS == "windows" {
+ t.Skip("host cannot mount Hermes session stores")
+ }
+ t.Fatal("Hermes session store was not mounted")
+ }
+ if err := os.WriteFile(filepath.Join(first.HermesHome, "state.db"), []byte("synthetic transcript"), 0o600); err != nil {
+ t.Fatal(err)
+ }
+ for _, tt := range []struct {
+ name, agentID, issueID string
+ want bool
+ }{
+ {"same conversation in fresh workdir", "agent-1", "issue-1", true},
+ {"different issue", "agent-1", "issue-2", false},
+ {"different agent", "agent-2", "issue-1", false},
+ } {
+ t.Run(tt.name, func(t *testing.T) {
+ env := prepare(t, tt.agentID, tt.issueID)
+ if sameExistingDir(first.WorkDir, env.WorkDir) {
+ t.Fatal("fixture must prepare a different task workdir")
+ }
+ task := Task{PriorSessionID: "session-1", PriorWorkDir: first.WorkDir}
+ taskCtx := execenv.TaskContextForEnv{PriorSessionResumed: true}
+ if got := gateResumeToReachableSession(&task, &taskCtx, "hermes", env.WorkDir,
+ sessionHomeReachable("hermes", env, false), false, slog.Default()); got != tt.want {
+ t.Fatalf("reachable = %v, want %v", got, tt.want)
+ }
+ })
+ }
+}
+
func TestGatePiResumeToSessionFile(t *testing.T) {
t.Parallel()
diff --git a/server/internal/daemon/execenv/hermes_sessions.go b/server/internal/daemon/execenv/hermes_sessions.go
index 140ccac15f2..98f55be6bbf 100644
--- a/server/internal/daemon/execenv/hermes_sessions.go
+++ b/server/internal/daemon/execenv/hermes_sessions.go
@@ -210,10 +210,27 @@ func mountHermesSessionDB(hermesHome, storeDir string, logger *slog.Logger) (her
const hermesSessionLinkStagingEntry = ".multica-session-link"
// hermesStoreHasSessionDB reports whether storeDir holds a session database
-// with content. A zero-length file is what SQLite leaves after an `open` that
-// never wrote a page, and resuming against it is the same amnesia as an absent
-// one — so it counts as no history, not as history.
+// with readable content. A zero-length file is what SQLite leaves after an
+// `open` that never wrote a page; resuming against it is the same as an absent
+// database, so it counts as no history.
func hermesStoreHasSessionDB(storeDir string) bool {
+ path := filepath.Join(storeDir, hermesSessionDBEntry)
+ fi, err := os.Stat(path)
+ if err != nil || !fi.Mode().IsRegular() || fi.Size() == 0 {
+ return false
+ }
+ db, err := os.Open(path)
+ if err != nil {
+ return false
+ }
+ defer db.Close()
+ return true
+}
+
+// hermesStoreContainsSessionDB protects existing history during migration,
+// even when it cannot currently be opened for resume. Read permissions do not
+// prevent unlinking a database from a writable directory.
+func hermesStoreContainsSessionDB(storeDir string) bool {
fi, err := os.Stat(filepath.Join(storeDir, hermesSessionDBEntry))
return err == nil && fi.Mode().IsRegular() && fi.Size() > 0
}
@@ -234,7 +251,7 @@ func hermesStoreHasSessionDB(storeDir string) bool {
// Only an empty store is migrated into — a store that already holds a database
// is this conversation's real history and must never be overwritten.
func migrateHermesTaskSessionDB(hermesHome, storeDir string, logger *slog.Logger) error {
- if hermesStoreHasSessionDB(storeDir) {
+ if hermesStoreContainsSessionDB(storeDir) {
return nil // store already holds this conversation — never overwrite it
}
@@ -306,7 +323,7 @@ func migrateHermesTaskSessionDB(hermesHome, storeDir string, logger *slog.Logger
func publishHermesSessionStaging(staging, storeDir string) (bool, error) {
hermesSessionPublishMu.Lock()
defer hermesSessionPublishMu.Unlock()
- if hermesStoreHasSessionDB(storeDir) {
+ if hermesStoreContainsSessionDB(storeDir) {
return false, nil // a competitor published a real transcript first
}
if hermesSessionPublishBarrier != nil {
@@ -315,7 +332,7 @@ func publishHermesSessionStaging(staging, storeDir string) (bool, error) {
if err := removeHermesSessionDBFamily(storeDir); err != nil {
return false, err
}
- return promoteHermesStoreStaging(staging, storeDir, hermesStoreHasSessionDB)
+ return promoteHermesStoreStaging(staging, storeDir, hermesStoreContainsSessionDB)
}
// hermesSessionPublishMu serializes every session-store publish in this
diff --git a/server/internal/daemon/execenv/hermes_sessions_test.go b/server/internal/daemon/execenv/hermes_sessions_test.go
index 2996ededbd5..e2404a2ea0c 100644
--- a/server/internal/daemon/execenv/hermes_sessions_test.go
+++ b/server/internal/daemon/execenv/hermes_sessions_test.go
@@ -229,6 +229,53 @@ func TestPrepareHermesHomeMigrationNeverOverwritesStore(t *testing.T) {
}
}
+func TestPrepareHermesHomeMigrationPreservesUnreadableStore(t *testing.T) {
+ if runtime.GOOS == "windows" {
+ t.Skip("Unix permissions required")
+ }
+ sharedHome := t.TempDir()
+ store := t.TempDir()
+ hermesHome := filepath.Join(t.TempDir(), "hermes-home")
+ skills := []SkillContextForEnv{{Name: "deploy", Content: "# Deploy"}}
+ db := filepath.Join(store, "state.db")
+ mustWrite(t, db, "the real transcript")
+ mustWrite(t, filepath.Join(store, "state.db-wal"), "the real uncheckpointed history")
+ if err := os.Chmod(db, 0); err != nil {
+ t.Fatal(err)
+ }
+ t.Cleanup(func() { _ = os.Chmod(db, 0o600) })
+ if f, err := os.Open(db); err == nil {
+ f.Close()
+ t.Skip("current user can read files despite permissions")
+ }
+ staging := t.TempDir()
+ mustWrite(t, filepath.Join(staging, "state.db"), "competing transcript")
+ mustWrite(t, filepath.Join(staging, "state.db-wal"), "competing WAL")
+ if published, err := publishHermesSessionStaging(staging, store); err != nil || published {
+ t.Fatalf("publish over unreadable history = %v, %v; want false, nil", published, err)
+ }
+ if _, err := prepareHermesHome(hermesHome, sharedHome, false, skills, nil, "", "", testLogger()); err != nil {
+ t.Fatal(err)
+ }
+ mustWrite(t, filepath.Join(hermesHome, "state.db"), "stray task-local database")
+ mustWrite(t, filepath.Join(hermesHome, "state.db-wal"), "stray task-local WAL")
+ if _, err := prepareHermesHome(hermesHome, sharedHome, false, skills, nil, "", store, testLogger()); err != nil {
+ t.Fatal(err)
+ }
+ if hermesStoreHasSessionDB(store) {
+ t.Error("unreadable history must not be resumable")
+ }
+ if err := os.Chmod(db, 0o600); err != nil {
+ t.Fatal(err)
+ }
+ for name, want := range map[string]string{"state.db": "the real transcript", "state.db-wal": "the real uncheckpointed history"} {
+ got, err := os.ReadFile(filepath.Join(store, name))
+ if err != nil || string(got) != want {
+ t.Errorf("persistent %s = %q, %v; want %q", name, got, err, want)
+ }
+ }
+}
+
// TestPrepareHermesHomeSessionMountIsIdempotent covers Reuse: rebuilding the
// overlay for a follow-up turn in the same task directory must leave the link
// (and therefore the live database) alone.
@@ -843,3 +890,23 @@ func requireSymlinks(t *testing.T) {
t.Skipf("symlinks unavailable on this host: %v", err)
}
}
+
+func TestHermesStoreHasSessionDBUnreadable(t *testing.T) {
+ if runtime.GOOS == "windows" {
+ t.Skip("Unix permissions required")
+ }
+ store := t.TempDir()
+ db := filepath.Join(store, "state.db")
+ mustWrite(t, db, "transcript")
+ if err := os.Chmod(db, 0); err != nil {
+ t.Fatal(err)
+ }
+ t.Cleanup(func() { _ = os.Chmod(db, 0o600) })
+ if f, err := os.Open(db); err == nil {
+ f.Close()
+ t.Skip("current user can read files despite permissions")
+ }
+ if hermesStoreHasSessionDB(store) {
+ t.Fatal("unreadable session database reported as reachable")
+ }
+}
diff --git a/server/internal/events/bus.go b/server/internal/events/bus.go
index 3636cadbc4c..95e035afb99 100644
--- a/server/internal/events/bus.go
+++ b/server/internal/events/bus.go
@@ -5,6 +5,14 @@ import (
"sync"
)
+// ChannelReactionTarget retains the non-secret cleanup anchor when a transaction
+// deletes the delivery row. It is internal bus metadata, never a client payload.
+type ChannelReactionTarget struct {
+ ChannelType string
+ InstallationID string
+ MessageID string
+}
+
// Event represents a domain event published by handlers or services.
type Event struct {
Type string // e.g. "issue:created", "inbox:new"
@@ -17,8 +25,9 @@ type Event struct {
// event to a more specific scope than `workspace:{WorkspaceID}`. When set
// these tell the listener which Redis stream / Hub room to publish on
// without re-deserializing Payload. See MUL-1138 phase 1.
- TaskID string
- ChatSessionID string
+ TaskID string
+ ChatSessionID string
+ ChannelReactionTarget *ChannelReactionTarget `json:"-"`
}
// Handler is a function that processes an event.
diff --git a/server/internal/handler/agent.go b/server/internal/handler/agent.go
index fc6f50c9db2..d082d6ec709 100644
--- a/server/internal/handler/agent.go
+++ b/server/internal/handler/agent.go
@@ -1599,7 +1599,7 @@ func (h *Handler) CreateAgent(w http.ResponseWriter, r *http.Request) {
return
}
slog.Warn("create agent failed", append(logger.RequestAttrs(r), "error", err, "workspace_id", workspaceID)...)
- writeError(w, http.StatusInternalServerError, "failed to create agent: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to create agent")
return
}
if err := replaceInvocationTargetsWithQueries(r.Context(), qtx, created.ID, parseUUID(ownerID), perm.targets); err != nil {
@@ -2250,7 +2250,7 @@ func (h *Handler) UpdateAgent(w http.ResponseWriter, r *http.Request) {
return
}
slog.Warn("update agent failed", append(logger.RequestAttrs(r), "error", err, "agent_id", id)...)
- writeError(w, http.StatusInternalServerError, "failed to update agent: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to update agent")
return
}
@@ -2261,7 +2261,7 @@ func (h *Handler) UpdateAgent(w http.ResponseWriter, r *http.Request) {
updated, err = h.Queries.ClearAgentMcpConfig(r.Context(), updated.ID)
if err != nil {
slog.Warn("clear agent mcp_config failed", append(logger.RequestAttrs(r), "error", err, "agent_id", id)...)
- writeError(w, http.StatusInternalServerError, "failed to clear mcp_config: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to clear mcp_config")
return
}
}
@@ -2269,7 +2269,7 @@ func (h *Handler) UpdateAgent(w http.ResponseWriter, r *http.Request) {
updated, err = h.Queries.ClearAgentThinkingLevel(r.Context(), updated.ID)
if err != nil {
slog.Warn("clear agent thinking_level failed", append(logger.RequestAttrs(r), "error", err, "agent_id", id)...)
- writeError(w, http.StatusInternalServerError, "failed to clear thinking_level: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to clear thinking_level")
return
}
}
@@ -2277,7 +2277,7 @@ func (h *Handler) UpdateAgent(w http.ResponseWriter, r *http.Request) {
updated, err = h.Queries.ClearAgentServiceTier(r.Context(), updated.ID)
if err != nil {
slog.Warn("clear agent service_tier failed", append(logger.RequestAttrs(r), "error", err, "agent_id", id)...)
- writeError(w, http.StatusInternalServerError, "failed to clear service_tier: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to clear service_tier")
return
}
}
@@ -2285,7 +2285,7 @@ func (h *Handler) UpdateAgent(w http.ResponseWriter, r *http.Request) {
updated, err = h.Queries.ClearAgentComposioToolkitAllowlist(r.Context(), updated.ID)
if err != nil {
slog.Warn("clear agent composio_toolkit_allowlist failed", append(logger.RequestAttrs(r), "error", err, "agent_id", id)...)
- writeError(w, http.StatusInternalServerError, "failed to clear composio_toolkit_allowlist: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to clear composio_toolkit_allowlist")
return
}
}
@@ -2296,7 +2296,7 @@ func (h *Handler) UpdateAgent(w http.ResponseWriter, r *http.Request) {
if replacePermissionTargets {
if err := h.replaceInvocationTargets(r.Context(), updated.ID, parseUUID(requestUserID(r)), resolvedPerm.targets); err != nil {
slog.Warn("update agent: persist invocation targets failed", append(logger.RequestAttrs(r), "error", err, "agent_id", id)...)
- writeError(w, http.StatusInternalServerError, "failed to update invocation targets: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to update invocation targets")
return
}
}
diff --git a/server/internal/handler/chat.go b/server/internal/handler/chat.go
index bf195405312..25cfb221696 100644
--- a/server/internal/handler/chat.go
+++ b/server/internal/handler/chat.go
@@ -14,6 +14,8 @@ import (
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgtype"
"github.com/multica-ai/multica/server/internal/analytics"
+ "github.com/multica-ai/multica/server/internal/events"
+ "github.com/multica-ai/multica/server/internal/logger"
obsmetrics "github.com/multica-ai/multica/server/internal/metrics"
"github.com/multica-ai/multica/server/internal/middleware"
"github.com/multica-ai/multica/server/internal/service"
@@ -584,6 +586,7 @@ func (h *Handler) SetChatSessionArchived(w http.ResponseWriter, r *http.Request)
// request unarchives and nil for a web-only chat; BroadcastCancelledTasks
// is a no-op on an empty slice.
var cancelled []db.AgentTaskQueue
+ var reactionTargets map[string]*events.ChannelReactionTarget
if req.Archived {
// Read the binding BEFORE the delete below, which is what erases the
@@ -636,6 +639,11 @@ func (h *Handler) SetChatSessionArchived(w http.ResponseWriter, r *http.Request)
writeError(w, http.StatusInternalServerError, "failed to read chat session channel binding")
return
}
+ reactionTargets, err = service.CaptureChannelReactionTargets(r.Context(), qtx, cancelled)
+ if err != nil {
+ writeError(w, http.StatusInternalServerError, "failed to capture channel reaction targets")
+ return
+ }
if err := qtx.DeleteChannelChatSessionBindingBySession(r.Context(), session.ID); err != nil {
writeError(w, http.StatusInternalServerError, "failed to clear chat session channel binding")
return
@@ -655,7 +663,7 @@ func (h *Handler) SetChatSessionArchived(w http.ResponseWriter, r *http.Request)
// clients drop the row instead of showing it queued until the next
// refresh, and wakes the runtime so a queued successor is claimed now
// rather than at the daemon's next poll.
- h.TaskService.BroadcastCancelledTasks(r.Context(), workspaceID, cancelled)
+ h.TaskService.BroadcastCancelledTasks(r.Context(), workspaceID, cancelled, reactionTargets)
resolvedSessionID := uuidToString(updated.ID)
status := updated.Status
@@ -727,6 +735,12 @@ func (h *Handler) DeleteChatSession(w http.ResponseWriter, r *http.Request) {
return
}
+ reactionTargets, err := service.CaptureChannelReactionTargets(r.Context(), qtx, cancelled)
+ if err != nil {
+ writeError(w, http.StatusInternalServerError, "failed to capture channel reaction targets")
+ return
+ }
+
// channel_chat_session_binding used to carry a chat_session FK with
// ON DELETE CASCADE; MUL-3515 §4 dropped every channel_* foreign key, so
// prune the binding here in the same tx that deletes its chat_session.
@@ -787,7 +801,7 @@ func (h *Handler) DeleteChatSession(w http.ResponseWriter, r *http.Request) {
// The workspace has to come from the session we just deleted: the tasks were
// cancelled and returned before the delete, so they still carry its id, but
// the row they would be resolved through is gone by now.
- h.TaskService.BroadcastCancelledTasks(r.Context(), workspaceID, cancelled)
+ h.TaskService.BroadcastCancelledTasks(r.Context(), workspaceID, cancelled, reactionTargets)
resolvedSessionID := uuidToString(session.ID)
h.publishChat(protocol.EventChatSessionDeleted, workspaceID, "member", userID, resolvedSessionID, protocol.ChatSessionDeletedPayload{
@@ -952,7 +966,8 @@ func (h *Handler) SendChatMessage(w http.ResponseWriter, r *http.Request) {
case errors.Is(err, service.ErrChatTaskAgentNoRuntime):
writeError(w, http.StatusConflict, "chat agent has no runtime")
default:
- writeError(w, http.StatusInternalServerError, "failed to send chat message: "+err.Error())
+ slog.Warn("send chat message failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to send chat message")
}
return
}
diff --git a/server/internal/handler/chat_reaction_target_test.go b/server/internal/handler/chat_reaction_target_test.go
new file mode 100644
index 00000000000..02246d81cad
--- /dev/null
+++ b/server/internal/handler/chat_reaction_target_test.go
@@ -0,0 +1,84 @@
+package handler
+
+import (
+ "context"
+ "encoding/json"
+ "errors"
+ "net/http"
+ "net/http/httptest"
+ "strings"
+ "testing"
+
+ "github.com/jackc/pgx/v5"
+ "github.com/multica-ai/multica/server/internal/events"
+ "github.com/multica-ai/multica/server/internal/testutil"
+ "github.com/multica-ai/multica/server/pkg/protocol"
+)
+
+func TestChatSessionRemovalCarriesReactionTargetsAfterCommit(t *testing.T) {
+ if testHandler == nil {
+ t.Skip("database not available")
+ }
+ for _, action := range []string{"delete", "archive"} {
+ t.Run(action, func(t *testing.T) {
+ f := newArchiveCancelFixture(t, "reaction targets "+action)
+ installationID := dbfx.Insert(t, "channel_installation", testutil.Cols{
+ "workspace_id": testWorkspaceID, "agent_id": f.agentID,
+ "channel_type": "feishu", "installer_user_id": testUserID,
+ })
+ bindingID := dbfx.Insert(t, "channel_chat_session_binding", testutil.Cols{
+ "chat_session_id": f.sessionID, "installation_id": installationID,
+ "channel_type": "feishu", "channel_chat_id": "reaction-room", "chat_type": "group",
+ })
+ expected := map[string]string{f.queuedTaskID: "queued-trigger", f.runningTaskID: "running-trigger"}
+ for taskID, messageID := range expected {
+ dbfx.InsertNoID(t, "channel_task_delivery", testutil.Cols{
+ "task_id": taskID, "binding_id": bindingID, "installation_id": installationID,
+ "channel_type": "feishu", "channel_chat_id": "reaction-room", "chat_type": "group",
+ "channel_message_id": messageID, "route_revision": 1,
+ }, "task_id = $1", taskID)
+ }
+ seen := make(chan events.Event, 2)
+ testHandler.Bus.Subscribe(protocol.EventTaskCancelled, func(e events.Event) {
+ if e.ChatSessionID != f.sessionID {
+ return
+ }
+ // Observe through the pool, not the transaction: the event must be post-commit.
+ _, err := testHandler.Queries.GetChannelTaskDelivery(context.Background(), parseUUID(e.TaskID))
+ if !errors.Is(err, pgx.ErrNoRows) {
+ t.Errorf("cancel published before delivery deletion committed: %v", err)
+ }
+ seen <- e
+ })
+ if action == "archive" {
+ f.archive(t)
+ } else {
+ req := httptest.NewRequest(http.MethodDelete, "/api/chat/sessions/"+f.sessionID, nil)
+ req.Header.Set("X-User-ID", testUserID)
+ req = withChatTestWorkspaceCtx(t, withURLParam(req, "sessionId", f.sessionID))
+ w := httptest.NewRecorder()
+ testHandler.DeleteChatSession(w, req)
+ if w.Code != http.StatusNoContent {
+ t.Fatalf("delete: %d %s", w.Code, w.Body.String())
+ }
+ }
+ if len(seen) != len(expected) {
+ t.Fatalf("got %d cancellation events, want %d", len(seen), len(expected))
+ }
+ for range expected {
+ e := <-seen
+ target := e.ChannelReactionTarget
+ if target == nil || target.ChannelType != "feishu" || target.InstallationID != installationID || target.MessageID != expected[e.TaskID] {
+ t.Fatalf("task %s lost its frozen reaction target: %+v", e.TaskID, target)
+ }
+ payload, err := json.Marshal(e)
+ if err != nil {
+ t.Fatal(err)
+ }
+ if strings.Contains(string(payload), target.MessageID) || strings.Contains(string(payload), installationID) {
+ t.Fatalf("internal cleanup metadata leaked into serialized event: %s", payload)
+ }
+ }
+ })
+ }
+}
diff --git a/server/internal/handler/comment.go b/server/internal/handler/comment.go
index 6cc243400a5..583279bc315 100644
--- a/server/internal/handler/comment.go
+++ b/server/internal/handler/comment.go
@@ -1922,7 +1922,7 @@ func (h *Handler) CreateComment(w http.ResponseWriter, r *http.Request) {
tx, beginErr := h.beginWakeupWrite(r.Context())
if beginErr != nil {
slog.Warn("create comment failed", append(logger.RequestAttrs(r), "error", beginErr, "issue_id", issueID)...)
- writeError(w, http.StatusInternalServerError, "failed to create comment: "+beginErr.Error())
+ writeError(w, http.StatusInternalServerError, "failed to create comment")
return
}
defer tx.Rollback(r.Context())
@@ -1958,7 +1958,7 @@ func (h *Handler) CreateComment(w http.ResponseWriter, r *http.Request) {
}
if err != nil {
slog.Warn("create comment failed", append(logger.RequestAttrs(r), "error", err, "issue_id", issueID)...)
- writeError(w, http.StatusInternalServerError, "failed to create comment: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to create comment")
return
}
comment := created.Comment()
diff --git a/server/internal/handler/daemon.go b/server/internal/handler/daemon.go
index 8945dbdb31b..a031c9bf554 100644
--- a/server/internal/handler/daemon.go
+++ b/server/internal/handler/daemon.go
@@ -25,6 +25,7 @@ import (
"github.com/multica-ai/multica/server/internal/daemonws"
"github.com/multica-ai/multica/server/internal/integrations/slack"
"github.com/multica-ai/multica/server/internal/issuestatus"
+ "github.com/multica-ai/multica/server/internal/logger"
obsmetrics "github.com/multica-ai/multica/server/internal/metrics"
"github.com/multica-ai/multica/server/internal/middleware"
"github.com/multica-ai/multica/server/internal/runtimeapps"
@@ -539,7 +540,8 @@ func (h *Handler) DaemonRegister(w http.ResponseWriter, r *http.Request) {
"db_error",
true,
))
- writeError(w, http.StatusInternalServerError, "failed to register runtime: "+err.Error())
+ slog.Warn("register runtime failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to register runtime")
return
}
provider = agent.ProfileRuntimeType(profile.RuntimeType, profile.ProtocolFamily)
@@ -585,7 +587,8 @@ func (h *Handler) DaemonRegister(w http.ResponseWriter, r *http.Request) {
"db_error",
true,
))
- writeError(w, http.StatusInternalServerError, "failed to register runtime: "+err.Error())
+ slog.Warn("register runtime failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to register runtime")
return
}
inserted = row.Inserted
@@ -1825,7 +1828,8 @@ func (h *Handler) ClaimTasksByRuntime(w http.ResponseWriter, r *http.Request) {
claimed, err := h.TaskService.ClaimTasksForRuntimes(r.Context(), authorized, maxTasks)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to claim tasks: "+err.Error())
+ slog.Warn("claim tasks failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to claim tasks")
return
}
@@ -3744,7 +3748,8 @@ func (h *Handler) ClaimTaskByRuntime(w http.ResponseWriter, r *http.Request) {
claimMs = time.Since(claimStart).Milliseconds()
if err != nil {
outcome = "error_claim"
- writeError(w, http.StatusInternalServerError, "failed to claim task: "+err.Error())
+ slog.Warn("claim task failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to claim task")
return
}
@@ -4333,7 +4338,7 @@ func (h *Handler) CompleteTask(w http.ResponseWriter, r *http.Request) {
// 5xx so the daemon retries the terminal callback and the completion —
// including the single chat outcome row — lands exactly once (MUL-4351).
slog.Warn("complete task failed", "task_id", taskID, "error", err)
- writeError(w, http.StatusInternalServerError, err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to complete task")
return
}
if !transitioned {
@@ -5039,7 +5044,7 @@ func (h *Handler) failTask(w http.ResponseWriter, r *http.Request, taskID, works
// isTransientError) — retries and the fail, gap flag, and retry land
// exactly once (MUL-5305). An invalid request body still returns 400 above.
slog.Warn("fail task failed", "task_id", taskID, "error", err)
- writeError(w, http.StatusInternalServerError, err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to fail task")
return
}
if !transitioned {
diff --git a/server/internal/handler/handler.go b/server/internal/handler/handler.go
index 25b478822f1..bfe72f9bf78 100644
--- a/server/internal/handler/handler.go
+++ b/server/internal/handler/handler.go
@@ -318,6 +318,8 @@ type Handler struct {
// where the storage backend exists; main.go starts it as an independent
// worker goroutine. Nil when no storage backend is configured.
ChannelMediaReconciler *service.ChannelMediaReconciler
+ // LarkTyping owns the durable reaction cleanup worker, started under sweepCtx.
+ LarkTyping *lark.TypingIndicatorManager
// SlackInstall owns the bring-your-own-app Slack install lifecycle (register
// pasted tokens / list / revoke) and the at-rest encryption of each app's bot
// + app tokens (MUL-3666). Nil unless MULTICA_SLACK_SECRET_KEY is set.
diff --git a/server/internal/handler/internal_error_boundary_test.go b/server/internal/handler/internal_error_boundary_test.go
new file mode 100644
index 00000000000..08e5795d8e4
--- /dev/null
+++ b/server/internal/handler/internal_error_boundary_test.go
@@ -0,0 +1,67 @@
+package handler
+
+import (
+ "go/ast"
+ "go/parser"
+ "go/token"
+ "os"
+ "path/filepath"
+ "strings"
+ "testing"
+)
+
+// TestInternalServerErrorsDoNotExposeCause keeps raw infrastructure errors on
+// the server side. It walks complete call expressions, so multiline payloads
+// and error variables with names other than err are covered too.
+func TestInternalServerErrorsDoNotExposeCause(t *testing.T) {
+ t.Parallel()
+
+ entries, err := os.ReadDir(".")
+ if err != nil {
+ t.Fatalf("read handler package: %v", err)
+ }
+ fset := token.NewFileSet()
+ for _, entry := range entries {
+ name := entry.Name()
+ if entry.IsDir() || !strings.HasSuffix(name, ".go") || strings.HasSuffix(name, "_test.go") {
+ continue
+ }
+ file, err := parser.ParseFile(fset, filepath.Clean(name), nil, 0)
+ if err != nil {
+ t.Fatalf("parse %s: %v", name, err)
+ }
+ ast.Inspect(file, func(node ast.Node) bool {
+ call, ok := node.(*ast.CallExpr)
+ if !ok || !callUsesInternalServerError(call) {
+ return true
+ }
+ ast.Inspect(call, func(child ast.Node) bool {
+ causeCall, ok := child.(*ast.CallExpr)
+ if !ok {
+ return true
+ }
+ selector, ok := causeCall.Fun.(*ast.SelectorExpr)
+ if ok && selector.Sel.Name == "Error" && len(causeCall.Args) == 0 {
+ pos := fset.Position(causeCall.Pos())
+ t.Errorf("%s:%d returns a raw error from an internal-server-error path; send stable public copy and log the cause", name, pos.Line)
+ }
+ return true
+ })
+ return false
+ })
+ }
+}
+
+func callUsesInternalServerError(call *ast.CallExpr) bool {
+ for _, arg := range call.Args {
+ selector, ok := arg.(*ast.SelectorExpr)
+ if !ok || selector.Sel.Name != "StatusInternalServerError" {
+ continue
+ }
+ pkg, ok := selector.X.(*ast.Ident)
+ if ok && pkg.Name == "http" {
+ return true
+ }
+ }
+ return false
+}
diff --git a/server/internal/handler/issue.go b/server/internal/handler/issue.go
index 05cae676388..09da855eb9a 100644
--- a/server/internal/handler/issue.go
+++ b/server/internal/handler/issue.go
@@ -774,11 +774,11 @@ func splitSearchTerms(q string) []string {
return terms
}
-// identifierNumberRe matches patterns like "MUL-123" or "ABC-45".
-var identifierNumberRe = regexp.MustCompile(`(?i)^[a-z]+-(\d+)$`)
+// identifierNumberRe matches letter-leading prefixes like "MUL-123" or "V2-12".
+var identifierNumberRe = regexp.MustCompile(`(?i)^[a-z][a-z0-9]*-(\d+)$`)
// parseQueryNumber extracts an issue number from the query if it looks like
-// an identifier (e.g. "MUL-123") or a bare number (e.g. "123").
+// an identifier (e.g. "MUL-123" or "V2-12") or a bare number (e.g. "123").
func parseQueryNumber(q string) (int, bool) {
q = strings.TrimSpace(q)
// Check for identifier pattern like "MUL-123"
@@ -3478,7 +3478,7 @@ func (h *Handler) CreateIssue(w http.ResponseWriter, r *http.Request) {
}
if err != nil {
slog.Warn("create issue failed", append(logger.RequestAttrs(r), "error", err, "workspace_id", workspaceID)...)
- writeError(w, http.StatusInternalServerError, "failed to create issue: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to create issue")
return
}
@@ -3994,7 +3994,7 @@ func (h *Handler) UpdateIssue(w http.ResponseWriter, r *http.Request) {
}
}
slog.Warn("update issue failed", append(logger.RequestAttrs(r), "error", err, "issue_id", id, "workspace_id", workspaceID)...)
- writeError(w, http.StatusInternalServerError, "failed to update issue: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to update issue")
return
}
diff --git a/server/internal/handler/issue_table_facets.go b/server/internal/handler/issue_table_facets.go
index 58b4a1f47f2..0b0481dc31b 100644
--- a/server/internal/handler/issue_table_facets.go
+++ b/server/internal/handler/issue_table_facets.go
@@ -265,13 +265,15 @@ GROUP BY r.agent_id`, len(compiled.args)+1, compiled.where)
// Unset issues count under the "__none__" bucket so the filter menu's
// "No value" option carries a real count (issues without the key).
query = fmt.Sprintf(`SELECT COALESCE(i.properties ->> %s, '__none__'), COUNT(*)::bigint FROM issue i WHERE %s AND (jsonb_typeof(i.properties -> %s) = 'string' OR NOT (i.properties ? %s)) GROUP BY 1`, propertyKey, compiled.where, propertyKey, propertyKey)
- case "text", "url", "date", "number":
- // Scalar values are free-form, so grouping by distinct value would
- // return one row per observed value with no bound. The filter menu
- // only reads the "__none__" bucket for these types, so compute just
- // the two bounded buckets instead of pulling the whole value space.
- // Key existence is exact for "no value": stored values can never be
- // null or empty, so a present key is always a set value.
+ case "text", "url", "date", "number", "multi_text", "multi_url":
+ // Scalar and free-form-list values are unbounded distinct-value
+ // spaces (multi_text/multi_url enumerate arbitrary user strings),
+ // so grouping by distinct value would return one row per observed
+ // value with no bound. The filter menu only reads the "__none__"
+ // bucket for these types, so compute just the two bounded buckets
+ // instead of pulling the whole value space. Key existence is exact
+ // for "no value": stored values can never be null or empty, so a
+ // present key is always a set value.
query = fmt.Sprintf(`SELECT CASE WHEN i.properties ? %s THEN '__set__' ELSE '__none__' END, COUNT(*)::bigint FROM issue i WHERE %s GROUP BY 1`, propertyKey, compiled.where)
case "multi_select", "multi_actor":
query = fmt.Sprintf(`SELECT COALESCE(property_value.value, '__none__'), COUNT(DISTINCT i.id)::bigint FROM issue i JOIN LATERAL (SELECT jsonb_array_elements_text(CASE WHEN jsonb_typeof(i.properties -> %s) = 'array' THEN i.properties -> %s ELSE '[]'::jsonb END) AS value UNION ALL SELECT NULL WHERE NOT (i.properties ? %s)) property_value(value) ON TRUE WHERE %s GROUP BY 1`, propertyKey, propertyKey, propertyKey, compiled.where)
diff --git a/server/internal/handler/mika_onboarding.go b/server/internal/handler/mika_onboarding.go
index 3f77ca05e2e..9e1f6f16c95 100644
--- a/server/internal/handler/mika_onboarding.go
+++ b/server/internal/handler/mika_onboarding.go
@@ -4,10 +4,12 @@ import (
"encoding/json"
"errors"
"fmt"
+ "log/slog"
"net/http"
"strings"
"github.com/go-chi/chi/v5"
+ "github.com/multica-ai/multica/server/internal/logger"
"github.com/multica-ai/multica/server/internal/service"
"github.com/multica-ai/multica/server/pkg/protocol"
)
@@ -170,7 +172,8 @@ func (h *Handler) StartMikaOnboarding(w http.ResponseWriter, r *http.Request) {
return
}
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to start Mika onboarding: "+err.Error())
+ slog.Warn("start Mika onboarding failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to start Mika onboarding")
return
}
diff --git a/server/internal/handler/property.go b/server/internal/handler/property.go
index bbd506fcdcd..d237899397c 100644
--- a/server/internal/handler/property.go
+++ b/server/internal/handler/property.go
@@ -52,7 +52,7 @@ const (
maxPropertyActorValues = 20
)
-var validPropertyTypes = []string{"text", "number", "select", "multi_select", "date", "checkbox", "url", "actor", "multi_actor"}
+var validPropertyTypes = []string{"text", "number", "select", "multi_select", "date", "checkbox", "url", "actor", "multi_actor", "multi_text", "multi_url"}
// Property icons use stable catalog keys that the Web client maps to Lucide
// glyphs. Keeping this allowlist at the API boundary prevents arbitrary text
diff --git a/server/internal/handler/property_test.go b/server/internal/handler/property_test.go
index a73d7cd9ae9..137b2383cca 100644
--- a/server/internal/handler/property_test.go
+++ b/server/internal/handler/property_test.go
@@ -15,6 +15,7 @@ import (
"github.com/go-chi/chi/v5"
"github.com/google/uuid"
+ "github.com/multica-ai/multica/server/internal/issueproperty"
"github.com/multica-ai/multica/server/internal/testutil"
db "github.com/multica-ai/multica/server/pkg/db/generated"
)
@@ -313,6 +314,30 @@ func TestIssuePropertyValues(t *testing.T) {
t.Fatalf("good number: expected 200, got %d: %s", w.Code, w.Body.String())
}
+ // multi_text: duplicates dropped, caller order kept.
+ listText := createTestProperty(t, map[string]any{"name": "Aliases" + uuid.NewString()[:8], "type": "multi_text"})
+ wl := setIssuePropertyRaw(t, issueID, listText.ID, []string{"beta", "alpha", "beta"})
+ if wl.Code != http.StatusOK {
+ t.Fatalf("multi_text set: expected 200, got %d: %s", wl.Code, wl.Body.String())
+ }
+ var listResp struct {
+ Properties map[string]any `json:"properties"`
+ }
+ json.NewDecoder(wl.Body).Decode(&listResp)
+ storedList, _ := listResp.Properties[listText.ID].([]any)
+ if len(storedList) != 2 || storedList[0] != "beta" || storedList[1] != "alpha" {
+ t.Fatalf("multi_text not stored in caller order: %v", storedList)
+ }
+
+ // multi_url: entry-level http(s) validation.
+ listURL := createTestProperty(t, map[string]any{"name": "Links" + uuid.NewString()[:8], "type": "multi_url"})
+ if w := setIssuePropertyRaw(t, issueID, listURL.ID, []string{"https://example.com/a", "https://example.com/b"}); w.Code != http.StatusOK {
+ t.Fatalf("good multi_url: expected 200, got %d: %s", w.Code, w.Body.String())
+ }
+ if w := setIssuePropertyRaw(t, issueID, listURL.ID, []string{"https://example.com/a", "ftp://example.com"}); w.Code != http.StatusBadRequest {
+ t.Fatalf("non-http(s) multi_url entry: expected 400, got %d", w.Code)
+ }
+
// Archived definitions reject new values but allow unset.
warch := httptest.NewRecorder()
req := newRequest("PATCH", "/api/properties/"+sel.ID, map[string]any{"archived": true})
@@ -362,6 +387,52 @@ func TestValidatePropertyValueUnit(t *testing.T) {
}
}
+func TestValidatePropertyListValuesUnit(t *testing.T) {
+ validate := func(def db.IssueProperty, raw string) ([]byte, error) {
+ return issueproperty.ValidateValue(def, json.RawMessage(raw))
+ }
+ textList := makePropertyDef("multi_text", nil)
+ if _, err := validate(textList, `[]`); err == nil {
+ t.Fatalf("empty multi_text array accepted")
+ }
+ if _, err := validate(textList, `"alpha"`); err == nil {
+ t.Fatalf("bare string into multi_text accepted")
+ }
+ if _, err := validate(textList, `["alpha", 3]`); err == nil {
+ t.Fatalf("non-string multi_text entry accepted")
+ }
+ if _, err := validate(textList, `["alpha", " "]`); err == nil {
+ t.Fatalf("blank multi_text entry accepted")
+ }
+ stored, err := validate(textList, `["alpha", "beta", "alpha"]`)
+ if err != nil {
+ t.Fatalf("valid multi_text rejected: %v", err)
+ }
+ if string(stored) != `["alpha","beta"]` {
+ t.Fatalf("multi_text not deduped in caller order: %s", stored)
+ }
+ over := make([]string, issueproperty.MaxListValues+1)
+ for i := range over {
+ over[i] = fmt.Sprintf("v%d", i)
+ }
+ overRaw, _ := json.Marshal(over)
+ if _, err := validate(textList, string(overRaw)); err == nil {
+ t.Fatalf("over-cap multi_text accepted")
+ }
+
+ urlList := makePropertyDef("multi_url", nil)
+ if _, err := validate(urlList, `["https://a.example", "javascript:alert(1)"]`); err == nil {
+ t.Fatalf("non-http(s) multi_url entry accepted")
+ }
+ stored, err = validate(urlList, `[" https://a.example/x ", "https://b.example", "https://a.example/x"]`)
+ if err != nil {
+ t.Fatalf("valid multi_url rejected: %v", err)
+ }
+ if string(stored) != `["https://a.example/x","https://b.example"]` {
+ t.Fatalf("multi_url not trimmed and deduped: %s", stored)
+ }
+}
+
func TestValidatePropertyNameReserved(t *testing.T) {
for _, name := range []string{"status", "Priority", "due date", "Due_Date", "START DATE", "labels"} {
if _, err := validatePropertyName(name); err == nil {
diff --git a/server/internal/handler/runtime_local_skills.go b/server/internal/handler/runtime_local_skills.go
index b1f88991a5e..e38d5befff6 100644
--- a/server/internal/handler/runtime_local_skills.go
+++ b/server/internal/handler/runtime_local_skills.go
@@ -14,6 +14,7 @@ import (
"github.com/go-chi/chi/v5"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgtype"
+ "github.com/multica-ai/multica/server/internal/logger"
obsmetrics "github.com/multica-ai/multica/server/internal/metrics"
"github.com/multica-ai/multica/server/internal/util"
db "github.com/multica-ai/multica/server/pkg/db/generated"
@@ -604,7 +605,8 @@ func (h *Handler) InitiateListLocalSkills(w http.ResponseWriter, r *http.Request
req, err := h.LocalSkillListStore.Create(r.Context(), rt.runtimeID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to enqueue local skills request: "+err.Error())
+ slog.Warn("enqueue local skills request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to enqueue local skills request")
return
}
h.requestDaemonPendingWork(rt.runtimeID, protocol.PendingWorkKindLocalSkills)
@@ -621,7 +623,8 @@ func (h *Handler) GetLocalSkillListRequest(w http.ResponseWriter, r *http.Reques
requestID := chi.URLParam(r, "requestId")
req, err := h.LocalSkillListStore.Get(r.Context(), requestID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load request: "+err.Error())
+ slog.Warn("load request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load request")
return
}
if req == nil || req.RuntimeID != rt.runtimeID {
@@ -689,7 +692,8 @@ func (h *Handler) InitiateImportLocalSkill(w http.ResponseWriter, r *http.Reques
SupportsConflict: req.SupportsConflict || req.Action == LocalSkillImportActionOverwrite,
})
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to enqueue local skill import: "+err.Error())
+ slog.Warn("enqueue local skill import failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to enqueue local skill import")
return
}
h.requestDaemonPendingWork(rt.runtimeID, protocol.PendingWorkKindLocalSkillImport)
@@ -706,7 +710,8 @@ func (h *Handler) GetLocalSkillImportRequest(w http.ResponseWriter, r *http.Requ
requestID := chi.URLParam(r, "requestId")
req, err := h.LocalSkillImportStore.Get(r.Context(), requestID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load request: "+err.Error())
+ slog.Warn("load request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load request")
return
}
if req == nil || req.RuntimeID != rt.runtimeID {
@@ -726,7 +731,8 @@ func (h *Handler) ReportLocalSkillListResult(w http.ResponseWriter, r *http.Requ
requestID := chi.URLParam(r, "requestId")
req, err := h.LocalSkillListStore.Get(r.Context(), requestID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load request: "+err.Error())
+ slog.Warn("load request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load request")
return
}
if req == nil || req.RuntimeID != runtimeID {
@@ -792,7 +798,8 @@ func (h *Handler) ReportLocalSkillImportResult(w http.ResponseWriter, r *http.Re
requestID := chi.URLParam(r, "requestId")
req, err := h.LocalSkillImportStore.Get(r.Context(), requestID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load request: "+err.Error())
+ slog.Warn("load request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load request")
return
}
if req == nil || req.RuntimeID != runtimeID {
@@ -884,16 +891,16 @@ func (h *Handler) ReportLocalSkillImportResult(w http.ResponseWriter, r *http.Re
Files: files,
})
if oerr != nil {
- failMsg := oerr.Error()
switch {
case errors.Is(oerr, errSkillOverwriteNotFound):
- failMsg = "target skill no longer exists"
+ h.failLocalSkillImport(w, r, requestID, "target skill no longer exists")
case errors.Is(oerr, errSkillOverwriteForbidden):
- failMsg = "you no longer have permission to overwrite this skill"
+ h.failLocalSkillImport(w, r, requestID, "you no longer have permission to overwrite this skill")
case errors.Is(oerr, errSkillOverwriteNameMismatch):
- failMsg = "target skill name no longer matches the imported skill"
+ h.failLocalSkillImport(w, r, requestID, "target skill name no longer matches the imported skill")
+ default:
+ h.failLocalSkillImportInternal(w, r, requestID, "failed to overwrite skill", oerr)
}
- h.failLocalSkillImport(w, r, requestID, failMsg)
return
}
if err := h.LocalSkillImportStore.Complete(r.Context(), requestID, resp); err != nil {
@@ -917,7 +924,7 @@ func (h *Handler) ReportLocalSkillImportResult(w http.ResponseWriter, r *http.Re
// can offer overwrite / rename / skip; older clients keep the legacy
// `failed` behavior (see resolveLocalSkillConflict).
if existing, found, lerr := h.lookupSkillByName(r.Context(), rt.WorkspaceID, sanitizeNullBytes(name)); lerr != nil {
- h.failLocalSkillImport(w, r, requestID, "failed to check for existing skill: "+lerr.Error())
+ h.failLocalSkillImportInternal(w, r, requestID, "failed to check for existing skill", lerr)
return
} else if found {
h.resolveLocalSkillConflict(w, r, req, existing)
@@ -946,7 +953,7 @@ func (h *Handler) ReportLocalSkillImportResult(w http.ResponseWriter, r *http.Re
h.failLocalSkillImport(w, r, requestID, "a skill with this name already exists")
return
}
- h.failLocalSkillImport(w, r, requestID, err.Error())
+ h.failLocalSkillImportInternal(w, r, requestID, "failed to create skill", err)
return
}
@@ -984,6 +991,14 @@ func (h *Handler) failLocalSkillImport(w http.ResponseWriter, r *http.Request, r
writeJSON(w, http.StatusOK, map[string]string{"status": "ok"})
}
+// failLocalSkillImportInternal keeps infrastructure details in server logs and
+// stores only stable copy in the user-polled import result.
+func (h *Handler) failLocalSkillImportInternal(w http.ResponseWriter, r *http.Request, requestID, publicMsg string, cause error) {
+ slog.Warn("runtime local skill import failed", append(logger.RequestAttrs(r),
+ "error", cause, "import_request_id", requestID, "public_message", publicMsg)...)
+ h.failLocalSkillImport(w, r, requestID, publicMsg)
+}
+
// resolveLocalSkillConflict terminates a same-name create import. Clients that
// opted into the structured-conflict contract (SupportsConflict) receive the
// `conflict` status plus metadata so they can offer overwrite / rename / skip;
diff --git a/server/internal/handler/runtime_local_skills_test.go b/server/internal/handler/runtime_local_skills_test.go
index 029f013eb33..d4b574b48cc 100644
--- a/server/internal/handler/runtime_local_skills_test.go
+++ b/server/internal/handler/runtime_local_skills_test.go
@@ -4,9 +4,11 @@ import (
"bytes"
"context"
"encoding/json"
+ "errors"
"fmt"
"net/http"
"net/http/httptest"
+ "strings"
"testing"
"time"
@@ -233,6 +235,45 @@ func TestInMemoryLocalSkillImportStore_TimesOutRunningRequests(t *testing.T) {
}
}
+func TestFailLocalSkillImportInternalRedactsCause(t *testing.T) {
+ ctx := context.Background()
+ store := NewInMemoryLocalSkillImportStore()
+ req, err := store.Create(ctx, LocalSkillImportRequestInput{
+ RuntimeID: "runtime-xyz",
+ CreatorID: "user-1",
+ SkillKey: "review-helper",
+ })
+ if err != nil {
+ t.Fatalf("create: %v", err)
+ }
+
+ h := Handler{LocalSkillImportStore: store}
+ w := httptest.NewRecorder()
+ r := httptest.NewRequest(http.MethodPost, "/api/daemon/local-skills/import/result", nil)
+ const leak = "postgres://secret@internal.example/skills"
+ h.failLocalSkillImportInternal(w, r, req.ID, "failed to create skill", errors.New(leak))
+
+ if w.Code != http.StatusOK {
+ t.Fatalf("status = %d, want %d", w.Code, http.StatusOK)
+ }
+ if strings.Contains(w.Body.String(), leak) {
+ t.Fatalf("daemon response leaked internal error: %s", w.Body.String())
+ }
+ got, err := store.Get(ctx, req.ID)
+ if err != nil {
+ t.Fatalf("get: %v", err)
+ }
+ if got == nil {
+ t.Fatal("expected stored import result")
+ }
+ if got.Error != "failed to create skill" {
+ t.Fatalf("stored error = %q, want stable public message", got.Error)
+ }
+ if strings.Contains(got.Error, leak) {
+ t.Fatalf("stored error leaked internal cause: %q", got.Error)
+ }
+}
+
// Capability discovery (list + poll) is readable by workspace members only
// after the runtime owner shares the machine with the workspace.
func TestListLocalSkills_AllowsNonOwnerForPublicRuntime(t *testing.T) {
diff --git a/server/internal/handler/runtime_models.go b/server/internal/handler/runtime_models.go
index 66c2df0b44c..ab57f6f7b1d 100644
--- a/server/internal/handler/runtime_models.go
+++ b/server/internal/handler/runtime_models.go
@@ -9,6 +9,7 @@ import (
"time"
"github.com/go-chi/chi/v5"
+ "github.com/multica-ai/multica/server/internal/logger"
obsmetrics "github.com/multica-ai/multica/server/internal/metrics"
"github.com/multica-ai/multica/server/pkg/protocol"
)
@@ -383,7 +384,8 @@ func (h *Handler) InitiateListModels(w http.ResponseWriter, r *http.Request) {
req, err := h.ModelListStore.Create(r.Context(), resolvedRuntimeID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to enqueue model list request: "+err.Error())
+ slog.Warn("enqueue model list request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to enqueue model list request")
return
}
h.requestDaemonPendingWork(resolvedRuntimeID, protocol.PendingWorkKindModelList)
@@ -468,7 +470,8 @@ func (h *Handler) GetModelListRequest(w http.ResponseWriter, r *http.Request) {
req, err := h.ModelListStore.Get(r.Context(), requestID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load request: "+err.Error())
+ slog.Warn("load request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load request")
return
}
if req == nil || req.RuntimeID != uuidToString(rt.ID) {
@@ -493,7 +496,8 @@ func (h *Handler) ReportModelListResult(w http.ResponseWriter, r *http.Request)
// run was a retry, and the original report already landed).
existing, err := h.ModelListStore.Get(r.Context(), requestID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load request: "+err.Error())
+ slog.Warn("load request failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load request")
return
}
if existing == nil || existing.RuntimeID != runtimeID {
diff --git a/server/internal/handler/runtime_update.go b/server/internal/handler/runtime_update.go
index 634c4ca197d..07f94c62618 100644
--- a/server/internal/handler/runtime_update.go
+++ b/server/internal/handler/runtime_update.go
@@ -10,6 +10,7 @@ import (
"time"
"github.com/go-chi/chi/v5"
+ "github.com/multica-ai/multica/server/internal/logger"
obsmetrics "github.com/multica-ai/multica/server/internal/metrics"
)
@@ -270,7 +271,8 @@ func (h *Handler) GetUpdate(w http.ResponseWriter, r *http.Request) {
update, err := h.UpdateStore.Get(r.Context(), updateID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load update: "+err.Error())
+ slog.Warn("load update failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load update")
return
}
if update == nil || update.RuntimeID != uuidToString(rt.ID) {
@@ -301,7 +303,8 @@ func (h *Handler) ReportUpdateResult(w http.ResponseWriter, r *http.Request) {
existing, err := h.UpdateStore.Get(r.Context(), updateID)
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to load update: "+err.Error())
+ slog.Warn("load update failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to load update")
return
}
if existing == nil || existing.RuntimeID != runtimeID {
diff --git a/server/internal/handler/search_identifier_test.go b/server/internal/handler/search_identifier_test.go
new file mode 100644
index 00000000000..8beb0fd4af8
--- /dev/null
+++ b/server/internal/handler/search_identifier_test.go
@@ -0,0 +1,97 @@
+package handler
+
+import (
+ "fmt"
+ "net/http"
+ "net/url"
+ "testing"
+
+ "github.com/google/uuid"
+ "github.com/multica-ai/multica/server/internal/testutil"
+)
+
+func TestParseQueryNumber(t *testing.T) {
+ for _, tc := range []struct {
+ query string
+ want int
+ }{
+ {"MUL-123", 123},
+ {"mul-123", 123},
+ {" 123 ", 123},
+ {"V2-12", 12},
+ {" v2-12 ", 12},
+ {"A1-7", 7},
+ {"a1b2-42", 42},
+ {"12-3", 0},
+ {"2FAS-1", 0},
+ {"V2-0", 0},
+ {"V2--12", 0},
+ {"V2-12x", 0},
+ {"V_2-12", 0},
+ {"MUL-", 0},
+ {"search", 0},
+ {"0", 0},
+ {"", 0},
+ } {
+ t.Run(tc.query, func(t *testing.T) {
+ got, ok := parseQueryNumber(tc.query)
+ if got != tc.want || ok != (tc.want > 0) {
+ t.Fatalf("parseQueryNumber(%q) = (%d, %t), want (%d, %t)", tc.query, got, ok, tc.want, tc.want > 0)
+ }
+ })
+ }
+}
+
+func TestIssueSearch_DigitPrefix(t *testing.T) {
+ workspaceID := dbfx.Workspace(t, "Digit prefix search", "digit-search-"+uuid.NewString(), testutil.Cols{"issue_prefix": "V2"})
+ dbfx.Member(t, workspaceID, testUserID, "owner")
+ targetID := dbfx.Issue(t, "Abandoned plan", testutil.Cols{
+ "workspace_id": workspaceID, "number": 12, "status": "cancelled",
+ })
+ liveID := dbfx.Issue(t, "Notes about V2-12", testutil.Cols{"workspace_id": workspaceID})
+ foreignWorkspaceID := dbfx.Workspace(t, "Other search workspace", "other-search-"+uuid.NewString())
+ dbfx.Issue(t, "V2-12", testutil.Cols{"workspace_id": foreignWorkspaceID, "number": 12})
+
+ for _, query := range []string{"V2-12", "v2-12", "12"} {
+ t.Run(query, func(t *testing.T) {
+ // The fallback API must retain the exact cancelled hit ahead of live
+ // text matches, even when the result window has only one slot.
+ for _, includeClosed := range []bool{true, false} {
+ path := fmt.Sprintf("/api/issues/search?q=%s&include_closed=%t&limit=1", url.QueryEscape(query), includeClosed)
+ req := newRequest(http.MethodGet, path, nil)
+ req.Header.Set("X-Workspace-ID", workspaceID)
+ var response struct {
+ Issues []SearchIssueResponse `json:"issues"`
+ }
+ testutil.Call(t, testHandler.SearchIssues, req).Want(http.StatusOK).JSON(&response)
+ wantID := liveID
+ if includeClosed {
+ wantID = targetID
+ }
+ if len(response.Issues) != 1 || response.Issues[0].ID != wantID {
+ t.Fatalf("search include_closed=%t: got %+v, want issue %s", includeClosed, response.Issues, wantID)
+ }
+ }
+
+ // List search uses the same parser while preserving active filters
+ // and the total for the filtered result set.
+ for _, status := range []string{"cancelled", "todo"} {
+ path := fmt.Sprintf("/api/issues?q=%s&status=%s&limit=1", url.QueryEscape(query), status)
+ req := newRequest(http.MethodGet, path, nil)
+ req.Header.Set("X-Workspace-ID", workspaceID)
+ var response struct {
+ Issues []IssueResponse `json:"issues"`
+ Total int64 `json:"total"`
+ }
+ testutil.Call(t, testHandler.ListIssues, req).Want(http.StatusOK).JSON(&response)
+ wantID := liveID
+ if status == "cancelled" {
+ wantID = targetID
+ }
+ if response.Total != 1 || len(response.Issues) != 1 || response.Issues[0].ID != wantID {
+ t.Fatalf("list status=%s: got %+v, want only issue %s", status, response, wantID)
+ }
+ }
+ })
+ }
+}
diff --git a/server/internal/handler/skill.go b/server/internal/handler/skill.go
index c570a3c1303..36c2b040f45 100644
--- a/server/internal/handler/skill.go
+++ b/server/internal/handler/skill.go
@@ -20,6 +20,7 @@ import (
"github.com/go-chi/chi/v5"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgtype"
+ "github.com/multica-ai/multica/server/internal/logger"
skillpkg "github.com/multica-ai/multica/server/internal/skill"
"github.com/multica-ai/multica/server/internal/util"
db "github.com/multica-ai/multica/server/pkg/db/generated"
@@ -569,7 +570,8 @@ func (h *Handler) CreateSkill(w http.ResponseWriter, r *http.Request) {
writeError(w, http.StatusConflict, "a skill with this name already exists")
return
}
- writeError(w, http.StatusInternalServerError, "failed to create skill: "+err.Error())
+ slog.Warn("create skill failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to create skill")
return
}
actorType, actorID := h.resolveActor(r, creatorID, workspaceID)
@@ -658,7 +660,8 @@ func (h *Handler) UpdateSkill(w http.ResponseWriter, r *http.Request) {
writeError(w, http.StatusConflict, "a skill with this name already exists")
return
}
- writeError(w, http.StatusInternalServerError, "failed to update skill: "+err.Error())
+ slog.Warn("update skill failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to update skill")
return
}
@@ -681,7 +684,8 @@ func (h *Handler) UpdateSkill(w http.ResponseWriter, r *http.Request) {
Content: sanitizeNullBytes(f.Content),
})
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to upsert skill file: "+err.Error())
+ slog.Warn("upsert skill file failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to upsert skill file")
return
}
fileResps = append(fileResps, skillFileToResponse(sf))
@@ -2235,7 +2239,7 @@ func skillImportOverwriteFailure(err error) (int, string) {
case errors.Is(err, errSkillOverwriteNameMismatch):
return http.StatusConflict, "target skill name no longer matches the imported skill"
default:
- return http.StatusInternalServerError, "failed to overwrite skill: " + err.Error()
+ return http.StatusInternalServerError, "failed to overwrite skill"
}
}
@@ -2269,6 +2273,9 @@ func (h *Handler) resolveImportSkillConflict(w http.ResponseWriter, r *http.Requ
})
if err != nil {
status, reason := skillImportOverwriteFailure(err)
+ if status == http.StatusInternalServerError {
+ slog.Warn("overwrite imported skill failed", append(logger.RequestAttrs(r), "error", err)...)
+ }
writeJSON(w, status, SkillImportResult{
Status: "failed",
Reason: reason,
@@ -2282,9 +2289,10 @@ func (h *Handler) resolveImportSkillConflict(w http.ResponseWriter, r *http.Requ
case importOnConflictRename:
resp, err := h.createRenamedImportedSkill(r.Context(), workspaceUUID, creatorUUID, name, imported, config, files)
if err != nil {
+ slog.Warn("create renamed skill failed", append(logger.RequestAttrs(r), "error", err)...)
writeJSON(w, http.StatusInternalServerError, SkillImportResult{
Status: "failed",
- Reason: "failed to create renamed skill: " + err.Error(),
+ Reason: "failed to create renamed skill",
ExistingSkill: &existingInfo,
})
return
@@ -2435,9 +2443,10 @@ func (h *Handler) finishSkillImport(w http.ResponseWriter, r *http.Request, work
if structuredResult {
if existing, found, lerr := h.lookupSkillByName(r.Context(), workspaceUUID, name); lerr != nil {
+ slog.Warn("look up existing skill failed", append(logger.RequestAttrs(r), "error", lerr)...)
writeJSON(w, http.StatusInternalServerError, SkillImportResult{
Status: "failed",
- Reason: "failed to check for existing skill: " + lerr.Error(),
+ Reason: "failed to check for existing skill",
})
return
} else if found {
@@ -2462,7 +2471,8 @@ func (h *Handler) finishSkillImport(w http.ResponseWriter, r *http.Request, work
}
return
}
- writeError(w, http.StatusInternalServerError, "failed to create skill: "+err.Error())
+ slog.Warn("create skill failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to create skill")
return
}
actorType, actorID := h.resolveActor(r, creatorID, workspaceID)
@@ -2545,7 +2555,8 @@ func (h *Handler) UpsertSkillFile(w http.ResponseWriter, r *http.Request) {
Content: sanitizeNullBytes(req.Content),
})
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to upsert skill file: "+err.Error())
+ slog.Warn("upsert skill file failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to upsert skill file")
return
}
@@ -2649,7 +2660,8 @@ func (h *Handler) SetAgentSkills(w http.ResponseWriter, r *http.Request) {
AgentID: agent.ID,
SkillID: skillID,
}); err != nil {
- writeError(w, http.StatusInternalServerError, "failed to add agent skill: "+err.Error())
+ slog.Warn("add agent skill failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to add agent skill")
return
}
}
@@ -2698,7 +2710,8 @@ func (h *Handler) AddAgentSkills(w http.ResponseWriter, r *http.Request) {
AgentID: agent.ID,
SkillID: skillID,
}); err != nil {
- writeError(w, http.StatusInternalServerError, "failed to add agent skill: "+err.Error())
+ slog.Warn("add agent skill failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to add agent skill")
return
}
}
diff --git a/server/internal/handler/skill_refresh.go b/server/internal/handler/skill_refresh.go
index 1e22ac31654..f1e1017f615 100644
--- a/server/internal/handler/skill_refresh.go
+++ b/server/internal/handler/skill_refresh.go
@@ -11,6 +11,7 @@ import (
"github.com/go-chi/chi/v5"
+ "github.com/multica-ai/multica/server/internal/logger"
"github.com/multica-ai/multica/server/pkg/protocol"
db "github.com/multica-ai/multica/server/pkg/db/generated"
@@ -191,7 +192,8 @@ func (h *Handler) RefreshSkill(w http.ResponseWriter, r *http.Request) {
case errors.Is(err, errSkillOverwriteNameConflict):
writeError(w, http.StatusConflict, "a skill named \""+newName+"\" already exists in this workspace")
default:
- writeError(w, http.StatusInternalServerError, "failed to update skill from source: "+err.Error())
+ slog.Warn("update skill from source failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to update skill from source")
}
return
}
diff --git a/server/internal/handler/workspace.go b/server/internal/handler/workspace.go
index 425658cc80a..0772ad3212e 100644
--- a/server/internal/handler/workspace.go
+++ b/server/internal/handler/workspace.go
@@ -271,7 +271,8 @@ func (h *Handler) CreateWorkspace(w http.ResponseWriter, r *http.Request) {
writeError(w, http.StatusConflict, "workspace slug already exists")
return
}
- writeError(w, http.StatusInternalServerError, "failed to create workspace: "+err.Error())
+ slog.Warn("create workspace failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to create workspace")
return
}
@@ -281,7 +282,8 @@ func (h *Handler) CreateWorkspace(w http.ResponseWriter, r *http.Request) {
Role: "owner",
})
if err != nil {
- writeError(w, http.StatusInternalServerError, "failed to add owner: "+err.Error())
+ slog.Warn("add owner failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to add owner")
return
}
@@ -289,7 +291,8 @@ func (h *Handler) CreateWorkspace(w http.ResponseWriter, r *http.Request) {
// workspace is never visible without its status catalog — an issue cannot
// be created before its status can be resolved. (MUL-6243)
if err := issuestatus.Ensure(r.Context(), qtx, ws.ID); err != nil {
- writeError(w, http.StatusInternalServerError, "failed to seed issue statuses: "+err.Error())
+ slog.Warn("seed issue statuses failed", append(logger.RequestAttrs(r), "error", err)...)
+ writeError(w, http.StatusInternalServerError, "failed to seed issue statuses")
return
}
@@ -451,7 +454,7 @@ func (h *Handler) UpdateWorkspace(w http.ResponseWriter, r *http.Request) {
ws, err := h.Queries.UpdateWorkspace(r.Context(), params)
if err != nil {
slog.Warn("update workspace failed", append(logger.RequestAttrs(r), "error", err, "workspace_id", id)...)
- writeError(w, http.StatusInternalServerError, "failed to update workspace: "+err.Error())
+ writeError(w, http.StatusInternalServerError, "failed to update workspace")
return
}
diff --git a/server/internal/handler/workspace_delete_manifest_test.go b/server/internal/handler/workspace_delete_manifest_test.go
index 584c7d0887a..cb9ede8950a 100644
--- a/server/internal/handler/workspace_delete_manifest_test.go
+++ b/server/internal/handler/workspace_delete_manifest_test.go
@@ -47,6 +47,7 @@ var workspaceDeletionManifest = map[string]workspaceDeleteAction{
"channel_outbound_message": workspaceDelete,
"channel_reply_delivery": workspaceDelete,
"channel_task_delivery": workspaceDelete,
+ "channel_typing_reaction": workspaceDeleteSettle, // Source deletion makes retained anchors eligible for the cleanup worker.
"channel_user_binding": workspaceDelete,
"chat_draft_restore": workspaceDelete,
"chat_message": workspaceDelete,
diff --git a/server/internal/handler/workspace_typing_cleanup_test.go b/server/internal/handler/workspace_typing_cleanup_test.go
new file mode 100644
index 00000000000..b7811e14518
--- /dev/null
+++ b/server/internal/handler/workspace_typing_cleanup_test.go
@@ -0,0 +1,127 @@
+package handler
+
+import (
+ "context"
+ "net/http"
+ "testing"
+
+ "github.com/multica-ai/multica/server/internal/testutil"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+func TestDeleteWorkspace_RetainsTypingCleanupUntilAcknowledged(t *testing.T) {
+ if testHandler == nil {
+ t.Skip("database not available")
+ }
+ for _, outcome := range []string{"acknowledged", "expired"} {
+ t.Run(outcome, func(t *testing.T) {
+ ctx := context.Background()
+ wsID := dbfx.Insert(t, "workspace", testutil.Cols{
+ "name": "Typing cleanup", "slug": "handler-delete-typing-cleanup",
+ })
+ dbfx.Exec(t, `INSERT INTO member (workspace_id, user_id, role) VALUES ($1, $2, 'owner')`, wsID, testUserID)
+ agentID := dbfx.Insert(t, "agent", testutil.Cols{
+ "workspace_id": wsID, "name": "Typing cleanup", "owner_id": testUserID, "runtime_mode": "cloud",
+ })
+ sessionID := dbfx.Insert(t, "chat_session", testutil.Cols{
+ "workspace_id": wsID, "agent_id": agentID, "creator_id": testUserID,
+ })
+ installationID := dbfx.Insert(t, "channel_installation", testutil.Cols{
+ "workspace_id": wsID, "agent_id": agentID, "channel_type": "feishu", "installer_user_id": testUserID,
+ })
+ dbfx.Insert(t, "channel_chat_session_binding", testutil.Cols{
+ "chat_session_id": sessionID, "installation_id": installationID,
+ "channel_type": "feishu", "channel_chat_id": "typing-cleanup-room", "chat_type": "group",
+ })
+ messageID := dbfx.Insert(t, "chat_message", testutil.Cols{
+ "chat_session_id": sessionID, "role": "user", "content": "pending input", "channel_ingested": true,
+ })
+ const snapshot = `{"app_id":"test-app","app_secret_enc":"test-encrypted-snapshot"}`
+ reactionID := dbfx.Insert(t, "channel_typing_reaction", testutil.Cols{
+ "id": testutil.Raw("gen_random_uuid()"), "workspace_id": wsID,
+ "chat_session_id": sessionID, "chat_message_id": messageID, "installation_id": installationID,
+ "channel_message_id": "remote-input", "installation_snapshot": snapshot,
+ "reaction_id": "remote-reaction", "add_finished": true,
+ })
+ queries := db.New(testPool)
+ rows, err := queries.ClaimChannelTypingReactionCleanup(ctx, parseUUID(sessionID))
+ if err != nil || len(rows) != 0 {
+ t.Fatalf("active input must not be claimed: rows=%d err=%v", len(rows), err)
+ }
+
+ req := withURLParam(newRequest(http.MethodDelete, "/api/workspaces/"+wsID, nil), "id", wsID)
+ testutil.Call(t, testHandler.DeleteWorkspace, req).Want(http.StatusNoContent)
+ for table, id := range map[string]string{
+ "workspace": wsID, "chat_session": sessionID, "channel_installation": installationID, "chat_message": messageID,
+ } {
+ var exists bool
+ dbfx.QueryRow(t, `SELECT EXISTS (SELECT 1 FROM `+table+` WHERE id = $1)`, id).Scan(&exists)
+ if exists {
+ t.Fatalf("%s source survived workspace deletion", table)
+ }
+ }
+
+ // A fresh worker can claim without joining to the now-deleted workspace,
+ // session or installation. The external anchor and encrypted snapshot survive.
+ queries = db.New(testPool)
+ rows, err = queries.ClaimChannelTypingReactionCleanup(ctx, parseUUID(sessionID))
+ if err != nil || len(rows) != 1 {
+ t.Fatalf("deleted workspace cleanup: rows=%d err=%v", len(rows), err)
+ }
+ row := rows[0]
+ if row.ID != parseUUID(reactionID) || row.ChannelMessageID != "remote-input" || row.ReactionID != "remote-reaction" || !row.CleanupRequired || row.CleanedAt.Valid {
+ t.Fatalf("cleanup anchor was lost or prematurely acknowledged: %+v", row)
+ }
+ var snapshotPreserved bool
+ dbfx.QueryRow(t, `SELECT installation_snapshot = $2::jsonb FROM channel_typing_reaction WHERE id = $1`, reactionID, snapshot).Scan(&snapshotPreserved)
+ if !snapshotPreserved {
+ t.Fatal("cleanup lost the encrypted installation snapshot")
+ }
+ // Pruning must retain unacknowledged cleanup even when source data is gone.
+ if err := queries.PruneChannelTypingReactionCleanup(ctx); err != nil {
+ t.Fatal(err)
+ }
+ var pending bool
+ dbfx.QueryRow(t, `SELECT EXISTS (SELECT 1 FROM channel_typing_reaction WHERE id = $1 AND cleaned_at IS NULL)`, reactionID).Scan(&pending)
+ if !pending {
+ t.Fatal("pruning dropped pending workspace cleanup")
+ }
+ if outcome == "expired" {
+ dbfx.Exec(t, "UPDATE channel_typing_reaction SET created_at=now()-interval '8 days' WHERE id=$1", reactionID)
+ expired, err := queries.ExpireChannelTypingReactionCleanup(ctx)
+ if err != nil {
+ t.Fatal(err)
+ }
+ found := false
+ for _, row := range expired {
+ if row.ID == parseUUID(reactionID) {
+ found = true
+ }
+ }
+ if !found {
+ t.Fatal("deleted workspace credential retention has no terminal outcome")
+ }
+ var redacted, abandoned, cleaned bool
+ dbfx.QueryRow(t, `SELECT installation_snapshot='{}'::jsonb, abandoned_at IS NOT NULL, cleaned_at IS NOT NULL FROM channel_typing_reaction WHERE id=$1`, reactionID).Scan(&redacted, &abandoned, &cleaned)
+ if !redacted || !abandoned || cleaned {
+ t.Fatal("expiry must erase credentials and record failure, not success")
+ }
+ rows, err := queries.ClaimChannelTypingReactionCleanup(ctx, parseUUID(sessionID))
+ if err != nil || len(rows) != 0 {
+ t.Fatalf("expired deletion still retries: %v %v", rows, err)
+ }
+ return
+ }
+ if err := queries.AcknowledgeChannelTypingReactionCleanup(ctx, db.AcknowledgeChannelTypingReactionCleanupParams{
+ ID: row.ID, ReactionID: row.ReactionID,
+ }); err != nil {
+ t.Fatal(err)
+ }
+ var cleaned bool
+ dbfx.QueryRow(t, `SELECT cleaned_at IS NOT NULL FROM channel_typing_reaction WHERE id = $1`, reactionID).Scan(&cleaned)
+ if !cleaned {
+ t.Fatal("cleanup could not be acknowledged after workspace deletion")
+ }
+ })
+ }
+}
diff --git a/server/internal/integrations/channel/engine/resolvers.go b/server/internal/integrations/channel/engine/resolvers.go
index cd8aeb9e807..44579aa4e8d 100644
--- a/server/internal/integrations/channel/engine/resolvers.go
+++ b/server/internal/integrations/channel/engine/resolvers.go
@@ -55,10 +55,12 @@ const (
// consumed by the outbound side (OutboundReplier / typing). It mirrors the
// legacy lark.DispatchResult.
type Result struct {
- Outcome Outcome
- DropReason DropReason
- InstallationID pgtype.UUID
- ChatSessionID pgtype.UUID
+ Outcome Outcome
+ DropReason DropReason
+ InstallationID pgtype.UUID
+ ChatSessionID pgtype.UUID
+ // ChatMessageID identifies this persisted input, including before debounce creates a task.
+ ChatMessageID pgtype.UUID
ChannelBindingID pgtype.UUID
ChannelRouteRevision int64
// Sender is the platform-native sender id (e.g. Lark open_id), so the
@@ -386,14 +388,14 @@ type OutboundReplier interface {
// it.
type TypingNotifier interface {
// OnIngested shows the indicator for a successfully ingested message.
- OnIngested(ctx context.Context, inst ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID)
+ OnIngested(ctx context.Context, inst ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID, chatMessageID pgtype.UUID)
// OnSettled clears the indicator for a session whose run trigger produced no
// task (agent offline / archived, or an enqueue failure). In that case no
// task lifecycle event is ever published, so the platform's own bus-driven
// clear (on chat-done / task-failed) would never fire and the indicator would
// stick. The Router calls this from the debounced flush. Idempotent: a
// session with no indicator is a no-op.
- OnSettled(ctx context.Context, sessionID pgtype.UUID)
+ OnSettled(ctx context.Context, sessionID pgtype.UUID, scope TypingSettlement)
}
// ResolverSet is the per-platform bundle the Router runs the pipeline through.
@@ -445,3 +447,10 @@ type SessionReader interface {
GetChatSession(ctx context.Context, id pgtype.UUID) (db.ChatSession, error)
GetWorkspace(ctx context.Context, id pgtype.UUID) (db.Workspace, error)
}
+
+// TypingSettlement identifies the failed flush's immutable input boundary.
+// It excludes later arrivals and other context generations in the same session.
+type TypingSettlement struct {
+ WorkspaceID, InstallationID, ThroughMessageID pgtype.UUID
+ ContextRevision int64
+}
diff --git a/server/internal/integrations/channel/engine/router.go b/server/internal/integrations/channel/engine/router.go
index d049cb5a50d..71f2bcaab33 100644
--- a/server/internal/integrations/channel/engine/router.go
+++ b/server/internal/integrations/channel/engine/router.go
@@ -274,7 +274,7 @@ func (r *Router) Handle(ctx context.Context, msg channel.InboundMessage) error {
go func() {
tctx, cancel := context.WithTimeout(context.Background(), r.replyTimeout)
defer cancel()
- set.Typing.OnIngested(tctx, inst, msg, res.ChatSessionID)
+ set.Typing.OnIngested(tctx, inst, msg, res.ChatSessionID, res.ChatMessageID)
}()
}
r.scheduleReply(set, inst, msg, res)
@@ -564,6 +564,7 @@ func (r *Router) processClaimed(ctx context.Context, set ResolverSet, msg channe
Outcome: OutcomeIngested,
InstallationID: inst.ID,
ChatSessionID: sessionID,
+ ChatMessageID: appendRes.MessageID,
ChannelBindingID: appendRes.BindingID,
ChannelRouteRevision: appendRes.RouteRevision,
Sender: msg.Source.SenderID,
@@ -674,12 +675,12 @@ func (r *Router) processClaimed(ctx context.Context, set ResolverSet, msg channe
if revision == appendRes.ContextRevision {
r.scheduleRunWithFresh(
set, inst, msg, sessionID, identity.UserID,
- res.ChannelBindingID, res.ChannelRouteRevision, forceFresh, revision,
+ res.ChannelBindingID, res.ChannelRouteRevision, forceFresh, revision, appendRes.MessageID,
)
} else if pending.InitiatorUserID.Valid {
r.scheduleRecoveredRun(
set, inst, msg, sessionID, pending.InitiatorUserID,
- res.ChannelBindingID, res.ChannelRouteRevision, revision,
+ res.ChannelBindingID, res.ChannelRouteRevision, revision, appendRes.MessageID,
)
} else {
slog.Warn("skipping recovered channel context without initiator snapshot",
@@ -915,10 +916,11 @@ func (r *Router) scheduleRunWithFresh(
routeRevision int64,
fresh bool,
contextRevision int64,
+ throughMessageID pgtype.UUID,
) {
r.scheduleRunMode(
set, inst, msg, sessionID, initiatorUserID, bindingID,
- routeRevision, fresh, contextRevision, true,
+ routeRevision, fresh, contextRevision, true, throughMessageID,
)
}
@@ -928,10 +930,11 @@ func (r *Router) scheduleRecoveredRun(
msg channel.InboundMessage,
sessionID, initiatorUserID, bindingID pgtype.UUID,
routeRevision, contextRevision int64,
+ throughMessageID pgtype.UUID,
) {
r.scheduleRunMode(
set, inst, msg, sessionID, initiatorUserID, bindingID,
- routeRevision, false, contextRevision, false,
+ routeRevision, false, contextRevision, false, throughMessageID,
)
}
@@ -944,11 +947,12 @@ func (r *Router) scheduleRunMode(
fresh bool,
contextRevision int64,
replace bool,
+ throughMessageID pgtype.UUID,
) {
if r.batcher == nil {
r.flushChatRun(
set, inst, msg, sessionID, initiatorUserID, bindingID,
- routeRevision, fresh, contextRevision,
+ routeRevision, fresh, contextRevision, throughMessageID,
)
return
}
@@ -959,7 +963,7 @@ func (r *Router) scheduleRunMode(
// batch key; the pre-boundary flush remains armed independently.
r.flushChatRun(
set, inst, msg, sessionID, initiatorUserID, bindingID,
- routeRevision, fresh, contextRevision,
+ routeRevision, fresh, contextRevision, throughMessageID,
)
}
if replace {
@@ -984,6 +988,7 @@ func (r *Router) flushChatRun(
routeRevision int64,
forceFresh bool,
contextRevision int64,
+ throughMessageID pgtype.UUID,
) {
ctx, cancel := context.WithTimeout(context.Background(), chatRunFlushTimeout)
defer cancel()
@@ -992,7 +997,7 @@ func (r *Router) flushChatRun(
if err != nil {
r.logger.Error("channel router: flush reload chat session failed",
"chat_session_id", uuidString(sessionID), "err", err.Error())
- r.clearTyping(ctx, set, sessionID)
+ r.clearTyping(ctx, set, sessionID, TypingSettlement{WorkspaceID: inst.WorkspaceID, InstallationID: inst.ID, ThroughMessageID: throughMessageID, ContextRevision: contextRevision})
return
}
if _, err := r.tasks.EnqueueChannelChatTask(
@@ -1002,7 +1007,7 @@ func (r *Router) flushChatRun(
// the platform's bus-driven typing clear can never fire. Clear the
// indicator here (before any notice) so the "processing" reaction does
// not stick on the user's message.
- r.clearTyping(ctx, set, sessionID)
+ r.clearTyping(ctx, set, sessionID, TypingSettlement{WorkspaceID: inst.WorkspaceID, InstallationID: inst.ID, ThroughMessageID: throughMessageID, ContextRevision: contextRevision})
switch {
case errors.Is(err, service.ErrChatTaskAgentNoRuntime):
r.emitFlushReply(ctx, set, inst, msg, sessionID, bindingID, routeRevision, OutcomeAgentOffline)
@@ -1018,9 +1023,9 @@ func (r *Router) flushChatRun(
// clearTyping asks the platform to drop the "processing" indicator for a session
// whose flush produced no task run. A nil TypingNotifier (platform without the
// feature) is a no-op.
-func (r *Router) clearTyping(ctx context.Context, set ResolverSet, sessionID pgtype.UUID) {
+func (r *Router) clearTyping(ctx context.Context, set ResolverSet, sessionID pgtype.UUID, scope TypingSettlement) {
if set.Typing != nil {
- set.Typing.OnSettled(ctx, sessionID)
+ set.Typing.OnSettled(ctx, sessionID, scope)
}
}
diff --git a/server/internal/integrations/channel/engine/router_test.go b/server/internal/integrations/channel/engine/router_test.go
index 5de2571f24a..a3750e9250c 100644
--- a/server/internal/integrations/channel/engine/router_test.go
+++ b/server/internal/integrations/channel/engine/router_test.go
@@ -216,20 +216,27 @@ func (f *fakeReplier) calls() []Result {
}
type fakeTyping struct {
- mu sync.Mutex
- count int
- settled int
+ settledScope TypingSettlement
+ sequence []string
+ messageID pgtype.UUID
+ mu sync.Mutex
+ count int
+ settled int
}
-func (f *fakeTyping) OnIngested(_ context.Context, _ ResolvedInstallation, _ channel.InboundMessage, _ pgtype.UUID) {
+func (f *fakeTyping) OnIngested(_ context.Context, _ ResolvedInstallation, _ channel.InboundMessage, _ pgtype.UUID, chatMessageID pgtype.UUID) {
f.mu.Lock()
defer f.mu.Unlock()
f.count++
+ f.messageID = chatMessageID
+ f.sequence = append(f.sequence, "ingested")
}
-func (f *fakeTyping) OnSettled(_ context.Context, _ pgtype.UUID) {
+func (f *fakeTyping) OnSettled(_ context.Context, _ pgtype.UUID, scope TypingSettlement) {
f.mu.Lock()
defer f.mu.Unlock()
f.settled++
+ f.settledScope = scope
+ f.sequence = append(f.sequence, "settled")
}
func (f *fakeTyping) calls() int { f.mu.Lock(); defer f.mu.Unlock(); return f.count }
func (f *fakeTyping) settledCalls() int { f.mu.Lock(); defer f.mu.Unlock(); return f.settled }
@@ -783,6 +790,12 @@ func TestRouter_Ingested_InTxMark_FinalizeNone(t *testing.T) {
if !waitFor(time.Second, func() bool { return h.typing.calls() == 1 }) {
t.Fatalf("ingest must show the typing indicator")
}
+ h.typing.mu.Lock()
+ messageID := h.typing.messageID
+ h.typing.mu.Unlock()
+ if messageID != h.binder.appendResult.MessageID {
+ t.Fatalf("typing lost persisted input identity: %v", messageID)
+ }
// Media resolution runs on its own goroutine (r.mediaWg), and the binding
// happens only after it returns, so both of these are downstream of work
// that Handle does not wait for. Reading them bare raced with that
@@ -1003,8 +1016,8 @@ func TestRouter_ContextGenerationsUseIndependentBatchWindows(t *testing.T) {
sessionID := h.binder.ensureID
initiator := h.ident.id.UserID
- h.router.scheduleRunWithFresh(h.router.sets[channel.TypeFeishu], h.inst.inst, msg, sessionID, initiator, pgtype.UUID{}, 1, false, 1)
- h.router.scheduleRunWithFresh(h.router.sets[channel.TypeFeishu], h.inst.inst, msg, sessionID, initiator, pgtype.UUID{}, 1, false, 2)
+ h.router.scheduleRunWithFresh(h.router.sets[channel.TypeFeishu], h.inst.inst, msg, sessionID, initiator, pgtype.UUID{}, 1, false, 1, pgtype.UUID{})
+ h.router.scheduleRunWithFresh(h.router.sets[channel.TypeFeishu], h.inst.inst, msg, sessionID, initiator, pgtype.UUID{}, 1, false, 2, pgtype.UUID{})
if got := h.router.batcher.pendingCount(); got != 2 {
t.Fatalf("pending generation windows = %d, want 2", got)
}
@@ -1086,7 +1099,8 @@ func TestRouter_RecoveryDoesNotDelayLiveOlderGeneration(t *testing.T) {
h.router.batcher = newTestBatcher(timers)
msg := p2pMessage(t)
h.router.scheduleRunWithFresh(h.router.sets[channel.TypeFeishu], h.inst.inst, msg,
- h.binder.ensureID, h.ident.id.UserID, pgtype.UUID{}, 1, false, 1)
+ h.binder.ensureID, h.ident.id.UserID, pgtype.UUID{}, 1, false, 1, pgtype.UUID{},
+ )
h.binder.appendResult.ContextRevision = 2
h.binder.appendResult.PendingContexts = []PendingContext{
@@ -1577,11 +1591,23 @@ func TestRouter_FlushOffline_RepliesAgentOffline(t *testing.T) {
}) {
t.Fatalf("agent-no-runtime must emit an AgentOffline reply")
}
- // The reaction was added on ingest but no task will run, so the bus-driven
- // clear never fires — the flush must clear the typing indicator itself.
+ // Inline failed enqueue precedes the detached ingestion hook. Its persisted
+ // settlement boundary must suppress that later Add without a task event.
if !waitFor(time.Second, func() bool { return h.typing.settledCalls() == 1 }) {
t.Fatalf("offline flush must clear the typing indicator, got %d OnSettled calls", h.typing.settledCalls())
}
+ if !waitFor(time.Second, func() bool { return h.typing.calls() == 1 }) {
+ t.Fatal("detached ingestion hook did not run")
+ }
+ h.typing.mu.Lock()
+ defer h.typing.mu.Unlock()
+ if h.typing.settledScope.ThroughMessageID != h.binder.appendResult.MessageID || h.typing.settledScope.WorkspaceID != h.inst.inst.WorkspaceID || h.typing.settledScope.InstallationID != h.inst.inst.ID {
+ t.Fatalf("failed flush lost its durable input boundary: %+v", h.typing.settledScope)
+ }
+ if len(h.typing.sequence) != 2 || h.typing.sequence[0] != "settled" || h.typing.sequence[1] != "ingested" {
+ t.Fatalf("unexpected inline flush/async Add ordering: %+v", h.typing.sequence)
+ }
+
}
func TestRouter_FlushArchived_ClearsTyping(t *testing.T) {
diff --git a/server/internal/integrations/dingtalk/ack.go b/server/internal/integrations/dingtalk/ack.go
index 56e23e4fbd9..a34e9a0e37c 100644
--- a/server/internal/integrations/dingtalk/ack.go
+++ b/server/internal/integrations/dingtalk/ack.go
@@ -57,7 +57,7 @@ func NewAckNotifier(client *Client, decrypt Decrypter, logger *slog.Logger, inpu
return &ackNotifier{client: client, decrypt: decrypt, logger: logger, inputs: inputs, active: make(map[string][]*ackState)}
}
-func (n *ackNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID) {
+func (n *ackNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID, chatMessageID pgtype.UUID) {
if !sessionID.Valid || msg.MessageID == "" || msg.Source.ChatID == "" {
return
}
@@ -96,7 +96,7 @@ func (n *ackNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstal
// session ID, so it cannot distinguish overlapping failed/pending generations.
// Task terminal events instead use onInputsSettled with their owned input IDs.
// This hook never marks input Done.
-func (n *ackNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID) {
+func (n *ackNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID, scope engine.TypingSettlement) {
key := util.UUIDToString(sessionID)
n.mu.Lock()
states := n.active[key]
diff --git a/server/internal/integrations/dingtalk/ack_archive_test.go b/server/internal/integrations/dingtalk/ack_archive_test.go
index afc008c323d..4d4f1d9ff45 100644
--- a/server/internal/integrations/dingtalk/ack_archive_test.go
+++ b/server/internal/integrations/dingtalk/ack_archive_test.go
@@ -21,7 +21,7 @@ func TestOutboundArchiveClearsPendingReceiptsOnlyForItsAgent(t *testing.T) {
sid := sessionUUID(92)
add := func(inst engine.ResolvedInstallation, name string, input byte, session byte) {
n.client.rememberReplySource(inst.ID, sessionUUID(input), sessionUUID(session), groupReactionMessage(name))
- n.OnIngested(context.Background(), inst, groupReactionMessage(name), sessionUUID(session))
+ n.OnIngested(context.Background(), inst, groupReactionMessage(name), sessionUUID(session), sessionUUID(session))
}
add(inst, "pending", 1, 92)
add(inst, "another-session", 2, 93)
@@ -35,7 +35,7 @@ func TestOutboundArchiveClearsPendingReceiptsOnlyForItsAgent(t *testing.T) {
bus.Publish(e)
// A repeated hook for an archived input stays retired, even after restore.
bus.Publish(events.Event{Type: protocol.EventAgentRestored, Payload: e.Payload})
- n.OnIngested(context.Background(), inst, groupReactionMessage("pending"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("pending"), sid, sid)
add(inst, "after-restore", 4, 92)
want := []string{"add:pending:" + emotionAcknowledged, "add:another-session:" + emotionAcknowledged, "add:other-agent:" + emotionAcknowledged}
if !slices.Equal((*actions)[:3], want) {
@@ -72,7 +72,10 @@ func TestOutboundArchiveRecallsAddThatFinishesAfterArchive(t *testing.T) {
mu.Unlock()
return nil
}
- go func() { defer close(done); n.OnIngested(context.Background(), inst, msg, sessionUUID(92)) }()
+ go func() {
+ defer close(done)
+ n.OnIngested(context.Background(), inst, msg, sessionUUID(92), sessionUUID(92))
+ }()
select {
case <-started:
case <-time.After(time.Second):
@@ -95,7 +98,7 @@ func TestOutboundArchiveRecallsAddThatFinishesAfterArchive(t *testing.T) {
func TestOutboundArchiveIgnoresInvalidPayload(t *testing.T) {
for _, payload := range []any{nil, make(chan int), map[string]any{"agent": "wrong type"}, map[string]any{"agent": map[string]any{"id": "not-a-uuid"}}} {
n, actions := newTestAckWithMessageIDs(time.Now)
- n.OnIngested(context.Background(), engine.ResolvedInstallation{AgentID: sessionUUID(80)}, groupReactionMessage("pending"), sessionUUID(92))
+ n.OnIngested(context.Background(), engine.ResolvedInstallation{AgentID: sessionUUID(80)}, groupReactionMessage("pending"), sessionUUID(92), sessionUUID(92))
NewOutbound(nil, nil, n.client, n, nil).handleAgentArchived(events.Event{Payload: payload})
if !slices.Equal(*actions, []string{"add:pending:" + emotionAcknowledged}) {
t.Fatalf("invalid archive cleared receipt: %v", *actions)
@@ -106,7 +109,7 @@ func TestOutboundArchiveIgnoresInvalidPayload(t *testing.T) {
func TestOutboundArchiveWithoutNotifierOrAgentIsNoOp(t *testing.T) {
NewOutbound(nil, nil, nil, nil, nil).handleAgentArchived(events.Event{})
n, actions := newTestAckWithMessageIDs(time.Now)
- n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage("pending"), sessionUUID(92))
+ n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage("pending"), sessionUUID(92), sessionUUID(92))
n.onAgentArchived(context.Background(), engine.ResolvedInstallation{}.AgentID)
if !slices.Equal(*actions, []string{"add:pending:" + emotionAcknowledged}) {
t.Fatalf("missing archive owner cleared receipt: %v", *actions)
diff --git a/server/internal/integrations/dingtalk/ack_batch_test.go b/server/internal/integrations/dingtalk/ack_batch_test.go
index ffcdf1b34b0..cd6a570b4a3 100644
--- a/server/internal/integrations/dingtalk/ack_batch_test.go
+++ b/server/internal/integrations/dingtalk/ack_batch_test.go
@@ -78,7 +78,7 @@ func TestAckBatchUsesSealedOwnershipAndInputOrder(t *testing.T) {
slices.Reverse(order)
}
for _, name := range order {
- n.OnIngested(ctx, inst, groupReactionMessage(name), sid)
+ n.OnIngested(ctx, inst, groupReactionMessage(name), sid, sid)
}
if !slices.Equal(actions, tc.want) {
t.Fatalf("actions=%v want=%v", actions, tc.want)
@@ -105,11 +105,11 @@ func TestAckBatchDoesNotMoveAcrossTaskBoundary(t *testing.T) {
q.rows[id] = db.ChatMessage{ID: id, ChatSessionID: sid, Role: "user", ChannelIngested: true, ChannelContextRevision: pgtype.Int8{Int64: 1, Valid: true}}
n.client.rememberReplySource(inst.ID, id, sid, groupReactionMessage(name))
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid, sid)
a := q.rows[sessionUUID(92)]
a.TaskID = sessionUUID(94)
q.rows[a.ID] = a
- n.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid, sid)
if len(visible) != 2 {
t.Fatalf("new batch removed old receipt: %v", visible)
}
@@ -132,7 +132,7 @@ func TestAckTerminalBeforeIngestDoesNotRecreateReceipt(t *testing.T) {
// Even the local source capture may follow a fast terminal event.
n.onInputsSettled(context.Background(), sid, []db.ChatMessage{row})
n.client.rememberReplySource(inst.ID, id, sid, groupReactionMessage("a"))
- n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid, sid)
if calls != 0 || len(n.active) != 0 {
t.Fatalf("late hook recreated receipt: calls=%d active=%v", calls, n.active)
}
@@ -156,7 +156,7 @@ func TestAckBatchRetriesWhenInputSealsDuringLookup(t *testing.T) {
q.rows[id] = db.ChatMessage{ID: id, ChatSessionID: sid, Role: "user", ChannelIngested: true}
n.client.rememberReplySource(inst.ID, id, sid, groupReactionMessage(name))
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid, sid)
q.read = func(id pgtype.UUID) {
if id == sessionUUID(92) {
q.read = nil
@@ -166,7 +166,7 @@ func TestAckBatchRetriesWhenInputSealsDuringLookup(t *testing.T) {
}
}
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid, sid)
if len(visible) != 1 || !visible["b"] {
t.Fatalf("sealing race left wrong receipts: %v", visible)
}
@@ -198,13 +198,16 @@ func TestAckBatchRecallsSupersededInFlightAdd(t *testing.T) {
}
return nil
}
- go func() { defer close(done); n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid) }()
+ go func() {
+ defer close(done)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid, sid)
+ }()
select {
case <-started:
case <-time.After(time.Second):
t.Fatal("add did not begin")
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid, sid)
close(release)
select {
case <-done:
@@ -227,7 +230,7 @@ func TestAckBatchQueryFailureSkipsReceipt(t *testing.T) {
t.Fatal("query failure guessed a batch")
return nil
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid, sid)
if len(n.active) != 0 {
t.Fatal("failed query retained receipt")
}
@@ -255,7 +258,7 @@ func TestAckBatchRetriesFailedRecallOnNextMessage(t *testing.T) {
id := sessionUUID(byte(92 + i))
q.rows[id] = db.ChatMessage{ID: id, ChatSessionID: sid, Role: "user", ChannelIngested: true}
n.client.rememberReplySource(inst.ID, id, sid, groupReactionMessage(name))
- n.OnIngested(context.Background(), inst, groupReactionMessage(name), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage(name), sid, sid)
if name == "b" && len(visible) != 2 {
t.Fatalf("fixture did not leave failed recall visible: %v", visible)
}
@@ -279,7 +282,7 @@ func TestAckTerminalDuringBatchLookupDoesNotRecreateReceipt(t *testing.T) {
t.Fatal("terminal event during lookup recreated receipt")
return nil
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid, sid)
}
func TestAckBatchRejectsUnattributableInput(t *testing.T) {
@@ -310,7 +313,7 @@ func TestAckBatchRejectsUnattributableInput(t *testing.T) {
t.Fatal("unattributable input received reaction")
return nil
}
- n.OnIngested(context.Background(), inst, msg, sid)
+ n.OnIngested(context.Background(), inst, msg, sid, sid)
})
}
}
@@ -381,7 +384,7 @@ func TestAckConcurrentBatchKeepsLastInput(t *testing.T) {
hooks.Add(1)
go func() {
defer hooks.Done()
- n.OnIngested(ctx, inst, msg, sid)
+ n.OnIngested(ctx, inst, msg, sid, sid)
}()
}
hooks.Wait()
diff --git a/server/internal/integrations/dingtalk/ack_failure_boundaries_test.go b/server/internal/integrations/dingtalk/ack_failure_boundaries_test.go
index 5aec5de483f..609e6b9db1c 100644
--- a/server/internal/integrations/dingtalk/ack_failure_boundaries_test.go
+++ b/server/internal/integrations/dingtalk/ack_failure_boundaries_test.go
@@ -31,7 +31,7 @@ func TestAckBatchLookupFailuresKeepExistingReceipt(t *testing.T) {
q.rows[id] = db.ChatMessage{ID: id, ChatSessionID: sid, ChannelIngested: true, Role: "user", CreatedAt: pgtype.Timestamptz{Time: time.Unix(int64(i), 0), Valid: true}}
n.client.rememberReplySource(inst.ID, id, sid, groupReactionMessage(name))
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("existing"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("existing"), sid, sid)
reads := 0
q.read = func(pgtype.UUID) {
reads++
@@ -39,13 +39,13 @@ func TestAckBatchLookupFailuresKeepExistingReceipt(t *testing.T) {
q.err = errors.New("database unavailable")
}
}
- n.OnIngested(context.Background(), inst, groupReactionMessage("new"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("new"), sid, sid)
if reads != failureRead || !slices.Equal(*actions, []string{"add:existing:" + emotionAcknowledged}) {
t.Fatalf("failed lookup moved receipt: reads=%d actions=%v", reads, *actions)
}
// A failed optional lookup must not poison the next accepted hook.
q.read, q.err = nil, nil
- n.OnIngested(context.Background(), inst, groupReactionMessage("new"), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage("new"), sid, sid)
want := []string{"add:existing:" + emotionAcknowledged, "recall:existing:" + emotionAcknowledged, "add:new:" + emotionAcknowledged}
if !slices.Equal(*actions, want) {
t.Fatalf("recovered lookup: actions=%v want=%v", *actions, want)
@@ -82,7 +82,7 @@ func TestOutboundTerminalOwnerFailureKeepsReceipts(t *testing.T) {
inst := engine.ResolvedInstallation{ID: sessionUUID(90)}
for i, name := range []string{"existing", "new"} {
n.client.rememberReplySource(inst.ID, sessionUUID(byte(92+i)), sid, groupReactionMessage(name))
- n.OnIngested(context.Background(), inst, groupReactionMessage(name), sid)
+ n.OnIngested(context.Background(), inst, groupReactionMessage(name), sid, sid)
}
before := append([]string(nil), (*actions)...)
q := &terminalOwnerQueries{task: tc.task, taskErr: tc.err}
diff --git a/server/internal/integrations/dingtalk/ack_test.go b/server/internal/integrations/dingtalk/ack_test.go
index 2cfb0b9b07d..7ae4e11955f 100644
--- a/server/internal/integrations/dingtalk/ack_test.go
+++ b/server/internal/integrations/dingtalk/ack_test.go
@@ -52,10 +52,10 @@ func TestAckNotifierClearsSessionWithoutMarkingOtherInputsDone(t *testing.T) {
ctx := context.Background()
sid := sessionUUID(1)
inst := engine.ResolvedInstallation{ID: sessionUUID(9)}
- n.OnIngested(ctx, inst, groupReactionMessage("a"), sid)
- n.OnIngested(ctx, inst, groupReactionMessage("b"), sid)
- n.OnIngested(ctx, inst, groupReactionMessage("other"), sessionUUID(2))
- n.OnSettled(ctx, sid)
+ n.OnIngested(ctx, inst, groupReactionMessage("a"), sid, sid)
+ n.OnIngested(ctx, inst, groupReactionMessage("b"), sid, sid)
+ n.OnIngested(ctx, inst, groupReactionMessage("other"), sessionUUID(2), sessionUUID(2))
+ n.OnSettled(ctx, sid, engine.TypingSettlement{})
n.client.rememberReplySource(inst.ID, sessionUUID(10), sid, groupReactionMessage("a"))
n.OnReplyDelivered(ctx, inst, sessionUUID(10))
want := []string{"add:a:收到", "add:b:收到", "add:other:收到", "recall:a:收到", "recall:b:收到", "add:a:Done"}
@@ -67,8 +67,8 @@ func TestAckNotifierDuplicateIngestDoesNotDuplicateReaction(t *testing.T) {
n, actions := newTestAck(time.Now)
ctx := context.Background()
sid := sessionUUID(1)
- n.OnIngested(ctx, engine.ResolvedInstallation{}, groupReactionMessage("a"), sid)
- n.OnIngested(ctx, engine.ResolvedInstallation{}, groupReactionMessage("a"), sid)
+ n.OnIngested(ctx, engine.ResolvedInstallation{}, groupReactionMessage("a"), sid, sid)
+ n.OnIngested(ctx, engine.ResolvedInstallation{}, groupReactionMessage("a"), sid, sid)
if len(*actions) != 1 {
t.Fatalf("duplicate reactions: %v", *actions)
}
@@ -85,10 +85,10 @@ func TestAckNotifierFailedAddStillAttemptsBoundedRecall(t *testing.T) {
}
ctx, cancel := context.WithCancel(context.Background())
sid := sessionUUID(1)
- n.OnIngested(ctx, engine.ResolvedInstallation{}, groupReactionMessage("a"), sid)
+ n.OnIngested(ctx, engine.ResolvedInstallation{}, groupReactionMessage("a"), sid, sid)
cancel()
- n.OnSettled(ctx, sid)
- n.OnSettled(context.Background(), sid)
+ n.OnSettled(ctx, sid, engine.TypingSettlement{})
+ n.OnSettled(context.Background(), sid, engine.TypingSettlement{})
if calls != 2 || len(n.active) != 0 {
t.Fatalf("calls=%d active=%d", calls, len(n.active))
}
@@ -117,14 +117,14 @@ func TestAckNotifierRecallsAddThatFinishesAfterClear(t *testing.T) {
sid := sessionUUID(1)
go func() {
defer close(done)
- n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage("a"), sid)
+ n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage("a"), sid, sid)
}()
select {
case <-started:
case <-time.After(time.Second):
t.Fatal("add did not start")
}
- n.OnSettled(context.Background(), sid)
+ n.OnSettled(context.Background(), sid, engine.TypingSettlement{})
close(release)
select {
case <-done:
@@ -139,9 +139,9 @@ func TestAckNotifierRecallsAddThatFinishesAfterClear(t *testing.T) {
}
func TestAckNotifierInvalidCoordinatesDoNothing(t *testing.T) {
n, actions := newTestAck(time.Now)
- n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage("a"), pgtype.UUID{})
- n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage(""), sessionUUID(1))
- n.OnIngested(context.Background(), engine.ResolvedInstallation{}, channel.InboundMessage{MessageID: "a"}, sessionUUID(1))
+ n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage("a"), pgtype.UUID{}, pgtype.UUID{})
+ n.OnIngested(context.Background(), engine.ResolvedInstallation{}, groupReactionMessage(""), sessionUUID(1), sessionUUID(1))
+ n.OnIngested(context.Background(), engine.ResolvedInstallation{}, channel.InboundMessage{MessageID: "a"}, sessionUUID(1), sessionUUID(1))
n.OnReplyDelivered(context.Background(), engine.ResolvedInstallation{}, pgtype.UUID{})
if len(*actions) != 0 {
t.Fatalf("invalid coordinates sent: %v", *actions)
diff --git a/server/internal/integrations/dingtalk/outbound_db_test.go b/server/internal/integrations/dingtalk/outbound_db_test.go
index d76ef4ba110..1a35c847334 100644
--- a/server/internal/integrations/dingtalk/outbound_db_test.go
+++ b/server/internal/integrations/dingtalk/outbound_db_test.go
@@ -115,7 +115,7 @@ func testOutboundSealedInput(t *testing.T, scenario string, restart bool) {
if err != nil {
t.Fatal(err)
}
- ack.OnIngested(ctx, inst, first, sid)
+ ack.OnIngested(ctx, inst, first, sid, sid)
const firstBody = "> **Quoted author:**\n>\n> [quoted content unavailable]\n\nfirst question"
var persisted string
if err := pool.QueryRow(ctx, "SELECT content FROM chat_message WHERE id = $1", appended.MessageID).Scan(&persisted); err != nil || persisted != firstBody {
@@ -147,7 +147,7 @@ func testOutboundSealedInput(t *testing.T, scenario string, restart bool) {
if _, err := set.Session.AppendMessage(ctx, engine.AppendParams{SessionID: sid, Sender: userID, InstallationID: inst.ID, Message: second}); err != nil {
t.Fatal(err)
}
- ack.OnIngested(ctx, inst, second, sid)
+ ack.OnIngested(ctx, inst, second, sid, sid)
wantReceipts := 2
if scenario == "merged" {
wantReceipts = 1
diff --git a/server/internal/integrations/dingtalk/outbound_test.go b/server/internal/integrations/dingtalk/outbound_test.go
index 7691abf590c..7f8df26a39d 100644
--- a/server/internal/integrations/dingtalk/outbound_test.go
+++ b/server/internal/integrations/dingtalk/outbound_test.go
@@ -217,7 +217,7 @@ func TestOutboundTerminalReactionLifecycle(t *testing.T) {
actions = append(actions, verb+msg.MessageID+":"+name)
return nil
}
- ack.OnIngested(context.Background(), engine.ResolvedInstallation{ID: q.installation.ID}, groupReactionMessage("source"), sid)
+ ack.OnIngested(context.Background(), engine.ResolvedInstallation{ID: q.installation.ID}, groupReactionMessage("source"), sid, sid)
o := NewOutbound(q, nil, client, ack, nil)
err = o.processEvent(ctx, events.Event{Type: tc.eventType, TaskID: util.UUIDToString(taskID), ChatSessionID: util.UUIDToString(sid), Payload: tc.payload})
if (err != nil) != tc.sendFails {
@@ -252,9 +252,9 @@ func TestOutboundTerminalWithoutDeliverySettlesOnlyOwnedInput(t *testing.T) {
for i, name := range []string{"a", "b", "other"} {
ack.client.rememberReplySource(inst.ID, sessionUUID(byte(70+i)), sid, groupReactionMessage(name))
}
- ack.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid)
- ack.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid)
- ack.OnIngested(context.Background(), inst, groupReactionMessage("other"), sessionUUID(62))
+ ack.OnIngested(context.Background(), inst, groupReactionMessage("a"), sid, sid)
+ ack.OnIngested(context.Background(), inst, groupReactionMessage("b"), sid, sid)
+ ack.OnIngested(context.Background(), inst, groupReactionMessage("other"), sessionUUID(62), sessionUUID(62))
q := missingDeliveryInputQueries{deliveryOnlyOutboundQueries{
task: db.AgentTaskQueue{ChatInputTaskID: sessionUUID(63)},
input: []db.ChatMessage{{ID: sessionUUID(70), ChannelIngested: true}},
@@ -297,7 +297,7 @@ func TestOutboundPartialReplyFailureDoesNotMarkDone(t *testing.T) {
}
ack, actions := newTestAckWithMessageIDs(time.Now)
ack.client.rememberReplySource(q.installation.ID, q.input[0].ID, sid, groupReactionMessage("source"))
- ack.OnIngested(context.Background(), engine.ResolvedInstallation{ID: q.installation.ID}, groupReactionMessage("source"), sid)
+ ack.OnIngested(context.Background(), engine.ResolvedInstallation{ID: q.installation.ID}, groupReactionMessage("source"), sid, sid)
err = NewOutbound(q, nil, client, ack, nil).processEvent(context.Background(), events.Event{
Type: protocol.EventChatDone, TaskID: util.UUIDToString(tid), ChatSessionID: util.UUIDToString(sid), Payload: protocol.ChatDonePayload{Content: strings.Repeat("answer\n", 6000)},
})
diff --git a/server/internal/integrations/dingtalk/reaction_interest_test.go b/server/internal/integrations/dingtalk/reaction_interest_test.go
index ff43561edfd..d6deca1733a 100644
--- a/server/internal/integrations/dingtalk/reaction_interest_test.go
+++ b/server/internal/integrations/dingtalk/reaction_interest_test.go
@@ -131,7 +131,7 @@ func TestTerminalBeforeSourceCaptureFencesLateReaction(t *testing.T) {
t.Fatal("late callback recreated terminal reaction")
return nil
}
- n.OnIngested(context.Background(), inst, msg, sid)
+ n.OnIngested(context.Background(), inst, msg, sid, sid)
if q.taskReads != 1 || q.inputReads != 1 || len(n.active) != 0 {
t.Fatalf("terminal fence not exercised: task=%d input=%d active=%v", q.taskReads, q.inputReads, n.active)
}
@@ -246,7 +246,7 @@ func TestActiveReceiptRemainsDiscoverableAfterSourceEviction(t *testing.T) {
return nil
}
n.client.rememberReplySource(inst.ID, id, sid, msg)
- n.OnIngested(context.Background(), inst, msg, sid)
+ n.OnIngested(context.Background(), inst, msg, sid, sid)
for i := 0; i < maxReplySources; i++ {
n.client.rememberReplySource(inst.ID, dbid.NewV7(), sessionUUID(54), groupReactionMessage("other"))
}
@@ -290,7 +290,7 @@ func TestIngestKeepsTerminalInterestAcrossSourceEviction(t *testing.T) {
t.Fatal("eviction allowed terminal reaction to reappear")
return nil
}
- n.OnIngested(context.Background(), inst, msg, sid)
+ n.OnIngested(context.Background(), inst, msg, sid, sid)
if q.inputReads != 1 || len(n.active) != 0 || n.client.hasReplySession(sid) {
t.Fatal("in-flight interest was lost or leaked")
}
diff --git a/server/internal/integrations/dingtalk/reply_boundaries_test.go b/server/internal/integrations/dingtalk/reply_boundaries_test.go
index 695ccdbb025..85ae8291415 100644
--- a/server/internal/integrations/dingtalk/reply_boundaries_test.go
+++ b/server/internal/integrations/dingtalk/reply_boundaries_test.go
@@ -26,8 +26,8 @@ func TestAckNotifierCredentialFailuresAreOptional(t *testing.T) {
t.Run(tc.name, func(t *testing.T) {
d := newDingtalkSendServer(t)
n := NewAckNotifier(NewClient(nil, d.srv.URL), tc.decrypt, nil, nil)
- n.OnIngested(context.Background(), engine.ResolvedInstallation{Platform: tc.platform}, groupReactionMessage("source"), sessionUUID(91))
- n.OnSettled(context.Background(), sessionUUID(91))
+ n.OnIngested(context.Background(), engine.ResolvedInstallation{Platform: tc.platform}, groupReactionMessage("source"), sessionUUID(91), sessionUUID(91))
+ n.OnSettled(context.Background(), sessionUUID(91), engine.TypingSettlement{})
if len(d.sendBodies) != 0 || len(n.active) != 0 {
t.Fatal("unusable credentials sent a reaction or prevented local cleanup")
}
diff --git a/server/internal/integrations/lark/client.go b/server/internal/integrations/lark/client.go
index 8d53f984df3..df42a9c3a1b 100644
--- a/server/internal/integrations/lark/client.go
+++ b/server/internal/integrations/lark/client.go
@@ -145,6 +145,20 @@ type TokenCacheInvalidator interface {
InvalidateTokenCache(appID string)
}
+// ReactionLister is implemented by an APIClient that can list the reactions
+// already sitting on a message (GET /im/v1/messages/{message_id}/reactions).
+// The typing-indicator sweep uses it to find and delete the bot's own Typing
+// reactions without consulting any in-process state, which is what keeps the
+// badge removable after a restart or on another replica.
+//
+// It is deliberately separate from APIClient rather than a method on it,
+// mirroring TokenCacheInvalidator: only clients that can read Lark implement
+// it, and fakes that only exercise the add/delete pair are not forced to grow
+// list plumbing. Callers type-assert and skip the sweep when it is absent.
+type ReactionLister interface {
+ ListMessageReactions(ctx context.Context, p ListMessageReactionsParams) ([]MessageReaction, error)
+}
+
// ListMessagesParams selects a bounded, recent window of messages in a
// single Lark chat for the group-context prefetch. Only the fields the
// enricher needs today are exposed (ChatID, ThreadID, PageSize, EndTime);
@@ -344,6 +358,26 @@ type DeleteReactionParams struct {
ReactionID string
}
+// ListMessageReactionsParams is the input shape for listing the reactions
+// already on a message. EmojiType filters server-side (Lark's
+// reaction_type query parameter); empty lists every reaction.
+type ListMessageReactionsParams struct {
+ InstallationID InstallationCredentials
+ MessageID string
+ EmojiType string
+}
+
+// MessageReaction is one reaction on a message as Lark lists it. OperatorType
+// is "app" when the installed bot added it and "user" when a human did — the
+// distinction that lets the typing-indicator sweep delete only the badges the
+// bot itself put on.
+type MessageReaction struct {
+ ReactionID string
+ OperatorType string
+ OperatorID string
+ EmojiType string
+}
+
// InstallationCredentials is the per-installation transport context the
// client needs to authenticate against Lark on behalf of a workspace's
// bot. Passing these explicitly to each call (rather than constructing
diff --git a/server/internal/integrations/lark/feishu_resolvers.go b/server/internal/integrations/lark/feishu_resolvers.go
index 6bd001ba5f8..03a3043b5cc 100644
--- a/server/internal/integrations/lark/feishu_resolvers.go
+++ b/server/internal/integrations/lark/feishu_resolvers.go
@@ -320,19 +320,19 @@ func dispatchResultFromEngine(res engine.Result) DispatchResult {
type feishuTypingNotifier struct{ mgr *TypingIndicatorManager }
-func (r *feishuTypingNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID) {
+func (r *feishuTypingNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID, chatMessageID pgtype.UUID) {
larkInst, ok := inst.Platform.(Installation)
if !ok {
return
}
lm, _ := larkMsgFromRaw(msg)
- r.mgr.Add(ctx, larkInst, sessionID, msg.MessageID, lm.CreateTime)
+ r.mgr.Add(ctx, larkInst, sessionID, msg.MessageID, lm.CreateTime, chatMessageID)
}
// OnSettled clears the reaction when the run trigger enqueued no task (agent
// offline / archived, or an enqueue failure) — the Patcher's bus-driven clear on
// chat-done / task-failed never fires for those, so without this the Typing
// reaction sticks.
-func (r *feishuTypingNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID) {
- r.mgr.Clear(ctx, sessionID)
+func (r *feishuTypingNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID, scope engine.TypingSettlement) {
+ r.mgr.Settle(ctx, sessionID, scope)
}
diff --git a/server/internal/integrations/lark/http_client.go b/server/internal/integrations/lark/http_client.go
index 12df677269d..ece913366c5 100644
--- a/server/internal/integrations/lark/http_client.go
+++ b/server/internal/integrations/lark/http_client.go
@@ -1002,6 +1002,81 @@ func (c *httpAPIClient) DeleteMessageReaction(ctx context.Context, p DeleteReact
return nil
}
+// larkListReactionsMaxPageSize is Lark's page-size cap for the message
+// reaction list endpoint.
+const larkListReactionsMaxPageSize = 50
+
+// larkListReactionsMaxPages bounds the pagination loop. A Typing badge is a
+// handful of reactions at most; the bound only exists so a misbehaving
+// has_more cannot turn the sweep into an unbounded poll loop on the
+// synchronous bus-delivery path.
+const larkListReactionsMaxPages = 10
+
+// ListMessageReactions lists the reactions on a message via
+// GET /open-apis/im/v1/messages/{message_id}/reactions, optionally filtered
+// server-side by emoji type (Lark's reaction_type query parameter). All pages
+// are followed (bounded by larkListReactionsMaxPages) so the caller sees every
+// reaction, not just the first window.
+func (c *httpAPIClient) ListMessageReactions(ctx context.Context, p ListMessageReactionsParams) ([]MessageReaction, error) {
+ if p.MessageID == "" {
+ return nil, errors.New("lark http client: missing message_id")
+ }
+ base := "/open-apis/im/v1/messages/" + url.PathEscape(p.MessageID) + "/reactions"
+
+ var out []MessageReaction
+ pageToken := ""
+ for page := 0; page < larkListReactionsMaxPages; page++ {
+ q := url.Values{}
+ q.Set("page_size", strconv.Itoa(larkListReactionsMaxPageSize))
+ if p.EmojiType != "" {
+ q.Set("reaction_type", p.EmojiType)
+ }
+ if pageToken != "" {
+ q.Set("page_token", pageToken)
+ }
+ var resp struct {
+ Code int `json:"code"`
+ Msg string `json:"msg"`
+ Data struct {
+ Items []struct {
+ ReactionID string `json:"reaction_id"`
+ Operator struct {
+ OperatorType string `json:"operator_type"`
+ OperatorID string `json:"operator_id"`
+ } `json:"operator"`
+ ReactionType struct {
+ EmojiType string `json:"emoji_type"`
+ } `json:"reaction_type"`
+ } `json:"items"`
+ HasMore bool `json:"has_more"`
+ PageToken string `json:"page_token"`
+ } `json:"data"`
+ }
+ if err := c.doAuthedJSON(ctx, p.InstallationID, http.MethodGet, base+"?"+q.Encode(), nil, &resp); err != nil {
+ return nil, fmt.Errorf("lark http client: list message reactions: %w", err)
+ }
+ if resp.Code != 0 {
+ if isTokenError(resp.Code) {
+ c.invalidateToken(p.InstallationID.AppID)
+ }
+ return nil, fmt.Errorf("lark http client: list message reactions: code=%d msg=%q", resp.Code, resp.Msg)
+ }
+ for _, it := range resp.Data.Items {
+ out = append(out, MessageReaction{
+ ReactionID: it.ReactionID,
+ OperatorType: it.Operator.OperatorType,
+ OperatorID: it.Operator.OperatorID,
+ EmojiType: it.ReactionType.EmojiType,
+ })
+ }
+ if !resp.Data.HasMore || resp.Data.PageToken == "" {
+ return out, nil
+ }
+ pageToken = resp.Data.PageToken
+ }
+ return out, nil
+}
+
// BatchGetUsers resolves user open_ids to display names via
// GET /open-apis/contact/v3/users/batch?user_ids=…&user_id_type=open_id.
// It mirrors fetchBotUnionID's single-user contact lookup, batched. Only
diff --git a/server/internal/integrations/lark/http_client_test.go b/server/internal/integrations/lark/http_client_test.go
index e94335f8d01..9201efb58f0 100644
--- a/server/internal/integrations/lark/http_client_test.go
+++ b/server/internal/integrations/lark/http_client_test.go
@@ -1528,6 +1528,106 @@ func TestBindingPromptTemplate_Shape(t *testing.T) {
}
}
+// ListMessageReactions backs the typing-indicator sweep: it must filter
+// server-side by emoji type, follow every page Lark offers (a badge beyond the
+// first window must still be found), and carry each item's operator and emoji
+// fields through so the caller can tell the bot's own reactions from a human's.
+func TestHTTPClient_ListMessageReactions_FollowsPages(t *testing.T) {
+ fake := newLarkFake(t)
+ fake.stubToken("tok", 7200)
+
+ requests := 0
+ fake.mux.HandleFunc("/open-apis/im/v1/messages/", func(w http.ResponseWriter, r *http.Request) {
+ requests++
+ if r.Method != http.MethodGet {
+ t.Errorf("list reactions: want GET, got %s", r.Method)
+ }
+ if !strings.HasSuffix(r.URL.Path, "/om_trigger/reactions") {
+ t.Errorf("list reactions: unexpected path %q", r.URL.Path)
+ }
+ q := r.URL.Query()
+ if got := q.Get("reaction_type"); got != "Typing" {
+ t.Errorf("list reactions: reaction_type = %q, want Typing", got)
+ }
+ if got := q.Get("page_size"); got != "50" {
+ t.Errorf("list reactions: page_size = %q, want 50", got)
+ }
+ switch q.Get("page_token") {
+ case "":
+ writeJSON(w, map[string]any{
+ "code": 0,
+ "msg": "ok",
+ "data": map[string]any{
+ "items": []map[string]any{{
+ "reaction_id": "r-page1",
+ "operator": map[string]any{"operator_type": "app", "operator_id": "cli_app_xx"},
+ "reaction_type": map[string]any{"emoji_type": "Typing"},
+ }},
+ "has_more": true,
+ "page_token": "cursor-2",
+ },
+ })
+ case "cursor-2":
+ writeJSON(w, map[string]any{
+ "code": 0,
+ "msg": "ok",
+ "data": map[string]any{
+ "items": []map[string]any{{
+ "reaction_id": "r-page2",
+ "operator": map[string]any{"operator_type": "user", "operator_id": "ou_human"},
+ "reaction_type": map[string]any{"emoji_type": "Typing"},
+ }},
+ "has_more": false,
+ },
+ })
+ default:
+ t.Errorf("list reactions: unexpected page_token %q", q.Get("page_token"))
+ }
+ })
+
+ c := newTestClient(fake, time.Now)
+ got, err := c.ListMessageReactions(context.Background(), ListMessageReactionsParams{
+ InstallationID: testCreds(),
+ MessageID: "om_trigger",
+ EmojiType: "Typing",
+ })
+ if err != nil {
+ t.Fatalf("list message reactions: %v", err)
+ }
+ if requests != 2 {
+ t.Errorf("expected both pages to be fetched, got %d requests", requests)
+ }
+ if len(got) != 2 || got[0].ReactionID != "r-page1" || got[1].ReactionID != "r-page2" {
+ t.Fatalf("unexpected reactions: %+v", got)
+ }
+ if got[0].OperatorType != "app" || got[0].OperatorID != "cli_app_xx" || got[0].EmojiType != "Typing" {
+ t.Errorf("page-1 item fields not carried through: %+v", got[0])
+ }
+ if got[1].OperatorType != "user" {
+ t.Errorf("page-2 operator_type not carried through: %+v", got[1])
+ }
+}
+
+func TestHTTPClient_ListMessageReactions_BusinessError(t *testing.T) {
+ fake := newLarkFake(t)
+ fake.stubToken("tok", 7200)
+ fake.mux.HandleFunc("/open-apis/im/v1/messages/", func(w http.ResponseWriter, r *http.Request) {
+ writeJSON(w, map[string]any{"code": 230013, "msg": "invalid message id"})
+ })
+
+ c := newTestClient(fake, time.Now)
+ _, err := c.ListMessageReactions(context.Background(), ListMessageReactionsParams{
+ InstallationID: testCreds(),
+ MessageID: "om_bad",
+ })
+ if err == nil {
+ t.Fatal("expected error on non-zero Lark code")
+ }
+ if !strings.Contains(err.Error(), "230013") {
+ t.Errorf("error should surface the Lark code; got %v", err)
+ }
+}
+
// fakeClock is a minimal monotonic clock for tests that need to drive
// the cache TTL deterministically.
type fakeClock struct{ now time.Time }
diff --git a/server/internal/integrations/lark/outbound.go b/server/internal/integrations/lark/outbound.go
index 51944359499..ea42cc8aedc 100644
--- a/server/internal/integrations/lark/outbound.go
+++ b/server/internal/integrations/lark/outbound.go
@@ -12,6 +12,7 @@ import (
"github.com/jackc/pgx/v5/pgtype"
"github.com/multica-ai/multica/server/internal/events"
"github.com/multica-ai/multica/server/internal/integrations/channel/engine"
+ "github.com/multica-ai/multica/server/internal/util"
db "github.com/multica-ai/multica/server/pkg/db/generated"
"github.com/multica-ai/multica/server/pkg/protocol"
)
@@ -266,12 +267,8 @@ func (p *Patcher) SetTypingIndicatorManager(m *TypingIndicatorManager) {
// event with no workspace is dropped before it reaches the bus. It now
// takes the workspace from its caller. Archiving an agent also publishes
// task:cancelled for chat tasks after commit, clearing their reactions
-// through this subscription. An ending during Add can still clear
-// nothing, because Add records its state only after the Lark call
-// returns, so the badge lands after the clear with nothing left to
-// take it off. This race predates task:cancelled — chat-done
-// and task-failed race the add the same way — and closing it needs a
-// per-session generation the add can check when its call returns.
+// through this subscription. Clear marks in-flight adds as ended so a
+// late Add removes its own reaction when the HTTP call returns.
//
// We deliberately do NOT subscribe to EventTaskQueued / EventTaskRunning
// (no thinking-card lifecycle anymore — adds noise without value) or to
@@ -311,37 +308,24 @@ func (p *Patcher) processEvent(ctx context.Context, e events.Event) error {
// Issue / autopilot tasks have no chat_session.
return nil
}
- // A cancelled run has no reply to place, so the only thing owed to the user
- // is taking the Typing badge off. That runs before every lookup below,
- // because each of them can answer "no" for a run that still has a badge on
- // screen:
- //
- // - the binding is gone by the time a session delete's cancels are
- // broadcast (they fire after the transaction that dropped it commits);
- //
- // - the origin classification answers "does this answer belong on Lark",
- // and a task cancelled for owning an empty input batch — the failure
- // #6611 fixed the cause of — reports no channel-ingested messages, so a
- // clear behind it would be skipped on exactly the run that most needs
- // it. A cancellation has no answer to misroute, so the question does not
- // arise.
- //
- // Nothing is posted here, so neither gate is protecting anything: the badge
- // is Lark's own, and Clear only touches sessions this process put one on.
- // The clear is keyed by session rather than by turn, so cancelling one of
- // two turns in a session takes the badge off both; the worst that costs is a
- // missing badge on a turn still running.
+ // Cleanup runs after reply delivery (including failed sends and early
+ // returns), with its own budget. A slow reaction API cannot consume the
+ // reply's context deadline. The bus still waits for bounded cleanup.
+ if p.typingIndicator != nil {
+ defer func() {
+ cleanupCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ defer cancel()
+ covered := p.typingIndicator.Reconcile(cleanupCtx, chatSessionID)
+ p.sweepTypingForTask(cleanupCtx, taskID, e.ChannelReactionTarget, covered)
+ }()
+ }
if e.Type == protocol.EventTaskCancelled {
- if p.typingIndicator != nil {
- p.typingIndicator.Clear(ctx, chatSessionID)
- }
return nil
}
delivery, err := p.queries.GetChannelTaskDelivery(ctx, taskID)
if err != nil {
if errors.Is(err, pgx.ErrNoRows) {
- // Direct Multica task or violated snapshot invariant — fail closed.
return nil
}
return fmt.Errorf("lookup lark task delivery: %w", err)
@@ -393,13 +377,6 @@ func (p *Patcher) processEvent(ctx context.Context, e events.Event) error {
agentName = agent.Name
}
- // Clear the "processing" reaction before the reply is visible so the
- // user sees a clean transition. Best-effort: a failure here is logged
- // but does not block the actual reply.
- if p.typingIndicator != nil {
- p.typingIndicator.Clear(ctx, chatSessionID)
- }
-
switch e.Type {
case protocol.EventChatDone:
return p.sendChatReply(ctx, creds, binding, mentionOpenID(binding), e.Payload)
@@ -409,6 +386,55 @@ func (p *Patcher) processEvent(ctx context.Context, e events.Event) error {
return nil
}
+// sweepTypingForTask uses the frozen trigger message. Session deletion carries
+// that anchor on the internal event before deleting the delivery row.
+func (p *Patcher) sweepTypingForTask(ctx context.Context, taskID pgtype.UUID, target *events.ChannelReactionTarget, covered map[string]bool) {
+ delivery, err := p.queries.GetChannelTaskDelivery(ctx, taskID)
+ if errors.Is(err, pgx.ErrNoRows) && target != nil {
+ installationID, parseErr := util.ParseUUID(target.InstallationID)
+ if parseErr != nil {
+ return
+ }
+ delivery = db.ChannelTaskDelivery{
+ ChannelType: target.ChannelType, InstallationID: installationID,
+ ChannelMessageID: pgtype.Text{String: target.MessageID, Valid: target.MessageID != ""},
+ }
+ err = nil
+ }
+ if err != nil {
+ if !errors.Is(err, pgx.ErrNoRows) {
+ p.cfg.Logger.Warn("lark patcher: typing sweep delivery lookup failed",
+ "task_id", uuidString(taskID),
+ "error", err,
+ )
+ }
+ return
+ }
+ if delivery.ChannelType != channelTypeFeishu || !delivery.ChannelMessageID.Valid || delivery.ChannelMessageID.String == "" {
+ return
+ }
+ if covered[uuidString(delivery.InstallationID)+"/"+delivery.ChannelMessageID.String] {
+ return
+ }
+ inst, err := p.queries.GetLarkInstallation(ctx, delivery.InstallationID)
+ if err != nil {
+ p.cfg.Logger.Warn("lark patcher: typing sweep installation lookup failed",
+ "task_id", uuidString(taskID),
+ "error", err,
+ )
+ return
+ }
+ creds, err := p.installationCredentials(inst)
+ if err != nil {
+ p.cfg.Logger.Warn("lark patcher: typing sweep credentials resolution failed",
+ "task_id", uuidString(taskID),
+ "error", err,
+ )
+ return
+ }
+ p.typingIndicator.SweepMessage(ctx, creds, delivery.ChannelMessageID.String)
+}
+
// mentionOpenID returns the Feishu open_id to @-mention on this reply, or ""
// for "send without a mention".
//
diff --git a/server/internal/integrations/lark/outbound_test.go b/server/internal/integrations/lark/outbound_test.go
index de445fdebd4..bce66a6da46 100644
--- a/server/internal/integrations/lark/outbound_test.go
+++ b/server/internal/integrations/lark/outbound_test.go
@@ -866,7 +866,7 @@ func TestPatcherClearsTypingOnTaskCancelled(t *testing.T) {
bus := events.New()
p.Register(bus)
- typing.Add(context.Background(), q.installation, q.binding.ChatSessionID, "om_trigger", "")
+ typing.Add(context.Background(), q.installation, q.binding.ChatSessionID, "om_trigger", "", q.binding.ChatSessionID)
if len(typingAPI.addCalled) != 1 {
t.Fatalf("setup: expected the Typing reaction to be added, got %d", len(typingAPI.addCalled))
}
@@ -928,7 +928,7 @@ func TestPatcherClearsTypingAfterSessionDeleteRemovedTheBinding(t *testing.T) {
p.Register(bus)
sessionID := q.binding.ChatSessionID
- typing.Add(context.Background(), q.installation, sessionID, "om_trigger", "")
+ typing.Add(context.Background(), q.installation, sessionID, "om_trigger", "", sessionID)
if len(typingAPI.addCalled) != 1 {
t.Fatalf("setup: expected the Typing reaction to be added, got %d", len(typingAPI.addCalled))
}
@@ -968,6 +968,137 @@ func TestPatcherClearsTypingAfterSessionDeleteRemovedTheBinding(t *testing.T) {
}
}
+// ---- the stateless sweep: restart- and replica-proof clearing ----
+
+// The in-memory state lives and dies with the process that ran Add. A restart
+// between the question and the answer — or a second replica handling the
+// completion event — empties the map while the badge is still on screen, and
+// the reply that follows clears nothing. The sweep closes that hole: it finds
+// the Typing reaction through Lark alone, on the trigger message the delivery
+// row froze, so the reply landing is enough to take the badge off.
+func TestPatcherSweepsTypingOnChatDoneAfterRestartEmptiedTheState(t *testing.T) {
+ p, q, api := newTestPatcher(t)
+ q.binding.LastMessageID = pgtype.Text{String: "om_trigger", Valid: true}
+ taskID := uuidFromString(t, "ee777777-ee77-ee77-ee77-eeeeeeeeeeee")
+ q.task = db.AgentTaskQueue{ChatInputTaskID: taskID}
+ q.taskChannelIngested = true
+
+ // A fresh manager: no Add ever ran in this process, the map is empty.
+ typingAPI := &fakeTypingAPIClient{
+ listReturn: []MessageReaction{
+ {ReactionID: "r-stale", OperatorType: "app", OperatorID: "cli_test_app", EmojiType: typingEmoji},
+ },
+ }
+ typing := NewTypingIndicatorManager(typingAPI, fakeTypingCreds{secret: "shh"},
+ &fakeTypingQueries{binding: q.binding, installation: q.installation}, newDiscardLogger())
+ p.SetTypingIndicatorManager(typing)
+
+ p.handleEvent(events.Event{
+ Type: protocol.EventChatDone,
+ TaskID: uuidString(taskID),
+ ChatSessionID: uuidString(q.binding.ChatSessionID),
+ Payload: protocol.ChatDonePayload{Content: "the reply arrived"},
+ })
+
+ api.mu.Lock()
+ defer api.mu.Unlock()
+ if len(api.textSent) != 1 {
+ t.Fatalf("the reply itself must still go out; textSent=%d", len(api.textSent))
+ }
+ if len(typingAPI.listCalled) != 1 || typingAPI.listCalled[0] != "om_trigger" {
+ t.Fatalf("the trigger message's reactions were not listed; lists=%v", typingAPI.listCalled)
+ }
+ if len(typingAPI.deleteCalled) != 1 ||
+ typingAPI.deleteCalled[0].messageID != "om_trigger" ||
+ typingAPI.deleteCalled[0].reactionID != "r-stale" {
+ t.Fatalf("the reply arrived but the Typing badge is still on om_trigger — "+
+ "the user sees a processing spinner under an answered question "+
+ "(deletes=%+v)", typingAPI.deleteCalled)
+ }
+}
+
+// The same hole on the cancellation path: a user pressing cancel on another
+// replica (or after a restart) must still take the badge off, through the
+// delivery row's frozen trigger message.
+func TestPatcherSweepsTypingOnTaskCancelledThroughTheDeliveryRow(t *testing.T) {
+ p, q, api := newTestPatcher(t)
+ q.binding.LastMessageID = pgtype.Text{String: "om_trigger", Valid: true}
+ taskID := uuidFromString(t, "ee666666-ee66-ee66-ee66-eeeeeeeeeeee")
+ q.task = db.AgentTaskQueue{ChatInputTaskID: taskID}
+ q.taskChannelIngested = false
+
+ typingAPI := &fakeTypingAPIClient{
+ listReturn: []MessageReaction{
+ {ReactionID: "r-stale", OperatorType: "app", OperatorID: "cli_test_app", EmojiType: typingEmoji},
+ },
+ }
+ typing := NewTypingIndicatorManager(typingAPI, fakeTypingCreds{secret: "shh"},
+ &fakeTypingQueries{binding: q.binding, installation: q.installation}, newDiscardLogger())
+ p.SetTypingIndicatorManager(typing)
+
+ p.handleEvent(events.Event{
+ Type: protocol.EventTaskCancelled,
+ TaskID: uuidString(taskID),
+ ChatSessionID: uuidString(q.binding.ChatSessionID),
+ Payload: map[string]any{
+ "task_id": uuidString(taskID),
+ "chat_session_id": uuidString(q.binding.ChatSessionID),
+ "status": "cancelled",
+ },
+ })
+
+ if len(typingAPI.deleteCalled) != 1 || typingAPI.deleteCalled[0].reactionID != "r-stale" {
+ t.Fatalf("the run was cancelled with no state on file and the badge stayed on "+
+ "om_trigger; deletes=%+v", typingAPI.deleteCalled)
+ }
+ api.mu.Lock()
+ defer api.mu.Unlock()
+ if len(api.sent) != 0 || len(api.textSent) != 0 || len(api.patched) != 0 {
+ t.Errorf("a cancelled run must post nothing; sent=%d textSent=%d patched=%d",
+ len(api.sent), len(api.textSent), len(api.patched))
+ }
+}
+
+// A web-UI turn shares the session with Lark turns but delivers nothing to
+// Lark, so its early return sits before the send path. The in-memory clear
+// must run on the way out anyway: it is the only thing that can take a badge
+// off in this process, and skipping it lets a stranded badge ride until some
+// later Lark turn.
+func TestPatcherClearsTypingStateWhenTaskIsNotChannelIngested(t *testing.T) {
+ p, q, api := newTestPatcher(t)
+ taskID := uuidFromString(t, "ee555555-ee55-ee55-ee55-eeeeeeeeeeee")
+ q.task = db.AgentTaskQueue{ChatInputTaskID: taskID}
+ q.taskChannelIngested = false
+
+ typingAPI := &fakeTypingAPIClient{addReturn: "reaction-web-turn"}
+ typing := NewTypingIndicatorManager(typingAPI, fakeTypingCreds{secret: "shh"},
+ &fakeTypingQueries{binding: q.binding, installation: q.installation}, newDiscardLogger())
+ p.SetTypingIndicatorManager(typing)
+
+ typing.Add(context.Background(), q.installation, q.binding.ChatSessionID, "om_earlier_turn", "", q.binding.ChatSessionID)
+ if len(typingAPI.addCalled) != 1 {
+ t.Fatalf("setup: expected the Typing reaction to be added, got %d", len(typingAPI.addCalled))
+ }
+
+ p.handleEvent(events.Event{
+ Type: protocol.EventChatDone,
+ TaskID: uuidString(taskID),
+ ChatSessionID: uuidString(q.binding.ChatSessionID),
+ Payload: protocol.ChatDonePayload{Content: "answered in Multica only"},
+ })
+
+ if len(typingAPI.deleteCalled) != 1 || typingAPI.deleteCalled[0].reactionID != "reaction-web-turn" {
+ t.Fatalf("a web-UI turn on a Lark-bound session skipped the clear; deletes=%+v",
+ typingAPI.deleteCalled)
+ }
+ api.mu.Lock()
+ defer api.mu.Unlock()
+ if len(api.sent) != 0 || len(api.textSent) != 0 || len(api.patched) != 0 {
+ t.Errorf("a non-channel-ingested turn must not deliver to Lark; sent=%d textSent=%d patched=%d",
+ len(api.sent), len(api.textSent), len(api.patched))
+ }
+}
+
// ---- native @-mention of the triggering member (#8234) ----
// groupPatcherWithSender wires the shared setup for the mention tests: a
diff --git a/server/internal/integrations/lark/typing_cleanup.go b/server/internal/integrations/lark/typing_cleanup.go
new file mode 100644
index 00000000000..0501a99c0c9
--- /dev/null
+++ b/server/internal/integrations/lark/typing_cleanup.go
@@ -0,0 +1,213 @@
+package lark
+
+import (
+ "context"
+ "encoding/json"
+ "errors"
+ "fmt"
+ "time"
+
+ "github.com/jackc/pgx/v5/pgtype"
+ "github.com/multica-ai/multica/server/internal/integrations/channel/engine"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+type typingLedgerQueries interface {
+ RegisterChannelTypingReaction(context.Context, db.RegisterChannelTypingReactionParams) (db.ChannelTypingReaction, error)
+ FinishChannelTypingReactionAdd(context.Context, db.FinishChannelTypingReactionAddParams) (db.ChannelTypingReaction, error)
+ AcknowledgeChannelTypingReactionCleanup(context.Context, db.AcknowledgeChannelTypingReactionCleanupParams) error
+ ClaimChannelTypingReactionCleanup(context.Context, pgtype.UUID) ([]db.ChannelTypingReaction, error)
+ SettleChannelTypingInputs(context.Context, db.SettleChannelTypingInputsParams) error
+ SkipChannelTypingReactionAdd(context.Context, pgtype.UUID) error
+ PruneChannelTypingReactionCleanup(context.Context) error
+ ExpireChannelTypingReactionCleanup(context.Context) ([]db.ExpireChannelTypingReactionCleanupRow, error)
+ ListRequestedChannelTypingReactions(context.Context, []pgtype.UUID) ([]pgtype.UUID, error)
+}
+
+// Settle persists the failed flush's input boundary before removing reactions.
+// It covers inputs whose detached Add has not even registered yet.
+func (m *TypingIndicatorManager) Settle(ctx context.Context, sessionID pgtype.UUID, scope engine.TypingSettlement) {
+ ctx, cancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ defer cancel()
+ if err := m.queries.SettleChannelTypingInputs(ctx, db.SettleChannelTypingInputsParams{
+ ThroughMessageID: scope.ThroughMessageID, ChatSessionID: sessionID,
+ WorkspaceID: scope.WorkspaceID, InstallationID: scope.InstallationID, ContextRevision: pgtype.Int8{Int64: scope.ContextRevision, Valid: true},
+ }); err != nil {
+ m.log.Warn("lark typing: persist taskless settlement", "err", err)
+ return
+ }
+ m.Reconcile(ctx, sessionID)
+}
+
+// Reconcile claims all terminal inputs, not just the task's delivery trigger.
+// A null session selects due work across all workspaces for the system worker.
+// The ledger survives source deletion; eligibility is based on committed rows,
+// so rolled-back cancellation/deletion never grants cleanup ownership.
+func (m *TypingIndicatorManager) Reconcile(ctx context.Context, sessionID pgtype.UUID) map[string]bool {
+ defer m.pruneLocal(ctx)
+ covered := make(map[string]bool)
+ rows, err := m.queries.ClaimChannelTypingReactionCleanup(ctx, sessionID)
+ if err != nil {
+ m.log.Warn("lark typing: claim cleanup", "err", err)
+ return covered
+ }
+ for _, row := range rows {
+ if ctx.Err() != nil {
+ break
+ } // Claimed but unfinished work remains due after its lease.
+ covered[uuidString(row.InstallationID)+"/"+row.ChannelMessageID] = true
+ m.mu.Lock()
+ key := uuidString(row.ChatSessionID)
+ for _, state := range append([]*TypingIndicatorState(nil), m.states[key]...) {
+ if state.LedgerID == row.ID {
+ state.ended = true
+ m.removeState(key, state)
+ }
+ }
+ m.mu.Unlock()
+ callCtx, cancel := context.WithTimeout(ctx, typingCleanupTimeout)
+ m.cleanupReaction(callCtx, row)
+ cancel()
+ }
+ return covered
+}
+
+func (m *TypingIndicatorManager) cleanupReaction(ctx context.Context, row db.ChannelTypingReaction) {
+ var snapshot Installation
+ if err := json.Unmarshal(row.InstallationSnapshot, &snapshot); err != nil {
+ m.log.Warn("lark typing: read cleanup snapshot", "err", err)
+ return
+ }
+ creds := m.credentialsForInstallation(ctx, uuidString(row.ChatSessionID), row.InstallationID, snapshot)
+ if creds == nil {
+ return
+ }
+ var err error
+ if row.ReactionID != "" {
+ err = m.client.DeleteMessageReaction(ctx, DeleteReactionParams{InstallationID: *creds, MessageID: row.ChannelMessageID, ReactionID: row.ReactionID})
+ // A different replica may already have deleted this exact reaction. Prove
+ // absence rather than retrying forever on the API's already-missing error.
+ if err != nil {
+ if lister, ok := m.client.(ReactionLister); ok {
+ reactions, listErr := lister.ListMessageReactions(ctx, ListMessageReactionsParams{InstallationID: *creds, MessageID: row.ChannelMessageID, EmojiType: typingEmoji})
+ if listErr == nil {
+ found := false
+ for _, r := range reactions {
+ if r.ReactionID == row.ReactionID {
+ found = true
+ }
+ }
+ if !found {
+ err = nil
+ }
+ }
+ }
+ }
+ } else {
+ err = m.sweepForCleanup(ctx, *creds, row.ChannelMessageID)
+ }
+ if err != nil {
+ m.log.Warn("lark typing: cleanup retained for retry", "cleanup_id", uuidString(row.ID), "attempt", row.Attempts, "err", err)
+ return
+ }
+ if err = m.queries.AcknowledgeChannelTypingReactionCleanup(ctx, db.AcknowledgeChannelTypingReactionCleanupParams{ID: row.ID, ReactionID: row.ReactionID}); err != nil {
+ m.log.Warn("lark typing: cleanup acknowledgement failed; retry retained", "cleanup_id", uuidString(row.ID), "err", err)
+ }
+}
+
+func (m *TypingIndicatorManager) sweepForCleanup(ctx context.Context, creds InstallationCredentials, messageID string) error {
+ if creds.AppID == "" {
+ return fmt.Errorf("reaction owner App ID unavailable")
+ }
+ lister, ok := m.client.(ReactionLister)
+ if !ok {
+ return fmt.Errorf("reaction listing unavailable")
+ }
+ reactions, err := lister.ListMessageReactions(ctx, ListMessageReactionsParams{InstallationID: creds, MessageID: messageID, EmojiType: typingEmoji})
+ if err != nil {
+ return err
+ }
+ for _, r := range reactions {
+ if r.OperatorType != "app" || r.OperatorID != creds.AppID || r.EmojiType != typingEmoji {
+ continue
+ }
+ err = errors.Join(err, m.client.DeleteMessageReaction(ctx, DeleteReactionParams{InstallationID: creds, MessageID: messageID, ReactionID: r.ReactionID}))
+ }
+ return err
+}
+
+// Run retries durable cleanup after transient errors, lost bus events and
+// process restarts. Claims use a 30s-1h capped backoff and SKIP LOCKED; each
+// pass is bounded. Expiration and GC have a separate budget, so slow remote
+// calls cannot indefinitely retain credential snapshots or terminal records.
+func (m *TypingIndicatorManager) Run(ctx context.Context) {
+ ticker := time.NewTicker(30 * time.Second)
+ defer ticker.Stop()
+ for {
+ m.maintainCleanup(ctx)
+ passCtx, cancel := context.WithTimeout(ctx, 10*time.Second)
+ m.Reconcile(passCtx, pgtype.UUID{})
+ cancel()
+ select {
+ case <-ctx.Done():
+ return
+ case <-ticker.C:
+ }
+ }
+}
+
+func (m *TypingIndicatorManager) maintainCleanup(ctx context.Context) {
+ expireCtx, cancel := context.WithTimeout(ctx, typingCleanupTimeout)
+ rows, err := m.queries.ExpireChannelTypingReactionCleanup(expireCtx)
+ cancel()
+ if err != nil && ctx.Err() == nil {
+ m.log.Error("lark typing: credential expiry maintenance failed", "err", err)
+ }
+ counts := make(map[pgtype.UUID]int)
+ for _, row := range rows {
+ counts[row.WorkspaceID]++
+ }
+ for workspaceID, count := range counts {
+ m.log.Warn("lark typing: cleanup abandoned at retention deadline; credentials erased", "workspace_id", uuidString(workspaceID), "count", count)
+ }
+ pruneCtx, pruneCancel := context.WithTimeout(ctx, typingCleanupTimeout)
+ defer pruneCancel()
+ if err := m.queries.PruneChannelTypingReactionCleanup(pruneCtx); err != nil && ctx.Err() == nil {
+ m.log.Warn("lark typing: prune terminal cleanup", "err", err)
+ }
+}
+
+// A different replica can acknowledge cleanup without touching our local map.
+// Evict those stale views even when no remote retry is left to claim.
+func (m *TypingIndicatorManager) pruneLocal(ctx context.Context) {
+ m.mu.RLock()
+ var ids []pgtype.UUID
+ for _, states := range m.states {
+ for _, state := range states {
+ ids = append(ids, state.LedgerID)
+ }
+ }
+ m.mu.RUnlock()
+ if len(ids) == 0 {
+ return
+ }
+ ended, err := m.queries.ListRequestedChannelTypingReactions(ctx, ids)
+ if err != nil {
+ m.log.Warn("lark typing: refresh local cleanup state", "err", err)
+ return
+ }
+ closed := make(map[pgtype.UUID]bool, len(ended))
+ for _, id := range ended {
+ closed[id] = true
+ }
+ m.mu.Lock()
+ defer m.mu.Unlock()
+ for key, states := range m.states {
+ for _, state := range append([]*TypingIndicatorState(nil), states...) {
+ if closed[state.LedgerID] {
+ state.ended = true
+ m.removeState(key, state)
+ }
+ }
+ }
+}
diff --git a/server/internal/integrations/lark/typing_cleanup_db_test.go b/server/internal/integrations/lark/typing_cleanup_db_test.go
new file mode 100644
index 00000000000..b70ac245a3e
--- /dev/null
+++ b/server/internal/integrations/lark/typing_cleanup_db_test.go
@@ -0,0 +1,184 @@
+package lark
+
+import (
+ "context"
+ "errors"
+ "sync"
+ "testing"
+
+ "github.com/jackc/pgx/v5/pgtype"
+ dbfx "github.com/multica-ai/multica/server/internal/testutil"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+type retryTypingAPI struct {
+ *crossProcessReactionAPI
+ failureMu sync.Mutex
+ fail bool
+}
+
+func (a *retryTypingAPI) DeleteMessageReaction(ctx context.Context, p DeleteReactionParams) error {
+ a.failureMu.Lock()
+ fail := a.fail
+ a.failureMu.Unlock()
+ if fail {
+ return errors.New("remote unavailable")
+ }
+ return a.crossProcessReactionAPI.DeleteMessageReaction(ctx, p)
+}
+
+// FIXME(test-integration): Failed withdrawal must retain a restart-safe anchor,
+// even after all session/installation source rows disappear.
+func TestTypingWithdrawalRetryAfterSourceDeletionDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ f.Exec(t, "UPDATE agent_task_queue SET status='cancelled' WHERE id=$1", f.taskID)
+ api := &retryTypingAPI{crossProcessReactionAPI: &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: map[string][]MessageReaction{"retry-input": {{ReactionID: "human", OperatorType: "user", EmojiType: typingEmoji}}}}, fail: true}
+ a := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ a.Add(context.Background(), f.inst, f.sessionID, "retry-input", "", f.messageID)
+ var pending int
+ f.QueryRow(t, "SELECT count(*) FROM channel_typing_reaction WHERE workspace_id=$1 AND cleanup_required AND cleaned_at IS NULL AND reaction_id<>''", f.inst.WorkspaceID).Scan(&pending)
+ if pending != 1 || len(api.remote["retry-input"]) != 2 {
+ t.Fatalf("failed withdrawal lost compensation: pending=%d remote=%+v", pending, api.remote)
+ }
+ f.Exec(t, "DELETE FROM channel_chat_session_binding WHERE chat_session_id=$1", f.sessionID)
+ f.Exec(t, "DELETE FROM chat_message WHERE chat_session_id=$1", f.sessionID)
+ f.Exec(t, "DELETE FROM chat_session WHERE id=$1", f.sessionID)
+ f.Exec(t, "DELETE FROM channel_installation WHERE id=$1", f.inst.ID)
+ b := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ // First worker failure leases the record instead of spinning or dropping it.
+ b.Reconcile(context.Background(), f.sessionID)
+ f.QueryRow(t, "SELECT count(*) FROM channel_typing_reaction WHERE workspace_id=$1 AND cleaned_at IS NULL AND attempts=1 AND retry_after>now()", f.inst.WorkspaceID).Scan(&pending)
+ if pending != 1 {
+ t.Fatal("worker failure was not durably rescheduled")
+ }
+ api.failureMu.Lock()
+ api.fail = false
+ api.failureMu.Unlock()
+ f.Exec(t, "UPDATE channel_typing_reaction SET retry_after=now() WHERE workspace_id=$1", f.inst.WorkspaceID)
+ b.Reconcile(context.Background(), f.sessionID)
+ f.QueryRow(t, "SELECT count(*) FROM channel_typing_reaction WHERE workspace_id=$1 AND cleaned_at IS NOT NULL", f.inst.WorkspaceID).Scan(&pending)
+ if pending != 1 || len(api.remote["retry-input"]) != 1 || api.remote["retry-input"][0].ReactionID != "human" {
+ t.Fatalf("restart compensation failed: ack=%d remote=%+v", pending, api.remote)
+ }
+}
+
+// FIXME(test-integration): Eligibility must be post-commit and input scoped.
+func TestTypingCleanupPreservesRollbackAndNextInputDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ next := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "next", "channel_ingested": true})
+ api := &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: make(map[string][]MessageReaction)}
+ a := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ a.Add(context.Background(), f.inst, f.sessionID, "old", "", f.messageID)
+ a.Add(context.Background(), f.inst, f.sessionID, "next", "", uuidFromString(t, next))
+ tx, err := f.Pool.Begin(context.Background())
+ if err != nil {
+ t.Fatal(err)
+ }
+ defer tx.Rollback(context.Background())
+ if _, err = tx.Exec(context.Background(), "UPDATE agent_task_queue SET status='cancelled' WHERE id=$1", f.taskID); err != nil {
+ t.Fatal(err)
+ }
+ b := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ b.Reconcile(context.Background(), f.sessionID)
+ if len(api.remote["old"]) != 1 || len(api.remote["next"]) != 1 {
+ t.Fatalf("uncommitted terminal state removed reactions: %+v", api.remote)
+ }
+ if err = tx.Rollback(context.Background()); err != nil {
+ t.Fatal(err)
+ }
+ f.Exec(t, "UPDATE agent_task_queue SET status='cancelled' WHERE id=$1", f.taskID)
+ b.Reconcile(context.Background(), f.sessionID)
+ a.Reconcile(context.Background(), f.sessionID) // Also evict A's already-acknowledged local view.
+ if states := a.states[uuidString(f.sessionID)]; len(states) != 1 || states[0].MessageID != "next" {
+ t.Fatalf("local cleanup lost next input or retained old: %+v", states)
+ }
+ if len(api.remote["old"]) != 0 || len(api.remote["next"]) != 1 {
+ t.Fatalf("committed cleanup damaged next input: %+v", api.remote)
+ }
+}
+
+type lostTypingFinish struct{ *ChannelStore }
+
+func (q lostTypingFinish) FinishChannelTypingReactionAdd(context.Context, db.FinishChannelTypingReactionAddParams) (db.ChannelTypingReaction, error) {
+ return db.ChannelTypingReaction{}, errors.New("database unavailable after remote Add")
+}
+
+// FIXME(test-integration): Unknown HTTP outcomes are swept, never acknowledged
+// as complete merely because one early list was empty.
+func TestTypingUnrecordedAddRetainsRecoveryAnchorDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ api := &retryTypingAPI{crossProcessReactionAPI: &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: make(map[string][]MessageReaction)}, fail: true}
+ a := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, lostTypingFinish{f.store}, newDiscardLogger())
+ a.Add(context.Background(), f.inst, f.sessionID, "unknown", "", f.messageID)
+ f.Exec(t, "UPDATE channel_typing_reaction SET created_at=now()-interval '2 minutes',retry_after=now() WHERE workspace_id=$1", f.inst.WorkspaceID)
+ api.failureMu.Lock()
+ api.fail = false
+ api.failureMu.Unlock()
+ b := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ b.Reconcile(context.Background(), f.sessionID)
+ if len(api.remote["unknown"]) != 0 {
+ t.Fatal("unfinished Add was not compensated")
+ }
+ var retained int
+ f.QueryRow(t, "SELECT count(*) FROM channel_typing_reaction WHERE workspace_id=$1 AND cleanup_required AND cleaned_at IS NULL AND NOT add_finished AND retry_after>now()", f.inst.WorkspaceID).Scan(&retained)
+ if retained != 1 {
+ t.Fatal("uncertain Add must retain its anchor for subsequent remote changes")
+ }
+}
+
+// FIXME(test-integration): A forged settlement scope cannot mutate another
+// workspace; an aborted settlement transaction must leave the input live.
+func TestTypingSettlementWorkspaceAndRollbackDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ f.Exec(t, "UPDATE chat_message SET task_id=NULL WHERE id=$1", f.messageID)
+ arg := db.SettleChannelTypingInputsParams{ChatSessionID: f.sessionID, ThroughMessageID: f.messageID, WorkspaceID: f.inst.WorkspaceID, InstallationID: f.inst.ID, ContextRevision: pgtype.Int8{Int64: 1, Valid: true}}
+ tx, err := f.Pool.Begin(context.Background())
+ if err != nil {
+ t.Fatal(err)
+ }
+ defer tx.Rollback(context.Background())
+ if err = f.store.WithTx(tx).SettleChannelTypingInputs(context.Background(), arg); err != nil {
+ t.Fatal(err)
+ }
+ if err = tx.Rollback(context.Background()); err != nil {
+ t.Fatal(err)
+ }
+ arg.WorkspaceID = f.taskID
+ if err = f.store.SettleChannelTypingInputs(context.Background(), arg); err != nil {
+ t.Fatal(err)
+ }
+ var settled bool
+ f.QueryRow(t, "SELECT channel_typing_settled FROM chat_message WHERE id=$1", f.messageID).Scan(&settled)
+ if settled {
+ t.Fatal("rollback or foreign workspace settled the input")
+ }
+}
+
+// FIXME(test-integration): Every sealed input is enumerable independently of
+// delivery, while a new unowned input in the same session remains active.
+func TestTypingCleanupAllBatchInputsDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ api := &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: make(map[string][]MessageReaction)}
+ a := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ for i, name := range []string{"first", "middle", "trigger"} {
+ id := f.messageID
+ if i > 0 {
+ id = uuidFromString(t, f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": name, "task_id": f.taskID, "channel_ingested": true}))
+ }
+ api.remote[name] = []MessageReaction{{ReactionID: "human", OperatorType: "user", EmojiType: typingEmoji}}
+ a.Add(context.Background(), f.inst, f.sessionID, name, "", id)
+ }
+ next := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "next", "channel_ingested": true})
+ a.Add(context.Background(), f.inst, f.sessionID, "next", "", uuidFromString(t, next))
+ f.Exec(t, "UPDATE agent_task_queue SET status='completed' WHERE id=$1", f.taskID)
+ b := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ b.Reconcile(context.Background(), f.sessionID)
+ for _, name := range []string{"first", "middle", "trigger"} {
+ if r := api.remote[name]; len(r) != 1 || r[0].ReactionID != "human" {
+ t.Fatalf("batch input %s retains app or loses human: %+v", name, r)
+ }
+ }
+ if len(api.remote["next"]) != 1 {
+ t.Fatal("next input reaction removed")
+ }
+}
diff --git a/server/internal/integrations/lark/typing_cleanup_test.go b/server/internal/integrations/lark/typing_cleanup_test.go
new file mode 100644
index 00000000000..45dc8578f9c
--- /dev/null
+++ b/server/internal/integrations/lark/typing_cleanup_test.go
@@ -0,0 +1,84 @@
+package lark
+
+import (
+ "context"
+ "time"
+
+ "github.com/jackc/pgx/v5"
+ "github.com/jackc/pgx/v5/pgtype"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+func (f *fakeTypingQueries) RegisterChannelTypingReaction(_ context.Context, p db.RegisterChannelTypingReactionParams) (db.ChannelTypingReaction, error) {
+ f.ledgerMu.Lock()
+ defer f.ledgerMu.Unlock()
+ if f.ledger == nil {
+ f.ledger = make(map[pgtype.UUID]db.ChannelTypingReaction)
+ }
+ r := db.ChannelTypingReaction{ID: p.ID, WorkspaceID: p.WorkspaceID, ChatSessionID: p.ChatSessionID, ChatMessageID: p.ChatMessageID, InstallationID: p.InstallationID, ChannelMessageID: p.ChannelMessageID, InstallationSnapshot: p.InstallationSnapshot}
+ f.ledger[p.ID] = r
+ return r, nil
+}
+func (f *fakeTypingQueries) FinishChannelTypingReactionAdd(_ context.Context, p db.FinishChannelTypingReactionAddParams) (db.ChannelTypingReaction, error) {
+ f.ledgerMu.Lock()
+ defer f.ledgerMu.Unlock()
+ r, ok := f.ledger[p.ID]
+ if !ok {
+ return r, pgx.ErrNoRows
+ }
+ r.ReactionID = p.ReactionID
+ r.AddFinished = true
+ r.CleanupRequired = r.CleanupRequired || p.CleanupRequired
+ r.CleanedAt = pgtype.Timestamptz{}
+ f.ledger[p.ID] = r
+ return r, nil
+}
+func (f *fakeTypingQueries) AcknowledgeChannelTypingReactionCleanup(_ context.Context, p db.AcknowledgeChannelTypingReactionCleanupParams) error {
+ f.ledgerMu.Lock()
+ defer f.ledgerMu.Unlock()
+ r := f.ledger[p.ID]
+ if r.AddFinished && r.ReactionID != "" && r.ReactionID == p.ReactionID {
+ r.CleanedAt = pgtype.Timestamptz{Time: time.Now(), Valid: true}
+ f.ledger[p.ID] = r
+ }
+ return nil
+}
+
+// Existing event unit fixtures supply terminal events directly rather than
+// mutating DB tasks. Real eligibility is covered by the DB regression matrix.
+func (f *fakeTypingQueries) ClaimChannelTypingReactionCleanup(_ context.Context, session pgtype.UUID) ([]db.ChannelTypingReaction, error) {
+ f.ledgerMu.Lock()
+ defer f.ledgerMu.Unlock()
+ var rows []db.ChannelTypingReaction
+ for id, r := range f.ledger {
+ if !r.CleanedAt.Valid && (!session.Valid || r.ChatSessionID == session) {
+ r.CleanupRequired = true
+ f.ledger[id] = r
+ rows = append(rows, r)
+ }
+ }
+ return rows, nil
+}
+func (f *fakeTypingQueries) SettleChannelTypingInputs(context.Context, db.SettleChannelTypingInputsParams) error {
+ return nil
+}
+func (f *fakeTypingQueries) SkipChannelTypingReactionAdd(context.Context, pgtype.UUID) error {
+ return nil
+}
+func (f *fakeTypingQueries) PruneChannelTypingReactionCleanup(context.Context) error { return nil }
+
+func (f *fakeTypingQueries) ListRequestedChannelTypingReactions(_ context.Context, ids []pgtype.UUID) ([]pgtype.UUID, error) {
+ f.ledgerMu.Lock()
+ defer f.ledgerMu.Unlock()
+ var closed []pgtype.UUID
+ for _, id := range ids {
+ if f.ledger[id].CleanupRequired {
+ closed = append(closed, id)
+ }
+ }
+ return closed, nil
+}
+
+func (f *fakeTypingQueries) ExpireChannelTypingReactionCleanup(context.Context) ([]db.ExpireChannelTypingReactionCleanupRow, error) {
+ return nil, nil
+}
diff --git a/server/internal/integrations/lark/typing_coalesced_regression_test.go b/server/internal/integrations/lark/typing_coalesced_regression_test.go
new file mode 100644
index 00000000000..291612bd694
--- /dev/null
+++ b/server/internal/integrations/lark/typing_coalesced_regression_test.go
@@ -0,0 +1,133 @@
+package lark
+
+import (
+ "context"
+ "sync"
+ "testing"
+ "time"
+
+ "github.com/jackc/pgx/v5"
+ "github.com/jackc/pgx/v5/pgtype"
+ "github.com/multica-ai/multica/server/internal/events"
+ dbfx "github.com/multica-ai/multica/server/internal/testutil"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+ "github.com/multica-ai/multica/server/pkg/protocol"
+)
+
+type qaCoalescedCheckBarrier struct {
+ TypingIndicatorQueries
+ checked chan bool
+ release chan struct{}
+}
+
+func (q *qaCoalescedCheckBarrier) IsChannelMessageTypingActive(ctx context.Context, arg db.IsChannelMessageTypingActiveParams) (bool, error) {
+ active, err := q.TypingIndicatorQueries.IsChannelMessageTypingActive(ctx, arg)
+ if err != nil {
+ return active, err
+ }
+ q.checked <- active
+ select {
+ case <-q.release:
+ return active, nil
+ case <-ctx.Done():
+ return false, ctx.Err()
+ }
+}
+
+// A real committed DB read is paused after it observes active=true. Process B
+// then terminates the task and sweeps its single delivery trigger, which is not
+// the other input from the same sealed debounce batch.
+// FIXME(test-integration): Preserves QA's exact post-SELECT barrier and its
+// delivery-trigger control, including source deletion before terminal delivery.
+func TestTypingTrueCheckThenCrossProcessTerminalDB(t *testing.T) {
+ for _, removal := range []string{"retained", "deleted", "archived"} {
+ t.Run(removal, func(t *testing.T) {
+ t.Run("actual_delivery_trigger", func(t *testing.T) { qaRunTrueCheckThenTerminal(t, true, removal) })
+ t.Run("coalesced_non_trigger", func(t *testing.T) { qaRunTrueCheckThenTerminal(t, false, removal) })
+ })
+ }
+}
+
+func qaRunTrueCheckThenTerminal(t *testing.T, actualTrigger bool, removal string) {
+ f := newTypingDBFixture(t)
+ f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "delivery trigger in same batch", "task_id": f.taskID, "channel_ingested": true})
+ p, patcherQueries, _ := newTestPatcher(t)
+ patcherQueries.installation = f.inst
+ patcherQueries.binding.ChatSessionID = f.sessionID
+ patcherQueries.binding.InstallationID = f.inst.ID
+ patcherQueries.binding.LastMessageID = pgtype.Text{String: "delivery-trigger", Valid: true}
+ api := &crossProcessReactionAPI{
+ fakeTypingAPIClient: &fakeTypingAPIClient{},
+ remote: map[string][]MessageReaction{
+ "coalesced-input": {{ReactionID: "human-typing", OperatorType: "user", EmojiType: typingEmoji}},
+ "delivery-trigger": {{ReactionID: "trigger-app", OperatorType: "app", OperatorID: f.inst.AppID, EmojiType: typingEmoji}},
+ },
+ }
+ platformMessage := "coalesced-input"
+ wantTriggerRemaining := 0
+ if actualTrigger {
+ platformMessage = "delivery-trigger"
+ wantTriggerRemaining = 1
+ api.remote[platformMessage] = []MessageReaction{{ReactionID: "human-typing", OperatorType: "user", EmojiType: typingEmoji}}
+ }
+ barrier := &qaCoalescedCheckBarrier{TypingIndicatorQueries: f.store, checked: make(chan bool, 1), release: make(chan struct{})}
+ managerA := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, barrier, newDiscardLogger())
+ managerB := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, f.store, newDiscardLogger())
+ p.SetTypingIndicatorManager(managerB)
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ var release sync.Once
+ defer release.Do(func() { close(barrier.release) })
+ done := make(chan struct{})
+ go func() {
+ defer close(done)
+ managerA.Add(ctx, f.inst, f.sessionID, platformMessage, "", f.messageID)
+ }()
+ select {
+ case active := <-barrier.checked:
+ if !active {
+ t.Fatal("precondition: real DB must report old input active before termination")
+ }
+ case <-ctx.Done():
+ t.Fatal("real DB lifecycle check did not complete")
+ }
+ f.Exec(t, "UPDATE agent_task_queue SET status='cancelled' WHERE id=$1", f.taskID)
+ event := events.Event{Type: protocol.EventTaskCancelled, TaskID: uuidString(f.taskID), ChatSessionID: uuidString(f.sessionID)}
+ if removal == "deleted" {
+ f.Exec(t, "DELETE FROM channel_chat_session_binding WHERE chat_session_id=$1", f.sessionID)
+ f.Exec(t, "DELETE FROM chat_message WHERE chat_session_id=$1", f.sessionID)
+ f.Exec(t, "DELETE FROM chat_session WHERE id=$1", f.sessionID)
+ patcherQueries.deliveryErr = pgx.ErrNoRows
+ event.ChannelReactionTarget = &events.ChannelReactionTarget{ChannelType: channelTypeFeishu, InstallationID: uuidString(f.inst.ID), MessageID: "delivery-trigger"}
+ } else if removal == "archived" {
+ f.Exec(t, "UPDATE chat_session SET status='archived' WHERE id=$1", f.sessionID)
+ }
+ p.handleEvent(event)
+ api.mu.Lock()
+ triggerLeft, deletes, lists := len(api.remote["delivery-trigger"]), api.deletes, api.lists
+ api.mu.Unlock()
+ wantCleanups := 2
+ if actualTrigger {
+ wantCleanups = 1
+ }
+ if triggerLeft != wantTriggerRemaining || deletes != wantCleanups || lists != wantCleanups {
+ t.Fatalf("precondition: B must remove actual delivery trigger: trigger=%d deletes=%d lists=%d", triggerLeft, deletes, lists)
+ }
+ release.Do(func() { close(barrier.release) })
+ select {
+ case <-done:
+ case <-ctx.Done():
+ t.Fatal("Add did not complete")
+ }
+ active, err := f.store.IsChannelMessageTypingActive(ctx, db.IsChannelMessageTypingActiveParams{MessageID: f.messageID, ChatSessionID: f.sessionID, WorkspaceID: f.inst.WorkspaceID, InstallationID: f.inst.ID, ChannelType: channelTypeFeishu})
+ if err != nil || active {
+ t.Fatalf("precondition: committed terminal input must now be inactive: active=%v err=%v", active, err)
+ }
+ api.mu.Lock()
+ defer api.mu.Unlock()
+ remaining := api.remote[platformMessage]
+ t.Logf("after active SELECT -> B terminal sweep -> A return: lists=%d deletes=%d trigger=%d coalesced=%+v A_states=%d B_states=%d db_active=%v", api.lists, api.deletes, len(api.remote["delivery-trigger"]), remaining, len(managerA.states[uuidString(f.sessionID)]), len(managerB.states[uuidString(f.sessionID)]), active)
+ if len(remaining) != 1 || remaining[0].ReactionID != "human-typing" {
+ t.Fatalf("terminal cleanup must remove app reaction and preserve human Typing: want [human-typing], got %+v", remaining)
+ }
+}
diff --git a/server/internal/integrations/lark/typing_cross_process_test.go b/server/internal/integrations/lark/typing_cross_process_test.go
new file mode 100644
index 00000000000..02856cc9470
--- /dev/null
+++ b/server/internal/integrations/lark/typing_cross_process_test.go
@@ -0,0 +1,143 @@
+package lark
+
+import (
+ "context"
+ "sync"
+ "testing"
+ "time"
+
+ "github.com/jackc/pgx/v5"
+ "github.com/jackc/pgx/v5/pgtype"
+ "github.com/multica-ai/multica/server/internal/events"
+ dbfx "github.com/multica-ai/multica/server/internal/testutil"
+ "github.com/multica-ai/multica/server/pkg/protocol"
+)
+
+// Two independent managers share the remote API and durable DB, never local state.
+type crossProcessReactionAPI struct {
+ *fakeTypingAPIClient
+ mu sync.Mutex
+ started chan struct{}
+ release chan struct{}
+ remote map[string][]MessageReaction
+ lists int
+ deletes int
+}
+
+func (a *crossProcessReactionAPI) AddMessageReaction(ctx context.Context, p AddReactionParams) (string, error) {
+ if p.MessageID == "late-trigger" {
+ close(a.started)
+ select {
+ case <-a.release:
+ case <-ctx.Done():
+ return "", ctx.Err()
+ }
+ }
+ reactionID := "reaction-" + p.MessageID
+ a.mu.Lock()
+ a.remote[p.MessageID] = append(a.remote[p.MessageID], MessageReaction{ReactionID: reactionID, OperatorType: "app", OperatorID: p.InstallationID.AppID, EmojiType: typingEmoji})
+ a.mu.Unlock()
+ return reactionID, nil
+}
+
+func (a *crossProcessReactionAPI) ListMessageReactions(_ context.Context, p ListMessageReactionsParams) ([]MessageReaction, error) {
+ a.mu.Lock()
+ defer a.mu.Unlock()
+ a.lists++
+ return append([]MessageReaction(nil), a.remote[p.MessageID]...), nil
+}
+
+func (a *crossProcessReactionAPI) DeleteMessageReaction(_ context.Context, p DeleteReactionParams) error {
+ a.mu.Lock()
+ defer a.mu.Unlock()
+ a.deletes++
+ for i, reaction := range a.remote[p.MessageID] {
+ if reaction.ReactionID == p.ReactionID {
+ a.remote[p.MessageID] = append(a.remote[p.MessageID][:i], a.remote[p.MessageID][i+1:]...)
+ break
+ }
+ }
+ return nil
+}
+
+// FIXME(test-integration): Both QA orderings use a real shared DB and deterministic HTTP barriers.
+func TestTypingCrossProcessLateAddAfterTerminalSweepDB(t *testing.T) {
+ for _, deleted := range []bool{false, true} {
+ name := "delivery_retained"
+ if deleted {
+ name = "session_deleted_captured_target"
+ }
+ t.Run(name, func(t *testing.T) {
+ f := newTypingDBFixture(t)
+ p, q, _ := newTestPatcher(t)
+ q.installation = f.inst
+ q.binding.ChatSessionID = f.sessionID
+ q.binding.InstallationID = f.inst.ID
+ q.binding.LastMessageID = pgtype.Text{String: "late-trigger", Valid: true}
+ if deleted {
+ q.deliveryErr = pgx.ErrNoRows
+ }
+ api := &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, started: make(chan struct{}), release: make(chan struct{}), remote: map[string][]MessageReaction{"late-trigger": {{ReactionID: "human-typing", OperatorType: "user", EmojiType: typingEmoji}}}}
+ managerA := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, f.store, newDiscardLogger())
+ managerB := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, f.store, newDiscardLogger())
+ p.SetTypingIndicatorManager(managerB)
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ var release sync.Once
+ defer release.Do(func() { close(api.release) })
+ done := make(chan struct{})
+ go func() {
+ defer close(done)
+ managerA.Add(ctx, q.installation, q.binding.ChatSessionID, "late-trigger", "", f.messageID)
+ }()
+ select {
+ case <-api.started:
+ case <-ctx.Done():
+ t.Fatal("Add did not start")
+ }
+ f.Exec(t, "UPDATE agent_task_queue SET status='cancelled' WHERE id=$1", f.taskID)
+ if deleted {
+ f.Exec(t, "DELETE FROM channel_chat_session_binding WHERE chat_session_id=$1", f.sessionID)
+ f.Exec(t, "DELETE FROM chat_message WHERE chat_session_id=$1", f.sessionID)
+ f.Exec(t, "DELETE FROM chat_session WHERE id=$1", f.sessionID)
+ }
+ e := events.Event{Type: protocol.EventTaskCancelled, TaskID: uuidString(f.taskID), ChatSessionID: uuidString(q.binding.ChatSessionID)}
+ if deleted {
+ e.ChannelReactionTarget = &events.ChannelReactionTarget{ChannelType: channelTypeFeishu, InstallationID: uuidString(q.installation.ID), MessageID: "late-trigger"}
+ }
+ p.handleEvent(e)
+ api.mu.Lock()
+ lists, deletes := api.lists, api.deletes
+ t.Logf("terminal completed before Add: lists=%d deletes=%d remote=%d", lists, deletes, len(api.remote["late-trigger"]))
+ api.mu.Unlock()
+ if lists != 1 || deletes != 0 {
+ t.Fatalf("B must finish its app-empty sweep while A is blocked: lists=%d deletes=%d", lists, deletes)
+ }
+ if !deleted {
+ nextTask := f.Task(t, uuidString(f.inst.AgentID), dbfx.Cols{"chat_session_id": f.sessionID, "status": "running", "runtime_id": f.runtimeID})
+ nextMessage := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "next", "task_id": nextTask, "channel_ingested": true})
+ managerA.Add(ctx, f.inst, f.sessionID, "next-trigger", "", uuidFromString(t, nextMessage))
+ }
+ release.Do(func() { close(api.release) })
+ select {
+ case <-done:
+ case <-ctx.Done():
+ t.Fatal("Add did not finish")
+ }
+ api.mu.Lock()
+ defer api.mu.Unlock()
+ states := managerA.states[uuidString(f.sessionID)]
+ t.Logf("after late Add: lists=%d deletes=%d old_message_reactions=%+v process_A_states=%d process_B_states=%d", api.lists, api.deletes, api.remote["late-trigger"], len(states), len(managerB.states[uuidString(f.sessionID)]))
+ if reactions := api.remote["late-trigger"]; len(reactions) != 1 || reactions[0].ReactionID != "human-typing" || api.deletes != 1 {
+ t.Fatalf("late Add must delete exactly its own reaction and preserve human Typing: %+v deletes=%d", reactions, api.deletes)
+ }
+ if deleted {
+ if len(states) != 0 {
+ t.Fatalf("deleted session retains local state: %+v", states)
+ }
+ } else if len(states) != 1 || states[0].MessageID != "next-trigger" || len(api.remote["next-trigger"]) != 1 {
+ t.Fatalf("late cleanup damaged next turn: states=%+v remote=%+v", states, api.remote)
+ }
+ })
+ }
+}
diff --git a/server/internal/integrations/lark/typing_indicator.go b/server/internal/integrations/lark/typing_indicator.go
index b7edf792ba8..5adff7c5c31 100644
--- a/server/internal/integrations/lark/typing_indicator.go
+++ b/server/internal/integrations/lark/typing_indicator.go
@@ -2,14 +2,19 @@ package lark
import (
"context"
+ "encoding/json"
"errors"
"log/slog"
"strconv"
+ "strings"
"sync"
"time"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgtype"
+
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+ "github.com/multica-ai/multica/server/pkg/dbid"
)
// typingEmoji is the Lark emoji_type used for the "processing" indicator.
@@ -21,37 +26,22 @@ const typingEmoji = "Typing"
// reconnect replays old events. Aligned with OpenClaw's 2-minute bound.
const typingIndicatorMaxAge = 2 * time.Minute
-// TypingIndicatorState holds the identifiers needed to remove a reaction, plus
-// the installation whose app credentials added it. The installation id is
-// recorded at add time because that is the last moment it is certainly
-// resolvable: it is reachable from the session's channel_chat_session_binding
-// row, and a session delete drops that row while the cancel it triggers is
-// still on its way to the Patcher.
-type TypingIndicatorState struct {
- MessageID string
- ReactionID string
- InstallationID pgtype.UUID
+// typingCleanupTimeout bounds best-effort cleanup independently of reply delivery.
+const typingCleanupTimeout = 2 * time.Second
- // installSnapshot is the installation row as it stood when the reaction was
- // added, kept for the one case where the id is no longer enough: a runtime
- // teardown deletes the installation inside the same transaction that
- // cancels the tasks (handler/runtime.go,
- // DeleteChannelInstallationsBySystemRuntimeAgents), so by the time the
- // cancel reaches Clear there is no row to resolve.
- //
- // It is a FALLBACK, never the primary. A live lookup picks up a credential
- // rotation between add and clear; a snapshot cannot, so it is consulted
- // only when the row is genuinely gone.
- //
- // It does not weaken "no decrypted secret lives in the state map": what is
- // held here is the same encrypted blob the database holds, and
- // DecryptAppSecret still runs at clear time.
- installSnapshot Installation
+// TypingIndicatorState is a process-local view of one durable cleanup record.
+// It never owns the only copy of a remote reaction's cleanup anchor.
+type TypingIndicatorState struct {
+ LedgerID pgtype.UUID
+ MessageID, ReactionID string
+ ended bool // guarded by mu; Reconcile may end an Add before HTTP returns.
}
// TypingIndicatorQueries is the narrow DB surface the manager needs.
type TypingIndicatorQueries interface {
+ typingLedgerQueries
GetLarkInstallation(ctx context.Context, id pgtype.UUID) (Installation, error)
+ IsChannelMessageTypingActive(ctx context.Context, arg db.IsChannelMessageTypingActiveParams) (bool, error)
}
// TypingIndicatorManager owns the "processing" reaction lifecycle for
@@ -89,14 +79,15 @@ func NewTypingIndicatorManager(client APIClient, credentials CredentialsResolver
}
// Add sends a Typing reaction to the given message and records the state
-// under the chat session. It is synchronous — the caller decides whether
+// under the chat session. chatMessageID is the immutable persisted user input,
+// not the platform message ID or the latest task in the session. It is synchronous — the caller decides whether
// to run it in a detached goroutine. Errors are logged and swallowed.
//
// createTime is Lark's epoch-millisecond string (InboundMessage.CreateTime).
// Messages older than typingIndicatorMaxAge are silently skipped so that
// WebSocket replays and stale reconnects do not surface misleading "processing"
// badges on long-finished conversations.
-func (m *TypingIndicatorManager) Add(ctx context.Context, inst Installation, chatSessionID pgtype.UUID, messageID string, createTime string) {
+func (m *TypingIndicatorManager) Add(ctx context.Context, inst Installation, chatSessionID pgtype.UUID, messageID string, createTime string, chatMessageID pgtype.UUID) {
if messageID == "" {
return
}
@@ -118,12 +109,60 @@ func (m *TypingIndicatorManager) Add(ctx context.Context, inst Installation, cha
return
}
- reactionID, err := m.client.AddMessageReaction(ctx, AddReactionParams{
+ // The external anchor must be committed before HTTP can create a badge.
+ // No registration means no Add: terminal workers must never miss an input.
+ snapshot, err := json.Marshal(Installation{ID: inst.ID, WorkspaceID: inst.WorkspaceID, AppID: inst.AppID, AppSecretEncrypted: inst.AppSecretEncrypted, TenantKey: inst.TenantKey, Region: inst.Region})
+ if err != nil {
+ m.log.Warn("lark typing: encode installation snapshot", "err", err)
+ return
+ }
+ registrationCtx, registrationCancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ row, err := m.queries.RegisterChannelTypingReaction(registrationCtx, db.RegisterChannelTypingReactionParams{
+ ID: dbid.NewV7(), WorkspaceID: inst.WorkspaceID, ChatSessionID: chatSessionID,
+ ChatMessageID: chatMessageID, InstallationID: inst.ID, ChannelMessageID: messageID, InstallationSnapshot: snapshot,
+ })
+ registrationCancel()
+ if err != nil {
+ if errors.Is(err, pgx.ErrNoRows) {
+ m.log.Warn("lark typing: Add skipped; input unavailable or workspace quota occupied", "workspace_id", uuidString(inst.WorkspaceID))
+ return
+ }
+ m.log.Warn("lark typing: register cleanup anchor", "message_id", messageID, "err", err)
+ return
+ }
+ if row.CleanupRequired {
+ skipCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ defer cancel()
+ if err := m.queries.SkipChannelTypingReactionAdd(skipCtx, row.ID); err != nil {
+ m.log.Warn("lark typing: acknowledge skipped add", "err", err)
+ }
+ return
+ }
+
+ key := uuidString(chatSessionID)
+ state := &TypingIndicatorState{LedgerID: row.ID, MessageID: messageID}
+ m.mu.Lock()
+ m.states[key] = append(m.states[key], state)
+ m.mu.Unlock()
+
+ addCtx, addCancel := context.WithTimeout(ctx, typingCleanupTimeout)
+ reactionID, err := m.client.AddMessageReaction(addCtx, AddReactionParams{
InstallationID: creds,
MessageID: messageID,
EmojiType: typingEmoji,
})
+ addCancel()
if err != nil {
+ m.mu.Lock()
+ m.removeState(key, state)
+ m.mu.Unlock()
+ finishCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ defer cancel()
+ // A transport failure does not prove the remote side did not add it.
+ // Keep the anchor in the durable retry queue, including across restart.
+ if _, finishErr := m.queries.FinishChannelTypingReactionAdd(finishCtx, db.FinishChannelTypingReactionAddParams{ID: row.ID, CleanupRequired: true}); finishErr != nil {
+ m.log.Warn("lark typing: record uncertain Add", "err", finishErr)
+ }
m.log.Warn("lark typing indicator: add reaction failed",
"chat_session_id", uuidString(chatSessionID),
"message_id", messageID,
@@ -132,15 +171,50 @@ func (m *TypingIndicatorManager) Add(ctx context.Context, inst Installation, cha
return
}
- key := uuidString(chatSessionID)
- m.mu.Lock()
- m.states[key] = append(m.states[key], &TypingIndicatorState{
- MessageID: messageID,
- ReactionID: reactionID,
- InstallationID: inst.ID,
- installSnapshot: inst,
+ // The terminal event may have been handled by another process while the
+ // remote Add was in flight. Re-read this exact persisted input after the
+ // remote write so a committed terminal state retracts the late result.
+ // A session's latest
+ // task/delivery is not sufficient: debounce can seal several inputs and a
+ // newer turn may already exist. Missing/deleted inputs are no longer live.
+ checkCtx, checkCancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ active, checkErr := m.queries.IsChannelMessageTypingActive(checkCtx, db.IsChannelMessageTypingActiveParams{
+ MessageID: chatMessageID, ChatSessionID: chatSessionID,
+ WorkspaceID: inst.WorkspaceID, InstallationID: inst.ID,
+ ChannelType: channelTypeFeishu,
})
+ checkCancel()
+ if checkErr != nil {
+ // An unverified processing badge must not survive indefinitely.
+ m.log.Warn("lark typing indicator: input lifecycle lookup failed", "message_id", messageID, "err", checkErr)
+ }
+
+ m.mu.Lock()
+ ended := state.ended || !active || checkErr != nil
+ m.mu.Unlock()
+ finishCtx, finishCancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ finished, finishErr := m.queries.FinishChannelTypingReactionAdd(finishCtx, db.FinishChannelTypingReactionAddParams{ID: row.ID, ReactionID: reactionID, CleanupRequired: ended})
+ finishCancel()
+ if finishErr != nil {
+ m.log.Warn("lark typing: record Add result; durable unfinished anchor remains", "err", finishErr)
+ row.ReactionID = reactionID
+ row.CleanupRequired = true
+ finished = row
+ }
+ m.mu.Lock()
+ ended = ended || state.ended || finished.CleanupRequired || finishErr != nil
+ if ended {
+ m.removeState(key, state)
+ } else {
+ state.ReactionID = reactionID
+ }
m.mu.Unlock()
+ if ended {
+ cleanupCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), typingCleanupTimeout)
+ defer cancel()
+ m.cleanupReaction(cleanupCtx, finished)
+ return
+ }
m.log.Debug("lark typing indicator: reaction added",
"chat_session_id", key,
@@ -149,69 +223,77 @@ func (m *TypingIndicatorManager) Add(ctx context.Context, inst Installation, cha
)
}
-// Clear removes every tracked Typing reaction for the chat session and
-// drops the state entry. It is synchronous so the reaction is gone before
-// the agent's reply is sent, giving the user a clean visual transition.
-// Individual delete failures are logged but do not abort the loop.
+// removeState removes only this Add's entry, preserving later inputs. mu must be held.
+func (m *TypingIndicatorManager) removeState(key string, state *TypingIndicatorState) {
+ for i, pending := range m.states[key] {
+ if pending == state {
+ m.states[key] = append(m.states[key][:i], m.states[key][i+1:]...)
+ if len(m.states[key]) == 0 {
+ delete(m.states, key)
+ }
+ return
+ }
+ }
+}
+
+// SweepMessage removes every Typing reaction the bot itself put on a message,
+// found by listing the message's reactions from Lark rather than by consulting
+// the in-memory state. This is the authoritative half of the lifecycle: the
+// state map only exists in the process that ran Add, so a restart between add
+// and clear or a second replica handling completion leaves the map empty
+// while the badge is still on screen. In-flight adds are handled separately
+// by the durable pending-Add record. The sweep also retries reactions whose
+// exact deletion failed during reconciliation.
//
-// Credentials come from the installation each state recorded, not from the
-// session's binding, because a clear can outlive that binding: deleting a chat
-// session drops the binding row inside the same transaction that cancels the
-// session's tasks, and the task:cancelled events that reach the Patcher are
-// broadcast after that transaction commits. A binding lookup would miss, and
-// since the state has already been taken here, there would be nothing left to
-// clear from. Installation rows survive a session delete.
+// Both operator type and App ID must match. Other applications' reactions are
+// not ours to delete and must not turn an otherwise successful sweep into a retry.
//
-// They do NOT survive a runtime teardown, which deletes them in the same
-// transaction — so each state also carries the installation as it stood at add
-// time, consulted only when the row is gone. See TypingIndicatorState.
-func (m *TypingIndicatorManager) Clear(ctx context.Context, chatSessionID pgtype.UUID) {
- key := uuidString(chatSessionID)
- m.mu.Lock()
- states := m.states[key]
- delete(m.states, key)
- m.mu.Unlock()
-
- if len(states) == 0 {
+// credentials came from the caller, which resolves them from the same
+// installation that added the reaction; errors are logged and swallowed — the
+// indicator is best-effort and must never fail a reply. A client that does not
+// implement ReactionLister (the stub, minimal fakes) skips the sweep.
+func (m *TypingIndicatorManager) SweepMessage(ctx context.Context, creds InstallationCredentials, messageID string) {
+ if messageID == "" || creds.AppID == "" {
return
}
-
- // One session's reactions normally share an installation, so the resolved
- // credentials are memoised; a session rebound to another installation
- // mid-run still clears every reaction through the app that added it. A nil
- // entry records an installation that failed to resolve, so it is not
- // retried once per reaction.
- resolved := make(map[string]*InstallationCredentials, 1)
- for _, s := range states {
- if s.ReactionID == "" {
- continue
- }
- instKey := uuidString(s.InstallationID)
- creds, seen := resolved[instKey]
- if !seen {
- creds = m.credentialsForInstallation(ctx, key, s.InstallationID, s.installSnapshot)
- resolved[instKey] = creds
- }
- if creds == nil {
+ lister, ok := m.client.(ReactionLister)
+ if !ok {
+ m.log.Debug("lark typing indicator: client cannot list reactions, skipping sweep",
+ "message_id", messageID,
+ )
+ return
+ }
+ reactions, err := lister.ListMessageReactions(ctx, ListMessageReactionsParams{
+ InstallationID: creds,
+ MessageID: messageID,
+ EmojiType: typingEmoji,
+ })
+ if err != nil {
+ m.log.Warn("lark typing indicator: sweep list reactions failed",
+ "message_id", messageID,
+ "err", err,
+ )
+ return
+ }
+ for _, r := range reactions {
+ if !strings.EqualFold(r.EmojiType, typingEmoji) || r.OperatorType != "app" || r.OperatorID != creds.AppID {
continue
}
if err := m.client.DeleteMessageReaction(ctx, DeleteReactionParams{
- InstallationID: *creds,
- MessageID: s.MessageID,
- ReactionID: s.ReactionID,
+ InstallationID: creds,
+ MessageID: messageID,
+ ReactionID: r.ReactionID,
}); err != nil {
- m.log.Warn("lark typing indicator: delete reaction failed",
- "chat_session_id", key,
- "message_id", s.MessageID,
- "reaction_id", s.ReactionID,
+ m.log.Warn("lark typing indicator: sweep delete reaction failed",
+ "message_id", messageID,
+ "reaction_id", r.ReactionID,
"err", err,
)
continue
}
- m.log.Debug("lark typing indicator: reaction removed",
- "chat_session_id", key,
- "message_id", s.MessageID,
- "reaction_id", s.ReactionID,
+ m.log.Debug("lark typing indicator: sweep removed reaction",
+ "message_id", messageID,
+ "reaction_id", r.ReactionID,
)
}
}
diff --git a/server/internal/integrations/lark/typing_indicator_test.go b/server/internal/integrations/lark/typing_indicator_test.go
index d4876f35d2b..d5e836d43ff 100644
--- a/server/internal/integrations/lark/typing_indicator_test.go
+++ b/server/internal/integrations/lark/typing_indicator_test.go
@@ -4,17 +4,23 @@ import (
"context"
"errors"
"strconv"
+ "sync"
"testing"
"time"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgtype"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
)
// fakeTypingAPIClient records reaction calls and can be programmed to fail.
type fakeTypingAPIClient struct {
+ mu sync.Mutex
addCalled []addReactionCall
deleteCalled []deleteReactionCall
+ listCalled []string
+ listReturn []MessageReaction
+ listErr error
addErr error
deleteErr error
addReturn string
@@ -64,21 +70,47 @@ func (f *fakeTypingAPIClient) BatchGetUsers(context.Context, InstallationCredent
return nil, nil
}
func (f *fakeTypingAPIClient) AddMessageReaction(_ context.Context, p AddReactionParams) (string, error) {
+ f.mu.Lock()
+ defer f.mu.Unlock()
f.addCalled = append(f.addCalled, addReactionCall{p.InstallationID, p.MessageID, p.EmojiType})
return f.addReturn, f.addErr
}
func (f *fakeTypingAPIClient) DeleteMessageReaction(_ context.Context, p DeleteReactionParams) error {
+ f.mu.Lock()
+ defer f.mu.Unlock()
f.deleteCalled = append(f.deleteCalled, deleteReactionCall{p.InstallationID, p.MessageID, p.ReactionID})
return f.deleteErr
}
+func (f *fakeTypingAPIClient) ListMessageReactions(_ context.Context, p ListMessageReactionsParams) ([]MessageReaction, error) {
+ f.mu.Lock()
+ defer f.mu.Unlock()
+ f.listCalled = append(f.listCalled, p.MessageID)
+ return f.listReturn, f.listErr
+}
+
+// noListerClient wraps an APIClient without promoting any extra methods, so it
+// satisfies APIClient but not ReactionLister — the shape of a client that
+// cannot answer "what is already on this message".
+type noListerClient struct{ APIClient }
+
type fakeTypingQueries struct {
+ ledgerMu sync.Mutex
+ ledger map[pgtype.UUID]db.ChannelTypingReaction
+ active func(context.Context, db.IsChannelMessageTypingActiveParams) (bool, error)
binding ChatSessionBinding
installation Installation
bindingErr error
installErr error
}
+func (f *fakeTypingQueries) IsChannelMessageTypingActive(ctx context.Context, arg db.IsChannelMessageTypingActiveParams) (bool, error) {
+ if f.active != nil {
+ return f.active(ctx, arg)
+ }
+ return true, nil
+}
+
func (f *fakeTypingQueries) GetLarkChatSessionBindingBySession(context.Context, pgtype.UUID) (ChatSessionBinding, error) {
return f.binding, f.bindingErr
}
@@ -100,7 +132,7 @@ func TestTypingIndicatorAddRecordsState(t *testing.T) {
inst := Installation{AppID: "cli_test", Region: "feishu"}
session := pgtype.UUID{Bytes: [16]byte{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16}, Valid: true}
- mgr.Add(context.Background(), inst, session, "msg-1", "")
+ mgr.Add(context.Background(), inst, session, "msg-1", "", session)
if len(api.addCalled) != 1 {
t.Fatalf("expected 1 add call, got %d", len(api.addCalled))
@@ -125,7 +157,7 @@ func TestTypingIndicatorAddSkipsOnEmptyMessageID(t *testing.T) {
inst := Installation{AppID: "cli_test", Region: "feishu"}
session := pgtype.UUID{Bytes: [16]byte{1}, Valid: true}
- mgr.Add(context.Background(), inst, session, "", "")
+ mgr.Add(context.Background(), inst, session, "", "", session)
if len(api.addCalled) != 0 {
t.Fatalf("expected 0 add calls, got %d", len(api.addCalled))
@@ -140,7 +172,7 @@ func TestTypingIndicatorAddSkipsOldMessages(t *testing.T) {
session := pgtype.UUID{Bytes: [16]byte{1}, Valid: true}
oldTime := time.Now().Add(-3 * time.Minute).UnixMilli()
- mgr.Add(context.Background(), inst, session, "msg-old", strconv.FormatInt(oldTime, 10))
+ mgr.Add(context.Background(), inst, session, "msg-old", strconv.FormatInt(oldTime, 10), session)
if len(api.addCalled) != 0 {
t.Fatalf("expected 0 add calls for old message, got %d", len(api.addCalled))
@@ -154,7 +186,7 @@ func TestTypingIndicatorAddLogsOnAPIError(t *testing.T) {
inst := Installation{AppID: "cli_test", Region: "feishu"}
session := pgtype.UUID{Bytes: [16]byte{1}, Valid: true}
- mgr.Add(context.Background(), inst, session, "msg-1", "")
+ mgr.Add(context.Background(), inst, session, "msg-1", "", session)
if len(api.addCalled) != 1 {
t.Fatalf("expected 1 add call, got %d", len(api.addCalled))
@@ -169,7 +201,7 @@ func TestTypingIndicatorAddLogsOnAPIError(t *testing.T) {
}
}
-func TestTypingIndicatorClearDeletesReactions(t *testing.T) {
+func TestTypingIndicatorReconcileDeletesReactions(t *testing.T) {
api := &fakeTypingAPIClient{addReturn: "reaction-abc"}
queries := &fakeTypingQueries{
binding: ChatSessionBinding{
@@ -186,12 +218,12 @@ func TestTypingIndicatorClearDeletesReactions(t *testing.T) {
inst := Installation{AppID: "cli_test", Region: "feishu"}
session := pgtype.UUID{Bytes: [16]byte{1, 2, 3, 4}, Valid: true}
- mgr.Add(context.Background(), inst, session, "msg-1", "")
+ mgr.Add(context.Background(), inst, session, "msg-1", "", session)
if len(api.addCalled) != 1 {
t.Fatal("add should have been called")
}
- mgr.Clear(context.Background(), session)
+ mgr.Reconcile(context.Background(), session)
if len(api.deleteCalled) != 1 {
t.Fatalf("expected 1 delete call, got %d", len(api.deleteCalled))
@@ -209,7 +241,7 @@ func TestTypingIndicatorClearDeletesReactions(t *testing.T) {
}
}
-func TestTypingIndicatorClearNoOpWhenEmpty(t *testing.T) {
+func TestTypingIndicatorReconcileNoOpWhenEmpty(t *testing.T) {
api := &fakeTypingAPIClient{}
queries := &fakeTypingQueries{
binding: ChatSessionBinding{
@@ -224,14 +256,14 @@ func TestTypingIndicatorClearNoOpWhenEmpty(t *testing.T) {
mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, queries, newDiscardLogger())
session := pgtype.UUID{Bytes: [16]byte{1, 2, 3, 4}, Valid: true}
- mgr.Clear(context.Background(), session)
+ mgr.Reconcile(context.Background(), session)
if len(api.deleteCalled) != 0 {
t.Fatalf("expected 0 delete calls when empty, got %d", len(api.deleteCalled))
}
}
-func TestTypingIndicatorClearLogsOnDeleteError(t *testing.T) {
+func TestTypingIndicatorReconcileLogsOnDeleteError(t *testing.T) {
api := &fakeTypingAPIClient{addReturn: "reaction-xyz", deleteErr: errors.New("delete failed")}
queries := &fakeTypingQueries{
binding: ChatSessionBinding{
@@ -248,8 +280,8 @@ func TestTypingIndicatorClearLogsOnDeleteError(t *testing.T) {
inst := Installation{AppID: "cli_test", Region: "feishu"}
session := pgtype.UUID{Bytes: [16]byte{1, 2, 3, 4}, Valid: true}
- mgr.Add(context.Background(), inst, session, "msg-1", "")
- mgr.Clear(context.Background(), session)
+ mgr.Add(context.Background(), inst, session, "msg-1", "", session)
+ mgr.Reconcile(context.Background(), session)
if len(api.deleteCalled) != 1 {
t.Fatalf("expected 1 delete call attempt, got %d", len(api.deleteCalled))
@@ -273,14 +305,14 @@ func TestTypingIndicatorMultipleMessagesPerSession(t *testing.T) {
inst := Installation{AppID: "cli_test", Region: "feishu"}
session := pgtype.UUID{Bytes: [16]byte{1, 2, 3, 4}, Valid: true}
- mgr.Add(context.Background(), inst, session, "msg-a", "")
- mgr.Add(context.Background(), inst, session, "msg-b", "")
+ mgr.Add(context.Background(), inst, session, "msg-a", "", session)
+ mgr.Add(context.Background(), inst, session, "msg-b", "", session)
if len(api.addCalled) != 2 {
t.Fatalf("expected 2 add calls, got %d", len(api.addCalled))
}
- mgr.Clear(context.Background(), session)
+ mgr.Reconcile(context.Background(), session)
if len(api.deleteCalled) != 2 {
t.Fatalf("expected 2 delete calls, got %d", len(api.deleteCalled))
@@ -307,13 +339,13 @@ func TestTypingIndicatorConcurrentAddAndClear(t *testing.T) {
done := make(chan struct{})
go func() {
for i := 0; i < 50; i++ {
- mgr.Add(context.Background(), inst, session, "msg", "")
+ mgr.Add(context.Background(), inst, session, "msg", "", session)
}
close(done)
}()
go func() {
for i := 0; i < 50; i++ {
- mgr.Clear(context.Background(), session)
+ mgr.Reconcile(context.Background(), session)
}
}()
<-done
@@ -325,7 +357,7 @@ func TestTypingIndicatorConcurrentAddAndClear(t *testing.T) {
// the time the cancel arrives. The reaction is still on the message and the
// state has already been taken off the map, so the snapshot recorded at add
// time is the only thing left that can remove it.
-func TestTypingIndicatorClearsAfterTheInstallationIsDeleted(t *testing.T) {
+func TestTypingIndicatorReconcilesAfterTheInstallationIsDeleted(t *testing.T) {
api := &fakeTypingAPIClient{addReturn: "reaction-123"}
queries := &fakeTypingQueries{}
mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, queries, newDiscardLogger())
@@ -337,7 +369,7 @@ func TestTypingIndicatorClearsAfterTheInstallationIsDeleted(t *testing.T) {
}
session := pgtype.UUID{Bytes: [16]byte{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16}, Valid: true}
- mgr.Add(context.Background(), inst, session, "msg-1", "")
+ mgr.Add(context.Background(), inst, session, "msg-1", "", session)
if len(api.addCalled) != 1 {
t.Fatalf("setup: the reaction should be on the message; adds = %d", len(api.addCalled))
}
@@ -345,7 +377,7 @@ func TestTypingIndicatorClearsAfterTheInstallationIsDeleted(t *testing.T) {
// The teardown transaction has committed.
queries.installErr = pgx.ErrNoRows
- mgr.Clear(context.Background(), session)
+ mgr.Reconcile(context.Background(), session)
if len(api.deleteCalled) != 1 {
t.Fatalf("the runtime was torn down and its installation deleted, but the reaction is still "+
@@ -369,12 +401,115 @@ func TestTypingIndicatorDoesNotFallBackOnATransientLookupFailure(t *testing.T) {
}
session := pgtype.UUID{Bytes: [16]byte{1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16}, Valid: true}
- mgr.Add(context.Background(), inst, session, "msg-1", "")
+ mgr.Add(context.Background(), inst, session, "msg-1", "", session)
queries.installErr = errors.New("connection reset")
- mgr.Clear(context.Background(), session)
+ mgr.Reconcile(context.Background(), session)
if len(api.deleteCalled) != 0 {
t.Fatalf("a transient lookup failure fell back to the snapshot; deletes = %d", len(api.deleteCalled))
}
}
+
+// ---- SweepMessage: the restart- and replica-proof half of the lifecycle ----
+
+// The sweep is what clears the badge when the state map has nothing on file:
+// it must list the message's Typing reactions from Lark, delete the ones the
+// bot itself added, and leave everyone else's alone — a human who reacted with
+// the same emoji owns that reaction, and a bot cannot delete it anyway.
+func TestTypingIndicatorSweepDeletesOnlyTheBotTypingReactions(t *testing.T) {
+ api := &fakeTypingAPIClient{
+ listReturn: []MessageReaction{
+ {ReactionID: "r-bot", OperatorType: "app", OperatorID: "cli_test", EmojiType: typingEmoji},
+ {ReactionID: "r-human", OperatorType: "user", EmojiType: typingEmoji},
+ {ReactionID: "r-bot-lower", OperatorType: "app", OperatorID: "cli_test", EmojiType: "typing"},
+ {ReactionID: "r-bot-smile", OperatorType: "app", OperatorID: "cli_test", EmojiType: "SMILE"},
+ },
+ }
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{}, newDiscardLogger())
+
+ mgr.SweepMessage(context.Background(), InstallationCredentials{AppID: "cli_test"}, "om_trigger")
+
+ if len(api.listCalled) != 1 || api.listCalled[0] != "om_trigger" {
+ t.Fatalf("expected one list call for om_trigger, got %v", api.listCalled)
+ }
+ if len(api.deleteCalled) != 2 {
+ t.Fatalf("expected the two bot-authored Typing reactions to be deleted, deletes = %+v", api.deleteCalled)
+ }
+ deleted := map[string]bool{}
+ for _, d := range api.deleteCalled {
+ deleted[d.reactionID] = true
+ if d.messageID != "om_trigger" {
+ t.Errorf("deleted reaction %s from the wrong message %q", d.reactionID, d.messageID)
+ }
+ }
+ if !deleted["r-bot"] || !deleted["r-bot-lower"] {
+ t.Errorf("bot-authored Typing reactions survived the sweep: %+v", deleted)
+ }
+ if deleted["r-human"] || deleted["r-bot-smile"] {
+ t.Errorf("the sweep deleted reactions it did not own: %+v", deleted)
+ }
+}
+
+// A restart or a second replica empties the state map without touching Lark;
+// the sweep must clear the badge anyway, because it answers only to Lark.
+func TestTypingIndicatorSweepClearsWithoutAnyRecordedState(t *testing.T) {
+ api := &fakeTypingAPIClient{
+ listReturn: []MessageReaction{
+ {ReactionID: "r-from-another-process", OperatorType: "app", OperatorID: "cli_test", EmojiType: typingEmoji},
+ },
+ }
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{}, newDiscardLogger())
+
+ // No Add ever ran in this process: the map is empty by construction.
+ mgr.SweepMessage(context.Background(), InstallationCredentials{AppID: "cli_test"}, "om_left_behind")
+
+ if len(api.deleteCalled) != 1 || api.deleteCalled[0].reactionID != "r-from-another-process" {
+ t.Fatalf("the badge added by another process was not swept: deletes = %+v", api.deleteCalled)
+ }
+}
+
+// A client that cannot list reactions (the stub, minimal fakes) must skip the
+// sweep quietly instead of panicking or firing blind deletes.
+func TestTypingIndicatorSweepSkipsWhenClientCannotList(t *testing.T) {
+ inner := &fakeTypingAPIClient{}
+ api := &noListerClient{APIClient: inner}
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{}, newDiscardLogger())
+
+ mgr.SweepMessage(context.Background(), InstallationCredentials{AppID: "cli_test"}, "om_trigger")
+
+ if len(inner.listCalled) != 0 || len(inner.deleteCalled) != 0 {
+ t.Fatalf("a client without ReactionLister must not be consulted; lists = %d deletes = %d",
+ len(inner.listCalled), len(inner.deleteCalled))
+ }
+}
+
+// A failed list must not turn into deletes of a stale picture of the message.
+func TestTypingIndicatorSweepDoesNotDeleteWhenTheListFails(t *testing.T) {
+ api := &fakeTypingAPIClient{listErr: errors.New("lark 5xx")}
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{}, newDiscardLogger())
+
+ mgr.SweepMessage(context.Background(), InstallationCredentials{AppID: "cli_test"}, "om_trigger")
+
+ if len(api.deleteCalled) != 0 {
+ t.Fatalf("deleted reactions without a successful list; deletes = %+v", api.deleteCalled)
+ }
+}
+
+func TestTypingAddRetractsUnverifiedInputWithIndependentContext(t *testing.T) {
+ api := &fakeTypingAPIClient{addReturn: "unverified-reaction"}
+ queries := &fakeTypingQueries{active: func(ctx context.Context, _ db.IsChannelMessageTypingActiveParams) (bool, error) {
+ if ctx.Err() != nil {
+ t.Fatalf("lifecycle check inherited expired Add context: %v", ctx.Err())
+ }
+ return false, errors.New("database unavailable")
+ }}
+ manager := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, queries, newDiscardLogger())
+ ctx, cancel := context.WithCancel(context.Background())
+ cancel() // Model a successful response delivered after the caller cancelled.
+ session := uuidFromString(t, "ee777777-ee77-ee77-ee77-eeeeeeeeeeee")
+ manager.Add(ctx, Installation{AppID: "cli_test", Region: "feishu"}, session, "trigger", "", session)
+ if len(api.deleteCalled) != 1 || api.deleteCalled[0].reactionID != "unverified-reaction" || len(manager.states) != 0 {
+ t.Fatalf("unverified Add survived: deletes=%+v states=%+v", api.deleteCalled, manager.states)
+ }
+}
diff --git a/server/internal/integrations/lark/typing_lifecycle_db_test.go b/server/internal/integrations/lark/typing_lifecycle_db_test.go
new file mode 100644
index 00000000000..25976162c5a
--- /dev/null
+++ b/server/internal/integrations/lark/typing_lifecycle_db_test.go
@@ -0,0 +1,140 @@
+package lark
+
+import (
+ "context"
+ "os"
+ "testing"
+
+ "github.com/google/uuid"
+ "github.com/jackc/pgx/v5/pgtype"
+ "github.com/jackc/pgx/v5/pgxpool"
+ dbfx "github.com/multica-ai/multica/server/internal/testutil"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+type typingDBFixture struct {
+ *dbfx.Fixture
+ runtimeID string
+ store *ChannelStore
+ inst Installation
+ sessionID, messageID, taskID pgtype.UUID
+}
+
+func newTypingDBFixture(t *testing.T) *typingDBFixture {
+ t.Helper()
+ dsn := os.Getenv("DATABASE_URL")
+ if dsn == "" {
+ t.Skip("DATABASE_URL required for durable typing lifecycle regression")
+ }
+ pool, err := pgxpool.New(context.Background(), dsn)
+ if err != nil {
+ t.Fatal(err)
+ }
+ t.Cleanup(pool.Close)
+ fx := dbfx.New(pool, "", "")
+ suffix := uuid.NewString()
+ fx.UserID = fx.User(t, "Typing tester", "typing-"+suffix+"@multica.test")
+ fx.WorkspaceID = fx.Workspace(t, "Typing workspace", "typing-"+suffix)
+ runtimeID := fx.Runtime(t, "Typing runtime")
+ agentID := fx.Agent(t, "Typing agent", runtimeID)
+ instID := fx.Insert(t, "channel_installation", dbfx.Cols{
+ "workspace_id": fx.WorkspaceID, "agent_id": agentID, "channel_type": "feishu",
+ "installer_user_id": fx.UserID, "config": dbfx.Raw(`'{"app_id":"typing-app-` + suffix + `","region":"feishu"}'::jsonb`),
+ })
+ sid := fx.ChatSession(t, agentID)
+ fx.Insert(t, "channel_chat_session_binding", dbfx.Cols{
+ "chat_session_id": sid, "installation_id": instID, "channel_type": "feishu", "channel_chat_id": "typing-room", "chat_type": "group",
+ })
+ task := fx.Task(t, agentID, dbfx.Cols{"chat_session_id": sid, "status": "running", "runtime_id": runtimeID})
+ message := fx.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": sid, "role": "user", "content": "work", "task_id": task, "channel_ingested": true})
+ store := NewChannelStore(db.New(pool))
+ inst, err := store.GetLarkInstallation(context.Background(), uuidFromString(t, instID))
+ if err != nil {
+ t.Fatal(err)
+ }
+ fx.Cleanup(t, "DELETE FROM channel_typing_reaction WHERE workspace_id=$1", fx.WorkspaceID)
+ return &typingDBFixture{Fixture: fx, runtimeID: runtimeID, store: store, inst: inst, sessionID: uuidFromString(t, sid), messageID: uuidFromString(t, message), taskID: uuidFromString(t, task)}
+}
+
+// FIXME(test-integration): The shared committed input/task relation, including
+// debounce and rollback, is the cross-process lifecycle contract.
+func TestTypingInputLifecycleDB(t *testing.T) {
+ for _, scenario := range []string{"pending debounce", "running", "completed", "failed", "cancelled", "missing task", "deleted input", "deleted session", "archived session", "revoked installation", "foreign workspace", "foreign installation", "foreign task workspace", "foreign task session", "web input", "rollback", "next turn", "coalesced inputs"} {
+ t.Run(scenario, func(t *testing.T) {
+ f := newTypingDBFixture(t)
+ ctx := context.Background()
+ arg := db.IsChannelMessageTypingActiveParams{MessageID: f.messageID, ChatSessionID: f.sessionID, WorkspaceID: f.inst.WorkspaceID, InstallationID: f.inst.ID, ChannelType: channelTypeFeishu}
+ want := false
+ switch scenario {
+ case "pending debounce":
+ f.Exec(t, "UPDATE chat_message SET task_id=NULL WHERE id=$1", f.messageID)
+ want = true
+ case "running":
+ want = true
+ case "completed", "failed", "cancelled":
+ f.Exec(t, "UPDATE agent_task_queue SET status=$2 WHERE id=$1", f.taskID, scenario)
+ case "missing task":
+ f.Exec(t, "DELETE FROM agent_task_queue WHERE id=$1", f.taskID)
+ case "deleted input":
+ f.Exec(t, "DELETE FROM chat_message WHERE id=$1", f.messageID)
+ case "deleted session":
+ f.Exec(t, "DELETE FROM chat_session WHERE id=$1", f.sessionID)
+ case "archived session":
+ f.Exec(t, "UPDATE chat_session SET status='archived' WHERE id=$1", f.sessionID)
+ case "revoked installation":
+ f.Exec(t, "UPDATE channel_installation SET status='revoked' WHERE id=$1", f.inst.ID)
+ case "foreign workspace":
+ arg.WorkspaceID = uuidFromString(t, uuid.NewString())
+ case "foreign installation":
+ arg.InstallationID = uuidFromString(t, uuid.NewString())
+ case "foreign task workspace":
+ other := f.Workspace(t, "Other", "other-"+uuid.NewString())
+ otherAgent := dbfx.New(f.Pool, other, f.UserID).Agent(t, "Other agent", "")
+ f.Exec(t, "UPDATE agent_task_queue SET agent_id=$2 WHERE id=$1", f.taskID, otherAgent)
+ case "foreign task session":
+ other := f.ChatSession(t, uuidString(f.inst.AgentID))
+ f.Exec(t, "UPDATE agent_task_queue SET chat_session_id=$2 WHERE id=$1", f.taskID, other)
+ case "web input":
+ f.Exec(t, "UPDATE chat_message SET channel_ingested=false WHERE id=$1", f.messageID)
+ case "rollback":
+ tx, err := f.Pool.Begin(ctx)
+ if err != nil {
+ t.Fatal(err)
+ }
+ defer tx.Rollback(ctx)
+ if _, err = tx.Exec(ctx, "UPDATE agent_task_queue SET status='cancelled' WHERE id=$1", f.taskID); err != nil {
+ t.Fatal(err)
+ }
+ // A separate process must not observe an uncommitted terminal state.
+ live, err := f.store.IsChannelMessageTypingActive(ctx, arg)
+ if err != nil || !live {
+ t.Fatalf("uncommitted cancellation leaked: active=%v err=%v", live, err)
+ }
+ if err = tx.Rollback(ctx); err != nil {
+ t.Fatal(err)
+ }
+ want = true
+ case "next turn":
+ f.Exec(t, "UPDATE agent_task_queue SET status='completed' WHERE id=$1", f.taskID)
+ nextTask := f.Task(t, uuidString(f.inst.AgentID), dbfx.Cols{"chat_session_id": f.sessionID, "status": "running", "runtime_id": f.runtimeID})
+ next := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "next", "task_id": nextTask, "channel_ingested": true})
+ nextArg := arg
+ nextArg.MessageID = uuidFromString(t, next)
+ live, err := f.store.IsChannelMessageTypingActive(ctx, nextArg)
+ if err != nil || !live {
+ t.Fatalf("next input not active: %v %v", live, err)
+ }
+ case "coalesced inputs":
+ // The delivery trigger is only one message in a sealed batch. Every
+ // other input must still follow the same terminal task without delivery.
+ other := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "coalesced", "task_id": f.taskID, "channel_ingested": true})
+ arg.MessageID = uuidFromString(t, other)
+ f.Exec(t, "UPDATE agent_task_queue SET status='completed' WHERE id=$1", f.taskID)
+ }
+ active, err := f.store.IsChannelMessageTypingActive(ctx, arg)
+ if err != nil || active != want {
+ t.Fatalf("active=%v want=%v err=%v", active, want, err)
+ }
+ })
+ }
+}
diff --git a/server/internal/integrations/lark/typing_lifecycle_regression_test.go b/server/internal/integrations/lark/typing_lifecycle_regression_test.go
new file mode 100644
index 00000000000..cbd3c0c6364
--- /dev/null
+++ b/server/internal/integrations/lark/typing_lifecycle_regression_test.go
@@ -0,0 +1,175 @@
+package lark
+
+import (
+ "context"
+ "errors"
+ "sync"
+ "testing"
+ "time"
+
+ "github.com/jackc/pgx/v5"
+ "github.com/jackc/pgx/v5/pgtype"
+ "github.com/multica-ai/multica/server/internal/events"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+ "github.com/multica-ai/multica/server/pkg/protocol"
+)
+
+type blockedTypingAdd struct {
+ *fakeTypingAPIClient
+ started chan struct{}
+ release chan struct{}
+}
+
+func (a *blockedTypingAdd) AddMessageReaction(ctx context.Context, p AddReactionParams) (string, error) {
+ if p.MessageID != "trigger" {
+ return a.fakeTypingAPIClient.AddMessageReaction(ctx, p)
+ }
+ close(a.started)
+ <-a.release
+ return "late-reaction", nil
+}
+
+func (a *blockedTypingAdd) DeleteMessageReaction(ctx context.Context, p DeleteReactionParams) error {
+ if err := ctx.Err(); err != nil {
+ return err
+ }
+ return a.fakeTypingAPIClient.DeleteMessageReaction(ctx, p)
+}
+
+func TestPatcherTerminalEventDuringTypingAdd(t *testing.T) {
+ for _, eventType := range []string{protocol.EventChatDone, protocol.EventTaskFailed, protocol.EventTaskCancelled} {
+ t.Run(eventType, func(t *testing.T) {
+ p, q, _ := newTestPatcher(t)
+ q.taskChannelIngested = true
+ q.binding.LastMessageID = pgtype.Text{String: "trigger", Valid: true}
+ taskID := uuidFromString(t, "ee777777-ee77-ee77-ee77-eeeeeeeeeeee")
+ q.task = db.AgentTaskQueue{ChatInputTaskID: taskID}
+ api := &blockedTypingAdd{fakeTypingAPIClient: &fakeTypingAPIClient{addReturn: "next-reaction"}, started: make(chan struct{}), release: make(chan struct{})}
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{installation: q.installation}, newDiscardLogger())
+ p.SetTypingIndicatorManager(mgr)
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ var release sync.Once
+ defer release.Do(func() { close(api.release) })
+ done := make(chan struct{})
+ go func() {
+ defer close(done)
+ mgr.Add(ctx, q.installation, q.binding.ChatSessionID, "trigger", "", q.binding.ChatSessionID)
+ }()
+ select {
+ case <-api.started:
+ case <-ctx.Done():
+ t.Fatal("Add did not start")
+ }
+ p.handleEvent(events.Event{Type: eventType, TaskID: uuidString(taskID), ChatSessionID: uuidString(q.binding.ChatSessionID), Payload: protocol.ChatDonePayload{Content: "answer"}})
+ if len(api.listCalled) != 1 || len(api.deleteCalled) != 0 {
+ t.Fatal("terminal event must finish its empty sweep while Add is blocked")
+ }
+ // A later turn must survive cleanup of the older in-flight generation.
+ mgr.Add(ctx, q.installation, q.binding.ChatSessionID, "next-trigger", "", q.binding.ChatSessionID)
+ cancel() // Late cleanup must also survive cancellation of the add context.
+ release.Do(func() { close(api.release) })
+ select {
+ case <-done:
+ case <-time.After(5 * time.Second):
+ t.Fatal("Add did not finish")
+ }
+ if len(api.deleteCalled) != 1 || api.deleteCalled[0].reactionID != "late-reaction" {
+ t.Fatalf("late Add left its badge: %+v", api.deleteCalled)
+ }
+ states := mgr.states[uuidString(q.binding.ChatSessionID)]
+ if len(states) != 1 || states[0].MessageID != "next-trigger" {
+ t.Fatalf("late cleanup damaged next turn: %+v", states)
+ }
+ })
+ }
+}
+
+type deadlineTypingLister struct {
+ *fakeTypingAPIClient
+ replies *fakeAPIClient
+ t *testing.T
+ timedOut bool
+}
+
+func (a *deadlineTypingLister) ListMessageReactions(ctx context.Context, _ ListMessageReactionsParams) ([]MessageReaction, error) {
+ a.replies.mu.Lock()
+ sent := len(a.replies.textSent)
+ a.replies.mu.Unlock()
+ if sent != 1 {
+ a.t.Errorf("slow sweep started before reply delivery: replies=%d", sent)
+ }
+ <-ctx.Done()
+ a.timedOut = errors.Is(ctx.Err(), context.DeadlineExceeded)
+ return nil, ctx.Err()
+}
+
+func TestPatcherSlowSweepCannotConsumeReplyBudget(t *testing.T) {
+ p, q, replies := newTestPatcher(t)
+ taskID := uuidFromString(t, "ee777777-ee77-ee77-ee77-eeeeeeeeeeee")
+ q.task = db.AgentTaskQueue{ChatInputTaskID: taskID}
+ q.taskChannelIngested = true
+ q.binding.LastMessageID = pgtype.Text{String: "trigger", Valid: true}
+ api := &deadlineTypingLister{fakeTypingAPIClient: &fakeTypingAPIClient{}, replies: replies, t: t}
+ p.SetTypingIndicatorManager(NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{installation: q.installation}, newDiscardLogger()))
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ if err := p.processEvent(ctx, events.Event{Type: protocol.EventChatDone, TaskID: uuidString(taskID), ChatSessionID: uuidString(q.binding.ChatSessionID), Payload: protocol.ChatDonePayload{Content: "answer"}}); err != nil {
+ t.Fatal(err)
+ }
+ if !api.timedOut {
+ t.Fatal("list call did not reach its independent deadline")
+ }
+ if ctx.Err() != nil {
+ t.Fatalf("sweep exhausted reply context: %v", ctx.Err())
+ }
+ if len(replies.textSent) != 1 {
+ t.Fatal("reply lost")
+ }
+}
+
+func TestPatcherClearsTypingBeforeTerminalGates(t *testing.T) {
+ for _, eventType := range []string{protocol.EventChatDone, protocol.EventTaskFailed} {
+ for _, gate := range []string{"inactive installation", "missing delivery"} {
+ t.Run(eventType+"/"+gate, func(t *testing.T) {
+ p, q, replies := newTestPatcher(t)
+ taskID := uuidFromString(t, "ee777777-ee77-ee77-ee77-eeeeeeeeeeee")
+ q.task = db.AgentTaskQueue{ChatInputTaskID: taskID}
+ q.taskChannelIngested = true
+ api := &fakeTypingAPIClient{addReturn: "held-reaction"}
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{installation: q.installation}, newDiscardLogger())
+ p.SetTypingIndicatorManager(mgr)
+ mgr.Add(context.Background(), q.installation, q.binding.ChatSessionID, "trigger", "", q.binding.ChatSessionID)
+ if gate == "inactive installation" {
+ q.installation.Status = "revoked"
+ } else {
+ q.deliveryErr = pgx.ErrNoRows
+ }
+ p.handleEvent(events.Event{Type: eventType, TaskID: uuidString(taskID), ChatSessionID: uuidString(q.binding.ChatSessionID), Payload: protocol.ChatDonePayload{Content: "answer"}})
+ if len(api.deleteCalled) != 1 || api.deleteCalled[0].reactionID != "held-reaction" {
+ t.Fatalf("badge survived terminal gate: %+v", api.deleteCalled)
+ }
+ if len(mgr.states) != 0 {
+ t.Fatal("terminal state retained")
+ }
+ if len(replies.textSent)+len(replies.sent)+len(replies.mdCardSent) != 0 {
+ t.Fatal("terminal gate unexpectedly sent a reply")
+ }
+ })
+ }
+ }
+}
+
+func TestPatcherSessionDeleteSweepsCapturedTargetWithoutLocalState(t *testing.T) {
+ p, q, replies := newTestPatcher(t)
+ q.deliveryErr = pgx.ErrNoRows // Both binding and delivery disappeared at commit.
+ api := &fakeTypingAPIClient{listReturn: []MessageReaction{{ReactionID: "remote-reaction", OperatorType: "app", OperatorID: "cli_test_app", EmojiType: typingEmoji}}}
+ p.SetTypingIndicatorManager(NewTypingIndicatorManager(api, fakeTypingCreds{secret: "shh"}, &fakeTypingQueries{}, newDiscardLogger()))
+ p.handleEvent(events.Event{Type: protocol.EventTaskCancelled, TaskID: "ee777777-ee77-ee77-ee77-eeeeeeeeeeee", ChatSessionID: uuidString(q.binding.ChatSessionID), ChannelReactionTarget: &events.ChannelReactionTarget{ChannelType: channelTypeFeishu, InstallationID: uuidString(q.installation.ID), MessageID: "deleted-session-trigger"}})
+ if len(api.deleteCalled) != 1 || api.deleteCalled[0].messageID != "deleted-session-trigger" {
+ t.Fatalf("cross-process session delete stranded reaction: %+v", api.deleteCalled)
+ }
+ if len(replies.textSent)+len(replies.sent) != 0 {
+ t.Fatal("cancel sent a reply")
+ }
+}
diff --git a/server/internal/integrations/lark/typing_retention_test.go b/server/internal/integrations/lark/typing_retention_test.go
new file mode 100644
index 00000000000..1063e91e9b3
--- /dev/null
+++ b/server/internal/integrations/lark/typing_retention_test.go
@@ -0,0 +1,124 @@
+package lark
+
+import (
+ "context"
+ "sync"
+ "testing"
+
+ "github.com/google/uuid"
+ "github.com/jackc/pgx/v5"
+ "github.com/jackc/pgx/v5/pgtype"
+ dbfx "github.com/multica-ai/multica/server/internal/testutil"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+// FIXME(test-integration): Recreate the durable state left by a hard crash
+// after successful Add but before the in-memory debounce timer creates a task.
+func TestTypingTasklessCrashRecoveryDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ f.Exec(t, "UPDATE chat_message SET task_id=NULL WHERE id=$1", f.messageID)
+ api := &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: make(map[string][]MessageReaction)}
+ first := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ first.Add(context.Background(), f.inst, f.sessionID, "crashed", "", f.messageID)
+ f.Exec(t, "UPDATE channel_typing_reaction SET created_at=now()-interval '3 minutes' WHERE chat_message_id=$1", f.messageID)
+ next := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "next", "channel_ingested": true})
+ first.Add(context.Background(), f.inst, f.sessionID, "next", "", uuidFromString(t, next))
+ running := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "running", "task_id": f.taskID, "channel_ingested": true})
+ first.Add(context.Background(), f.inst, f.sessionID, "running", "", uuidFromString(t, running))
+ f.Exec(t, "UPDATE channel_typing_reaction SET created_at=now()-interval '3 minutes' WHERE chat_message_id=$1", running)
+ // A fresh process has no local states or timers, only the shared database.
+ fresh := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ fresh.Reconcile(context.Background(), pgtype.UUID{})
+ if len(api.remote["crashed"]) != 0 || len(api.remote["next"]) != 1 || len(api.remote["running"]) != 1 {
+ t.Fatalf("wrong cleanup scope: %+v", api.remote)
+ }
+ var settled bool
+ f.QueryRow(t, "SELECT channel_typing_settled FROM chat_message WHERE id=$1", f.messageID).Scan(&settled)
+ if settled {
+ t.Fatal("visual expiry must not settle or prevent future input processing")
+ }
+}
+
+// FIXME(test-integration): Cleanup retention is bounded even without source rows,
+// a working credential, a successful API call, or a known Add response.
+func TestTypingRetentionErasesCredentialsAndStopsRetriesDB(t *testing.T) {
+ for _, known := range []bool{false, true} {
+ t.Run(map[bool]string{false: "unknown_add", true: "permanent_failure"}[known], func(t *testing.T) {
+ f := newTypingDBFixture(t)
+ id, ws := uuid.NewString(), uuid.NewString()
+ reaction := ""
+ if known {
+ reaction = "remote-reaction"
+ }
+ f.Exec(t, `INSERT INTO channel_typing_reaction(id,workspace_id,chat_session_id,chat_message_id,installation_id,channel_message_id,installation_snapshot,reaction_id,add_finished,cleanup_required,created_at,attempts,quota_slot)
+ VALUES($1,$2,$3,$4,$5,'deleted','{"app_secret_enc":"synthetic-ciphertext"}',$6,$7,true,now()-interval '8 days',10000,1)`, id, ws, f.sessionID, f.messageID, f.inst.ID, reaction, known)
+ f.Cleanup(t, "DELETE FROM channel_typing_reaction WHERE id=$1", id)
+ api := &fakeTypingAPIClient{}
+ fresh := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ // Eligibility stops at the deadline even if maintenance has not run yet.
+ fresh.Reconcile(context.Background(), f.sessionID)
+ if len(api.deleteCalled)+len(api.listCalled) != 0 {
+ t.Fatal("expired credentials used for remote cleanup")
+ }
+ fresh.maintainCleanup(context.Background())
+ var abandoned, redacted, cleaned bool
+ f.QueryRow(t, `SELECT abandoned_at IS NOT NULL,installation_snapshot='{}'::jsonb,cleaned_at IS NOT NULL FROM channel_typing_reaction WHERE id=$1`, id).Scan(&abandoned, &redacted, &cleaned)
+ if !abandoned || !redacted || cleaned {
+ t.Fatalf("failed cleanup not preserved distinctly: abandoned=%v redacted=%v cleaned=%v", abandoned, redacted, cleaned)
+ }
+ _, err := f.store.FinishChannelTypingReactionAdd(context.Background(), db.FinishChannelTypingReactionAddParams{ID: uuidFromString(t, id), ReactionID: "late", CleanupRequired: true})
+ if err != pgx.ErrNoRows {
+ t.Fatalf("late Add resurrected expired ledger: %v", err)
+ }
+ fresh.Reconcile(context.Background(), f.sessionID)
+ if len(api.deleteCalled)+len(api.listCalled) != 0 {
+ t.Fatal("abandoned cleanup retried")
+ }
+ f.Exec(t, "UPDATE channel_typing_reaction SET abandoned_at=now()-interval '8 days' WHERE id=$1", id)
+ fresh.maintainCleanup(context.Background())
+ var exists bool
+ f.QueryRow(t, "SELECT EXISTS(SELECT 1 FROM channel_typing_reaction WHERE id=$1)", id).Scan(&exists)
+ if exists {
+ t.Fatal("expired non-secret tombstone was not pruned")
+ }
+ })
+ }
+}
+
+// FIXME(test-integration): The unique quota index arbitrates concurrent replicas;
+// the losing Add is skipped instead of creating an untracked remote reaction.
+func TestTypingWorkspaceQuotaConcurrentDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ f.Exec(t, `INSERT INTO channel_typing_reaction(id,workspace_id,chat_session_id,chat_message_id,installation_id,channel_message_id,installation_snapshot,quota_slot)
+ SELECT gen_random_uuid(),$1,$2,$3,$4,'occupied','{}',slot FROM generate_series(1,999) slot`, f.inst.WorkspaceID, f.sessionID, f.messageID, f.inst.ID)
+ api := &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: make(map[string][]MessageReaction)}
+ start := make(chan struct{})
+ var wg sync.WaitGroup
+ for i := 0; i < 8; i++ {
+ wg.Add(1)
+ go func() {
+ defer wg.Done()
+ <-start
+ NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger()).Add(context.Background(), f.inst, f.sessionID, uuid.NewString(), "", f.messageID)
+ }()
+ }
+ close(start)
+ wg.Wait()
+ var pending int
+ f.QueryRow(t, "SELECT count(*) FROM channel_typing_reaction WHERE workspace_id=$1 AND cleaned_at IS NULL AND abandoned_at IS NULL", f.inst.WorkspaceID).Scan(&pending)
+ if pending != 1000 || len(api.remote) != 1 {
+ t.Fatalf("quota bypass: pending=%d remoteAdds=%d", pending, len(api.remote))
+ }
+ other := newTypingDBFixture(t)
+ NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, other.store, newDiscardLogger()).Add(context.Background(), other.inst, other.sessionID, "neighbor", "", other.messageID)
+ if len(api.remote["neighbor"]) != 1 {
+ t.Fatal("workspace quota affected neighbor")
+ }
+ f.Exec(t, "UPDATE channel_typing_reaction SET created_at=now()-interval '8 days' WHERE workspace_id=$1", f.inst.WorkspaceID)
+ m := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ m.maintainCleanup(context.Background())
+ m.Add(context.Background(), f.inst, f.sessionID, "after-expiry", "", f.messageID)
+ if len(api.remote["after-expiry"]) != 1 {
+ t.Fatal("terminal slots were not released")
+ }
+}
diff --git a/server/internal/integrations/lark/typing_settled_regression_test.go b/server/internal/integrations/lark/typing_settled_regression_test.go
new file mode 100644
index 00000000000..98a86a1f3b8
--- /dev/null
+++ b/server/internal/integrations/lark/typing_settled_regression_test.go
@@ -0,0 +1,37 @@
+package lark
+
+import (
+ "context"
+ "github.com/multica-ai/multica/server/internal/integrations/channel"
+ "github.com/multica-ai/multica/server/internal/integrations/channel/engine"
+ dbfx "github.com/multica-ai/multica/server/internal/testutil"
+ "testing"
+)
+
+// The router's inline failed enqueue invokes OnSettled before Handle launches
+// the detached OnIngested. No task exists to supply a later terminal event.
+func TestTypingSettledBeforeAddRegistrationDB(t *testing.T) {
+ f := newTypingDBFixture(t)
+ f.Exec(t, "UPDATE chat_message SET task_id=NULL WHERE id=$1", f.messageID)
+ f.Exec(t, "DELETE FROM agent_task_queue WHERE id=$1", f.taskID)
+ api := &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: make(map[string][]MessageReaction)}
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "test"}, f.store, newDiscardLogger())
+ notifier := &feishuTypingNotifier{mgr: mgr}
+ next := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "next", "channel_ingested": true})
+ cancelled, cancel := context.WithCancel(context.Background())
+ cancel()
+ notifier.OnSettled(cancelled, f.sessionID, engine.TypingSettlement{WorkspaceID: f.inst.WorkspaceID, InstallationID: f.inst.ID, ThroughMessageID: f.messageID, ContextRevision: 1})
+ notifier.OnIngested(context.Background(), engine.ResolvedInstallation{ID: f.inst.ID, WorkspaceID: f.inst.WorkspaceID, Platform: f.inst}, channel.InboundMessage{MessageID: "taskless-input"}, f.sessionID, f.messageID)
+ notifier.OnIngested(context.Background(), engine.ResolvedInstallation{ID: f.inst.ID, WorkspaceID: f.inst.WorkspaceID, Platform: f.inst}, channel.InboundMessage{MessageID: "next-input"}, f.sessionID, uuidFromString(t, next))
+ if len(api.remote["next-input"]) != 1 {
+ t.Fatal("settling previous input removed next debounce input")
+ }
+ var tasks int
+ f.QueryRow(t, "SELECT count(*) FROM agent_task_queue WHERE chat_session_id=$1", f.sessionID).Scan(&tasks)
+ if tasks != 0 {
+ t.Fatalf("taskless settlement unexpectedly created %d tasks", tasks)
+ }
+ if len(api.remote["taskless-input"]) != 0 {
+ t.Fatalf("settled input without a task retained reaction: %+v", api.remote)
+ }
+}
diff --git a/server/internal/integrations/lark/typing_sweep_ownership_test.go b/server/internal/integrations/lark/typing_sweep_ownership_test.go
new file mode 100644
index 00000000000..1b7de52c6d7
--- /dev/null
+++ b/server/internal/integrations/lark/typing_sweep_ownership_test.go
@@ -0,0 +1,31 @@
+package lark
+
+import (
+ "context"
+ "testing"
+)
+
+func TestTypingSweepsPreserveForeignAppAndHuman(t *testing.T) {
+ for _, durable := range []bool{false, true} {
+ t.Run(map[bool]string{false: "fallback", true: "durable"}[durable], func(t *testing.T) {
+ api := &fakeTypingAPIClient{listReturn: []MessageReaction{
+ {ReactionID: "ours", OperatorType: "app", OperatorID: "cli_ours", EmojiType: typingEmoji},
+ {ReactionID: "foreign", OperatorType: "app", OperatorID: "cli_other", EmojiType: typingEmoji},
+ {ReactionID: "unknown", OperatorType: "app", EmojiType: typingEmoji},
+ {ReactionID: "human", OperatorType: "user", OperatorID: "ou_user", EmojiType: typingEmoji},
+ }}
+ m := NewTypingIndicatorManager(api, fakeTypingCreds{}, &fakeTypingQueries{}, newDiscardLogger())
+ creds := InstallationCredentials{AppID: "cli_ours"}
+ if durable {
+ if err := m.sweepForCleanup(context.Background(), creds, "message"); err != nil {
+ t.Fatal(err)
+ }
+ } else {
+ m.SweepMessage(context.Background(), creds, "message")
+ }
+ if len(api.deleteCalled) != 1 || api.deleteCalled[0].reactionID != "ours" {
+ t.Fatalf("foreign reaction targeted: %+v", api.deleteCalled)
+ }
+ })
+ }
+}
diff --git a/server/internal/integrations/lark/typing_withdrawal_regression_test.go b/server/internal/integrations/lark/typing_withdrawal_regression_test.go
new file mode 100644
index 00000000000..b8a25c72f0a
--- /dev/null
+++ b/server/internal/integrations/lark/typing_withdrawal_regression_test.go
@@ -0,0 +1,105 @@
+package lark
+
+import (
+ "context"
+ "errors"
+ "sync"
+ "testing"
+ "time"
+
+ "github.com/jackc/pgx/v5/pgtype"
+ dbfx "github.com/multica-ai/multica/server/internal/testutil"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+type qaR2CheckQueries struct {
+ *ChannelStore
+ oldID pgtype.UUID
+ fail bool
+ t *testing.T
+}
+
+func (q *qaR2CheckQueries) IsChannelMessageTypingActive(ctx context.Context, p db.IsChannelMessageTypingActiveParams) (bool, error) {
+ if p.MessageID == q.oldID {
+ deadline, ok := ctx.Deadline()
+ if ctx.Err() != nil || !ok || time.Until(deadline) > typingCleanupTimeout {
+ q.t.Errorf("lifecycle check lacks independent bounded context")
+ }
+ if q.fail {
+ return true, errors.New("QA database check unavailable")
+ }
+ }
+ return q.ChannelStore.IsChannelMessageTypingActive(ctx, p)
+}
+
+type qaR2DeleteBarrier struct {
+ *crossProcessReactionAPI
+ started, release chan struct{}
+ t *testing.T
+}
+
+func (a *qaR2DeleteBarrier) DeleteMessageReaction(ctx context.Context, p DeleteReactionParams) error {
+ if p.MessageID == "checked-trigger" {
+ deadline, ok := ctx.Deadline()
+ if ctx.Err() != nil || !ok || time.Until(deadline) > typingCleanupTimeout {
+ a.t.Errorf("reaction withdrawal lacks independent bounded context")
+ }
+ close(a.started)
+ select {
+ case <-a.release:
+ case <-ctx.Done():
+ return ctx.Err()
+ }
+ }
+ return a.crossProcessReactionAPI.DeleteMessageReaction(ctx, p)
+}
+
+func TestQAR2WithdrawalPreservesNextDebounceInput(t *testing.T) {
+ for _, failure := range []bool{false, true} {
+ name := "committed_terminal"
+ if failure {
+ name = "database_check_error"
+ }
+ t.Run(name, func(t *testing.T) {
+ f := newTypingDBFixture(t)
+ if !failure {
+ f.Exec(t, "UPDATE agent_task_queue SET status='completed' WHERE id=$1", f.taskID)
+ }
+ next := f.Insert(t, "chat_message", dbfx.Cols{"chat_session_id": f.sessionID, "role": "user", "content": "next debounce input", "channel_ingested": true})
+ api := &qaR2DeleteBarrier{crossProcessReactionAPI: &crossProcessReactionAPI{fakeTypingAPIClient: &fakeTypingAPIClient{}, remote: map[string][]MessageReaction{"checked-trigger": {{ReactionID: "human-typing", OperatorType: "user", EmojiType: typingEmoji}}}}, started: make(chan struct{}), release: make(chan struct{}), t: t}
+ queries := &qaR2CheckQueries{ChannelStore: f.store, oldID: f.messageID, fail: failure, t: t}
+ mgr := NewTypingIndicatorManager(api, fakeTypingCreds{secret: "fixture"}, queries, newDiscardLogger())
+ var once sync.Once
+ defer once.Do(func() { close(api.release) })
+ done := make(chan struct{})
+ ctx, cancel := context.WithCancel(context.Background())
+ cancel() // HTTP fake models a successful Add delivered after caller cancellation.
+ go func() { defer close(done); mgr.Add(ctx, f.inst, f.sessionID, "checked-trigger", "", f.messageID) }()
+ select {
+ case <-api.started:
+ case <-time.After(5 * time.Second):
+ t.Fatal("old reaction withdrawal did not start")
+ }
+ mgr.Add(context.Background(), f.inst, f.sessionID, "next-trigger", "", uuidFromString(t, next))
+ once.Do(func() { close(api.release) })
+ select {
+ case <-done:
+ case <-time.After(5 * time.Second):
+ t.Fatal("old withdrawal did not finish")
+ }
+ api.mu.Lock()
+ defer api.mu.Unlock()
+ if old := api.remote["checked-trigger"]; len(old) != 1 || old[0].ReactionID != "human-typing" {
+ t.Fatalf("old reaction or human reaction incorrect: %+v", old)
+ }
+ states := mgr.states[uuidString(f.sessionID)]
+ if len(api.remote["next-trigger"]) != 1 || len(states) != 1 || states[0].MessageID != "next-trigger" {
+ t.Fatalf("next debounce input lost: remote=%+v states=%+v", api.remote, states)
+ }
+ if api.deletes != 1 {
+ t.Fatalf("withdrawals=%d want 1", api.deletes)
+ }
+ t.Log("old app reaction removed; human Typing and next taskless input reaction/state retained; query and withdrawal independent of cancelled Add context")
+ })
+ }
+}
diff --git a/server/internal/integrations/slack/resolvers.go b/server/internal/integrations/slack/resolvers.go
index d47d7ae0873..b7caaf02dbd 100644
--- a/server/internal/integrations/slack/resolvers.go
+++ b/server/internal/integrations/slack/resolvers.go
@@ -413,7 +413,7 @@ type slackTypingNotifier struct{ mgr *TypingIndicatorManager }
// the bot is processing it. The resolved installation carries the bot token in
// its Config blob — the InstallationResolver stashed the db.ChannelInstallation
// row in Platform, the documented adapter boundary the core never reads.
-func (n *slackTypingNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID) {
+func (n *slackTypingNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID, chatMessageID pgtype.UUID) {
ci, ok := inst.Platform.(db.ChannelInstallation)
if !ok {
return
@@ -424,6 +424,6 @@ func (n *slackTypingNotifier) OnIngested(ctx context.Context, inst engine.Resolv
// OnSettled clears the reaction when the run trigger enqueued no task (agent
// offline / archived, or an enqueue failure) — the bus-driven clear on
// chat-done / task-failed never fires for those, so without this the 👀 sticks.
-func (n *slackTypingNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID) {
+func (n *slackTypingNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID, scope engine.TypingSettlement) {
n.mgr.Clear(ctx, sessionID)
}
diff --git a/server/internal/integrations/slack/typing_indicator_test.go b/server/internal/integrations/slack/typing_indicator_test.go
index 9dda7988ea1..b4dcaf7e676 100644
--- a/server/internal/integrations/slack/typing_indicator_test.go
+++ b/server/internal/integrations/slack/typing_indicator_test.go
@@ -4,6 +4,7 @@ import (
"context"
"errors"
"fmt"
+ "github.com/multica-ai/multica/server/internal/integrations/channel/engine"
"testing"
"time"
@@ -152,7 +153,7 @@ func TestSlackTypingNotifier_OnSettledClears(t *testing.T) {
m := newTestTyping(q, fr)
m.Add(context.Background(), db.ChannelInstallation{ID: uid(1), Config: slackInstallConfigJSON()}, sessionID, "C1", freshTS())
- (&slackTypingNotifier{mgr: m}).OnSettled(context.Background(), sessionID)
+ (&slackTypingNotifier{mgr: m}).OnSettled(context.Background(), sessionID, engine.TypingSettlement{})
if len(fr.removed) != 1 || fr.removed[0].Channel != "C1" {
t.Fatalf("OnSettled must clear the reaction, removed = %+v", fr.removed)
}
diff --git a/server/internal/integrations/telegram/resolvers.go b/server/internal/integrations/telegram/resolvers.go
index 9a05d100d4d..5daa707bbcc 100644
--- a/server/internal/integrations/telegram/resolvers.go
+++ b/server/internal/integrations/telegram/resolvers.go
@@ -318,7 +318,7 @@ func NewTypingNotifier(decrypt Decrypter, apiBase string, client *http.Client, l
return &typingNotifier{decrypt: decrypt, apiBase: apiBase, client: client, logger: logger}
}
-func (n *typingNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID) {
+func (n *typingNotifier) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID, chatMessageID pgtype.UUID) {
row, ok := inst.Platform.(db.ChannelInstallation)
if !ok {
return
@@ -341,4 +341,5 @@ func (n *typingNotifier) OnIngested(ctx context.Context, inst engine.ResolvedIns
}
}
-func (n *typingNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID) {}
+func (n *typingNotifier) OnSettled(ctx context.Context, sessionID pgtype.UUID, scope engine.TypingSettlement) {
+}
diff --git a/server/internal/integrations/wecom/stream_bubble_db_test.go b/server/internal/integrations/wecom/stream_bubble_db_test.go
index 8d822e8d8fb..f5817e5ced7 100644
--- a/server/internal/integrations/wecom/stream_bubble_db_test.go
+++ b/server/internal/integrations/wecom/stream_bubble_db_test.go
@@ -78,7 +78,7 @@ func (r *bubbleReplica) asked(t *testing.T, turn boundTurn, reqID string) {
Source: channel.Source{ChannelType: TypeWecom, ChatID: turn.chatID, ChatType: channel.ChatTypeP2P, SenderID: "USER_1"},
Raw: raw,
},
- sessionID)
+ sessionID, sessionID)
r.bus.Publish(events.Event{
Type: protocol.EventTaskQueued,
ChatSessionID: turn.sessionID,
diff --git a/server/internal/integrations/wecom/stream_bubble_test.go b/server/internal/integrations/wecom/stream_bubble_test.go
index 5839ce2db1b..6502e720717 100644
--- a/server/internal/integrations/wecom/stream_bubble_test.go
+++ b/server/internal/integrations/wecom/stream_bubble_test.go
@@ -319,7 +319,7 @@ func (r *bubbleRig) ask(t *testing.T, reqID string) {
Source: channel.Source{ChannelType: TypeWecom, ChatID: "CHAT_1", ChatType: channel.ChatTypeP2P, SenderID: "USER_1"},
Raw: raw,
},
- bubbleSessionID(t))
+ bubbleSessionID(t), bubbleSessionID(t))
}
// reconnect swaps the installation's live socket the way the Supervisor does
diff --git a/server/internal/integrations/wecom/stream_round_identity_test.go b/server/internal/integrations/wecom/stream_round_identity_test.go
index ddc0d76729c..63a77ba1463 100644
--- a/server/internal/integrations/wecom/stream_round_identity_test.go
+++ b/server/internal/integrations/wecom/stream_round_identity_test.go
@@ -13,6 +13,7 @@ package wecom
import (
"context"
+ "github.com/multica-ai/multica/server/internal/integrations/channel/engine"
"testing"
"time"
)
@@ -187,7 +188,7 @@ func TestAFlushThatStartedNoRunClosesTheBubbleWithNoRun(t *testing.T) {
rig.ran(t, "REQ-S1", "task-1") // running, bound
rig.ask(t, "REQ-S2") // the flush that found no runtime
- rig.typing.OnSettled(context.Background(), sessionID)
+ rig.typing.OnSettled(context.Background(), sessionID, engine.TypingSettlement{})
frames := rig.conn.streamFrames(t)
if len(frames) != 3 {
@@ -213,7 +214,7 @@ func TestASettledFlushLeavesARoundWaitingForItsRetry(t *testing.T) {
rig.ran(t, "REQ-RETRY-SETTLE", "task-1")
rig.failed(t, "task-1", true) // the attempt is being retried; the round waits
- rig.typing.OnSettled(context.Background(), bubbleSessionID(t))
+ rig.typing.OnSettled(context.Background(), bubbleSessionID(t), engine.TypingSettlement{})
if got := len(rig.conn.streamFrames(t)); got != 1 {
t.Fatalf("got %d stream frames, want 1 (the opening one) — a settled flush closed the "+
diff --git a/server/internal/integrations/wecom/typing_indicator.go b/server/internal/integrations/wecom/typing_indicator.go
index a1846ef1e59..7a33bf17386 100644
--- a/server/internal/integrations/wecom/typing_indicator.go
+++ b/server/internal/integrations/wecom/typing_indicator.go
@@ -185,7 +185,7 @@ func NewTypingIndicator(cfg TypingIndicatorConfig) *TypingIndicatorManager {
// nothing here needs to be quick for the ACK's sake — but everything here is
// best-effort: a bubble that fails to open costs the user a few seconds of
// uncertainty, and the answer still arrives as a plain message.
-func (m *TypingIndicatorManager) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID) {
+func (m *TypingIndicatorManager) OnIngested(ctx context.Context, inst engine.ResolvedInstallation, msg channel.InboundMessage, sessionID pgtype.UUID, chatMessageID pgtype.UUID) {
if m.senders == nil || m.streams == nil || !sessionID.Valid {
return
}
@@ -294,7 +294,7 @@ func (m *TypingIndicatorManager) OnIngested(ctx context.Context, inst engine.Res
//
// No bubble, nothing to say: the replier's notice is the whole of what the
// user is told, and there is no round left to address a second line to.
-func (m *TypingIndicatorManager) OnSettled(ctx context.Context, sessionID pgtype.UUID) {
+func (m *TypingIndicatorManager) OnSettled(ctx context.Context, sessionID pgtype.UUID, scope engine.TypingSettlement) {
if m.senders == nil || m.streams == nil || !sessionID.Valid {
return
}
diff --git a/server/internal/issueproperty/value.go b/server/internal/issueproperty/value.go
index 1fc3bc3b28f..42dc6d6af2d 100644
--- a/server/internal/issueproperty/value.go
+++ b/server/internal/issueproperty/value.go
@@ -25,6 +25,10 @@ const (
// MaxActorValues is exported so the handler's existing focused tests can
// continue to pin the public multi-actor limit after validation moved here.
MaxActorValues = 20
+ // Free-form list values (multi_text / multi_url) are capped the same way:
+ // the whole properties bag shares one 16KB row budget, and URL entries can
+ // individually reach 2048 bytes.
+ MaxListValues = 20
)
var actorKinds = []string{"member"}
@@ -160,6 +164,63 @@ func optionsHint(config propertyConfig) string {
return strings.Join(parts, ", ")
}
+// textItem validates one text value - a single `text` value or one
+// `multi_text` element. Text keeps interior spacing as written.
+func textItem(s string) (string, error) {
+ if strings.TrimSpace(s) == "" {
+ return "", errors.New("value cannot be empty (use DELETE to unset a property)")
+ }
+ if utf8.RuneCountInString(s) > maxTextValueLen {
+ return "", fmt.Errorf("value must be %d characters or fewer", maxTextValueLen)
+ }
+ return util.SanitizeTextForPostgres(s), nil
+}
+
+// urlItem validates and canonicalizes one http(s) URL value - a single `url`
+// value or one `multi_url` element.
+func urlItem(s string) (string, error) {
+ s = strings.TrimSpace(s)
+ if len(s) > maxURLValueLen {
+ return "", fmt.Errorf("value must be %d characters or fewer", maxURLValueLen)
+ }
+ parsed, err := url.Parse(s)
+ if err != nil || (parsed.Scheme != "http" && parsed.Scheme != "https") || parsed.Host == "" {
+ return "", errors.New("value must be an http(s) URL")
+ }
+ return s, nil
+}
+
+// validateStringList validates a multi_text / multi_url value: a non-empty
+// array whose every element passes the item validator. Duplicates are dropped
+// and the caller's order preserved, mirroring multi_actor.
+func validateStringList(value any, itemValidator func(string) (string, error)) ([]byte, error) {
+ items, ok := value.([]any)
+ if !ok || len(items) == 0 {
+ return nil, errors.New("value must be a non-empty array of strings")
+ }
+ if len(items) > MaxListValues {
+ return nil, fmt.Errorf("value cannot list more than %d entries", MaxListValues)
+ }
+ seen := make(map[string]struct{}, len(items))
+ out := make([]string, 0, len(items))
+ for _, item := range items {
+ text, ok := item.(string)
+ if !ok {
+ return nil, errors.New("value must be a non-empty array of strings")
+ }
+ canonical, err := itemValidator(text)
+ if err != nil {
+ return nil, err
+ }
+ if _, duplicate := seen[canonical]; duplicate {
+ continue
+ }
+ seen[canonical] = struct{}{}
+ out = append(out, canonical)
+ }
+ return json.Marshal(out)
+}
+
// ValidateValue checks a raw JSON value against the definition's type and
// returns the canonical JSON stored by every issue-property write path.
func ValidateValue(def db.IssueProperty, raw json.RawMessage) ([]byte, error) {
@@ -181,27 +242,25 @@ func ValidateValue(def db.IssueProperty, raw json.RawMessage) ([]byte, error) {
if !ok {
return nil, errors.New("value must be a string")
}
- if strings.TrimSpace(text) == "" {
- return nil, errors.New("value cannot be empty (use DELETE to unset a property)")
- }
- if utf8.RuneCountInString(text) > maxTextValueLen {
- return nil, fmt.Errorf("value must be %d characters or fewer", maxTextValueLen)
+ item, err := textItem(text)
+ if err != nil {
+ return nil, err
}
- return json.Marshal(util.SanitizeTextForPostgres(text))
+ return json.Marshal(item)
+ case "multi_text":
+ return validateStringList(value, textItem)
case "url":
text, ok := value.(string)
if !ok {
return nil, errors.New("value must be a URL string")
}
- text = strings.TrimSpace(text)
- if len(text) > maxURLValueLen {
- return nil, fmt.Errorf("value must be %d characters or fewer", maxURLValueLen)
- }
- parsed, err := url.Parse(text)
- if err != nil || (parsed.Scheme != "http" && parsed.Scheme != "https") || parsed.Host == "" {
- return nil, errors.New("value must be an http(s) URL")
+ item, err := urlItem(text)
+ if err != nil {
+ return nil, err
}
- return json.Marshal(text)
+ return json.Marshal(item)
+ case "multi_url":
+ return validateStringList(value, urlItem)
case "number":
if _, ok := value.(float64); !ok {
return nil, errors.New("value must be a number")
diff --git a/server/internal/service/builtin_skills/multica-platform/references/issues.md b/server/internal/service/builtin_skills/multica-platform/references/issues.md
index 29a40ceab66..d460e3bcfde 100644
--- a/server/internal/service/builtin_skills/multica-platform/references/issues.md
+++ b/server/internal/service/builtin_skills/multica-platform/references/issues.md
@@ -128,8 +128,8 @@ concurrent edits. There is no CLI bulk-export or `--all` mode.
Workspaces may define custom issue properties (Severity, Environment, QA
Status, Reviewer, ...). They are the place for durable, typed issue state:
values are validated against the definition (select options, date format,
-http(s) URL, member reference), visible in the issue sidebar, and addressed
-by name.
+http(s) URL, member reference, free-form text/URL lists), visible in the
+issue sidebar, and addressed by name.
- Read what exists before writing: `multica property list` shows the catalog;
`multica issue property list ` shows values set on the issue.
@@ -147,6 +147,14 @@ multica issue property unset --name Environment
workspace members only. `--value` takes a member name, email, UUID, short id,
or an explicit `member:`; `multi_actor` takes a comma-separated list
(duplicates dropped, order kept, max 20).
+- `multi_text` / `multi_url` properties hold free-form lists: `--value` is a
+ comma-separated list of strings / http(s) URLs (empty entries skipped,
+ duplicates dropped, order kept, max 20). They take no options. An entry that
+ itself contains a comma cannot survive that form — pass a JSON array instead:
+ `--value '["Smith, John","https://en.wikipedia.org/wiki/Washington,_D.C."]'`.
+ The array form is always safe; use it whenever any entry has a comma or the
+ input starts with `[`, for example `--value '["[draft] spec"]'`. The same
+ forms work during creation: `--property 'Aliases=["Smith, John","[draft] spec"]'`.
- Definitions may include an optional catalog icon for visual identification;
it does not change the property's type or value validation.
- Agents cannot create or edit property definitions (owner/admin humans only).
@@ -166,7 +174,8 @@ multica issue list --sort property:Impact --direction desc --output json
matches ANY of its values; different properties must ALL match. Values are
option names or ids (select types), `true`/`false` (checkbox), a member
name/email/id (actor types), or the value itself for text, url, number,
- and date (`YYYY-MM-DD`). The reserved value `__none__` matches
+ and date (`YYYY-MM-DD`); for `multi_text` / `multi_url` the value matches
+ any single element of the list exactly. The reserved value `__none__` matches
issues where the property is unset (works for every type; it is not
index-backed, so use it for targeted audits rather than as a default
listing filter). Only `=` is supported today; the `>=`, `<=` and `!=`
@@ -175,7 +184,7 @@ multica issue list --sort property:Impact --direction desc --output json
an ordinal scale (Low < Medium < High) sorts by meaning — and number/date/
text/url by value; issues without the property sort last either way.
Archived properties and types without an order (multi_select, checkbox,
- actor kinds) are rejected up front.
+ actor kinds, list types) are rejected up front.
- `issue list` and `issue get` return `properties` as a map of definition id
to stored value. Add `--resolve-properties` in JSON mode to get the rows
`issue property list` prints instead (name, type, stored value, display
@@ -187,8 +196,9 @@ multica issue list --status in_progress --output json --resolve-properties
multica issue get --resolve-properties
```
- Read `display` for a single value and `display_values` for a multi_select
- or multi_actor value; `value` keeps the stored ids.
+ Read `display` for a single value and `display_values` for a multi-value
+ property (multi_select, multi_actor, multi_text, multi_url); `value` keeps
+ the stored ids / strings.
## Status changes have server side effects
diff --git a/server/internal/service/channel_reaction_target.go b/server/internal/service/channel_reaction_target.go
new file mode 100644
index 00000000000..0cf9dfac721
--- /dev/null
+++ b/server/internal/service/channel_reaction_target.go
@@ -0,0 +1,40 @@
+package service
+
+import (
+ "context"
+ "errors"
+ "fmt"
+
+ "github.com/jackc/pgx/v5"
+ "github.com/jackc/pgx/v5/pgtype"
+ "github.com/multica-ai/multica/server/internal/events"
+ "github.com/multica-ai/multica/server/internal/util"
+ db "github.com/multica-ai/multica/server/pkg/db/generated"
+)
+
+// CaptureChannelReactionTargets runs in the cancellation transaction, before
+// deleting bindings and delivery rows. Pass its result to BroadcastCancelledTasks
+// after commit, so a different process can clean up without any Add state.
+func CaptureChannelReactionTargets(ctx context.Context, q interface {
+ GetChannelTaskDelivery(context.Context, pgtype.UUID) (db.ChannelTaskDelivery, error)
+}, tasks []db.AgentTaskQueue) (map[string]*events.ChannelReactionTarget, error) {
+ targets := make(map[string]*events.ChannelReactionTarget)
+ for _, task := range tasks {
+ delivery, err := q.GetChannelTaskDelivery(ctx, task.ID)
+ if errors.Is(err, pgx.ErrNoRows) {
+ continue
+ }
+ if err != nil {
+ return nil, fmt.Errorf("capture channel reaction target: %w", err)
+ }
+ if !delivery.ChannelMessageID.Valid || delivery.ChannelMessageID.String == "" {
+ continue
+ }
+ targets[util.UUIDToString(task.ID)] = &events.ChannelReactionTarget{
+ ChannelType: delivery.ChannelType,
+ InstallationID: util.UUIDToString(delivery.InstallationID),
+ MessageID: delivery.ChannelMessageID.String,
+ }
+ }
+ return targets, nil
+}
diff --git a/server/internal/service/task.go b/server/internal/service/task.go
index 0f890d5e6c2..1d4e5c6bc2d 100644
--- a/server/internal/service/task.go
+++ b/server/internal/service/task.go
@@ -2743,11 +2743,20 @@ func (s *TaskService) CancelTasksByTriggerComment(ctx context.Context, commentID
// showing a run that no longer exists. Each caller already knows the workspace
// — it is the one whose session, member or runtime is being torn down — so the
// lookup is not needed and cannot fail.
-func (s *TaskService) BroadcastCancelledTasks(ctx context.Context, workspaceID string, cancelled []db.AgentTaskQueue) {
+//
+// reactionTargets optionally carries anchors captured before deleting the
+// channel delivery rows. They stay on the internal event, outside its payload.
+func (s *TaskService) BroadcastCancelledTasks(ctx context.Context, workspaceID string, cancelled []db.AgentTaskQueue, reactionTargets ...map[string]*events.ChannelReactionTarget) {
for _, t := range cancelled {
s.captureTaskCancelled(ctx, t)
s.ReconcileAgentStatus(ctx, t.AgentID)
- s.publishTaskEvent(protocol.EventTaskCancelled, workspaceID, t)
+ if workspaceID != "" {
+ e := taskEvent(protocol.EventTaskCancelled, workspaceID, t)
+ if len(reactionTargets) > 0 {
+ e.ChannelReactionTarget = reactionTargets[0][util.UUIDToString(t.ID)]
+ }
+ s.Bus.Publish(e)
+ }
}
s.notifyTasksFinished(cancelled)
}
diff --git a/server/migrations/564_issue_property_list_types.down.sql b/server/migrations/564_issue_property_list_types.down.sql
new file mode 100644
index 00000000000..f05cb4ab9a3
--- /dev/null
+++ b/server/migrations/564_issue_property_list_types.down.sql
@@ -0,0 +1,26 @@
+-- Restore the pre-list type allowlist.
+--
+-- Fails closed when any multi_text / multi_url definition still exists, the
+-- same stance as 341's down: rewriting or deleting those rows would destroy
+-- user data keyed by definition id. Archived definitions count too; their
+-- values stay resolvable. Convert the definitions (and the issue values keyed
+-- to them) to a pre-list type before rolling back.
+DO $$
+DECLARE
+ list_defs BIGINT;
+BEGIN
+ SELECT count(*) INTO list_defs
+ FROM issue_property
+ WHERE type IN ('multi_text', 'multi_url');
+
+ IF list_defs > 0 THEN
+ RAISE EXCEPTION 'cannot roll back 564: % multi_text/multi_url property definition(s) still exist', list_defs
+ USING HINT = 'Convert those definitions and the issue values keyed to them to a pre-list type first; this migration will not delete user data.';
+ END IF;
+END
+$$;
+
+ALTER TABLE issue_property DROP CONSTRAINT IF EXISTS issue_property_type_check;
+ALTER TABLE issue_property ADD CONSTRAINT issue_property_type_check
+ CHECK (type IN ('text', 'number', 'select', 'multi_select', 'date', 'checkbox', 'url', 'actor', 'multi_actor')) NOT VALID;
+ALTER TABLE issue_property VALIDATE CONSTRAINT issue_property_type_check;
diff --git a/server/migrations/564_issue_property_list_types.up.sql b/server/migrations/564_issue_property_list_types.up.sql
new file mode 100644
index 00000000000..8aa37cac260
--- /dev/null
+++ b/server/migrations/564_issue_property_list_types.up.sql
@@ -0,0 +1,18 @@
+-- Custom issue property list types: multi_text / multi_url.
+--
+-- Adds 'multi_text' and 'multi_url' to the type allowlist. Both store an
+-- array of free-form strings (text entries or http(s) URLs) in insertion
+-- order; validation, dedup, and the entry cap live in the handler, which is
+-- also where each element is validated exactly like a single text / url
+-- value. This constraint is only the outer guard.
+--
+-- Like the actor types (341), the values ride the existing properties jsonb
+-- bag: the @> containment filter for element-equality, the jsonb_path_ops GIN
+-- index, and the client value schema all keep working unchanged.
+--
+-- NOT VALID + VALIDATE keeps the ACCESS EXCLUSIVE lock instantaneous;
+-- existing rows cannot carry the new types, so validation is a formality.
+ALTER TABLE issue_property DROP CONSTRAINT IF EXISTS issue_property_type_check;
+ALTER TABLE issue_property ADD CONSTRAINT issue_property_type_check
+ CHECK (type IN ('text', 'number', 'select', 'multi_select', 'date', 'checkbox', 'url', 'actor', 'multi_actor', 'multi_text', 'multi_url')) NOT VALID;
+ALTER TABLE issue_property VALIDATE CONSTRAINT issue_property_type_check;
diff --git a/server/migrations/565_channel_typing_cleanup.down.sql b/server/migrations/565_channel_typing_cleanup.down.sql
new file mode 100644
index 00000000000..96b342575bc
--- /dev/null
+++ b/server/migrations/565_channel_typing_cleanup.down.sql
@@ -0,0 +1,2 @@
+DROP TABLE IF EXISTS channel_typing_reaction;
+ALTER TABLE chat_message DROP COLUMN IF EXISTS channel_typing_settled;
diff --git a/server/migrations/565_channel_typing_cleanup.up.sql b/server/migrations/565_channel_typing_cleanup.up.sql
new file mode 100644
index 00000000000..6ab139bf1aa
--- /dev/null
+++ b/server/migrations/565_channel_typing_cleanup.up.sql
@@ -0,0 +1,21 @@
+-- Durable per-input settlement survives an OnSettled callback before Add starts.
+ALTER TABLE chat_message ADD COLUMN IF NOT EXISTS channel_typing_settled boolean NOT NULL DEFAULT false;
+
+-- No foreign keys: cleanup anchors must survive session, task and installation
+-- deletion. The snapshot contains encrypted installation credentials only.
+CREATE TABLE IF NOT EXISTS channel_typing_reaction (
+ id uuid NOT NULL,
+ workspace_id uuid NOT NULL,
+ chat_session_id uuid NOT NULL,
+ chat_message_id uuid NOT NULL,
+ installation_id uuid NOT NULL,
+ channel_message_id text NOT NULL,
+ installation_snapshot jsonb NOT NULL,
+ reaction_id text NOT NULL DEFAULT '',
+ add_finished boolean NOT NULL DEFAULT false,
+ cleanup_required boolean NOT NULL DEFAULT false,
+ cleaned_at timestamptz,
+ retry_after timestamptz NOT NULL DEFAULT now(),
+ attempts integer NOT NULL DEFAULT 0,
+ created_at timestamptz NOT NULL DEFAULT now()
+);
diff --git a/server/migrations/566_channel_typing_reaction_id_idx.down.sql b/server/migrations/566_channel_typing_reaction_id_idx.down.sql
new file mode 100644
index 00000000000..6805fe1656d
--- /dev/null
+++ b/server/migrations/566_channel_typing_reaction_id_idx.down.sql
@@ -0,0 +1 @@
+DROP INDEX CONCURRENTLY IF EXISTS channel_typing_reaction_id_idx;
diff --git a/server/migrations/566_channel_typing_reaction_id_idx.up.sql b/server/migrations/566_channel_typing_reaction_id_idx.up.sql
new file mode 100644
index 00000000000..51c9e4beedd
--- /dev/null
+++ b/server/migrations/566_channel_typing_reaction_id_idx.up.sql
@@ -0,0 +1 @@
+CREATE UNIQUE INDEX CONCURRENTLY IF NOT EXISTS channel_typing_reaction_id_idx ON channel_typing_reaction (id);
diff --git a/server/migrations/567_channel_typing_reaction_retry_idx.down.sql b/server/migrations/567_channel_typing_reaction_retry_idx.down.sql
new file mode 100644
index 00000000000..f88b5f45285
--- /dev/null
+++ b/server/migrations/567_channel_typing_reaction_retry_idx.down.sql
@@ -0,0 +1 @@
+DROP INDEX CONCURRENTLY IF EXISTS channel_typing_reaction_retry_idx;
diff --git a/server/migrations/567_channel_typing_reaction_retry_idx.up.sql b/server/migrations/567_channel_typing_reaction_retry_idx.up.sql
new file mode 100644
index 00000000000..48e2a821806
--- /dev/null
+++ b/server/migrations/567_channel_typing_reaction_retry_idx.up.sql
@@ -0,0 +1 @@
+CREATE INDEX CONCURRENTLY IF NOT EXISTS channel_typing_reaction_retry_idx ON channel_typing_reaction (retry_after, chat_session_id) WHERE cleaned_at IS NULL;
diff --git a/server/migrations/568_channel_typing_reaction_gc_idx.down.sql b/server/migrations/568_channel_typing_reaction_gc_idx.down.sql
new file mode 100644
index 00000000000..235a935d287
--- /dev/null
+++ b/server/migrations/568_channel_typing_reaction_gc_idx.down.sql
@@ -0,0 +1 @@
+DROP INDEX CONCURRENTLY IF EXISTS channel_typing_reaction_gc_idx;
diff --git a/server/migrations/568_channel_typing_reaction_gc_idx.up.sql b/server/migrations/568_channel_typing_reaction_gc_idx.up.sql
new file mode 100644
index 00000000000..ec77fbac4f1
--- /dev/null
+++ b/server/migrations/568_channel_typing_reaction_gc_idx.up.sql
@@ -0,0 +1 @@
+CREATE INDEX CONCURRENTLY IF NOT EXISTS channel_typing_reaction_gc_idx ON channel_typing_reaction (cleaned_at) WHERE cleaned_at IS NOT NULL;
diff --git a/server/migrations/569_channel_typing_limits.down.sql b/server/migrations/569_channel_typing_limits.down.sql
new file mode 100644
index 00000000000..b179ea11049
--- /dev/null
+++ b/server/migrations/569_channel_typing_limits.down.sql
@@ -0,0 +1,2 @@
+ALTER TABLE channel_typing_reaction DROP COLUMN IF EXISTS quota_slot;
+ALTER TABLE channel_typing_reaction DROP COLUMN IF EXISTS abandoned_at;
diff --git a/server/migrations/569_channel_typing_limits.up.sql b/server/migrations/569_channel_typing_limits.up.sql
new file mode 100644
index 00000000000..44a21cdc582
--- /dev/null
+++ b/server/migrations/569_channel_typing_limits.up.sql
@@ -0,0 +1,18 @@
+ALTER TABLE channel_typing_reaction ADD COLUMN IF NOT EXISTS abandoned_at timestamptz;
+ALTER TABLE channel_typing_reaction ADD COLUMN IF NOT EXISTS quota_slot integer;
+
+-- Give existing pending rows a bounded share of each workspace's quota.
+-- Excess rows become explicit failed outcomes; never retain their credentials.
+WITH ranked AS (
+ SELECT id, row_number() OVER (PARTITION BY workspace_id ORDER BY created_at DESC, id) AS slot
+ FROM channel_typing_reaction WHERE cleaned_at IS NULL AND abandoned_at IS NULL
+)
+UPDATE channel_typing_reaction r SET
+ quota_slot = CASE WHEN ranked.slot <= 1000 THEN ranked.slot::integer END,
+ abandoned_at = CASE WHEN ranked.slot > 1000 THEN now() END,
+ cleanup_required = cleanup_required OR ranked.slot > 1000,
+ installation_snapshot = CASE WHEN ranked.slot > 1000 THEN '{}'::jsonb ELSE installation_snapshot END
+FROM ranked WHERE r.id = ranked.id;
+
+UPDATE channel_typing_reaction SET installation_snapshot = '{}'::jsonb
+WHERE cleaned_at IS NOT NULL;
diff --git a/server/migrations/570_channel_typing_quota_idx.down.sql b/server/migrations/570_channel_typing_quota_idx.down.sql
new file mode 100644
index 00000000000..c8dbcfbcfc4
--- /dev/null
+++ b/server/migrations/570_channel_typing_quota_idx.down.sql
@@ -0,0 +1 @@
+DROP INDEX CONCURRENTLY IF EXISTS channel_typing_reaction_quota_idx;
diff --git a/server/migrations/570_channel_typing_quota_idx.up.sql b/server/migrations/570_channel_typing_quota_idx.up.sql
new file mode 100644
index 00000000000..c176ff4a0ea
--- /dev/null
+++ b/server/migrations/570_channel_typing_quota_idx.up.sql
@@ -0,0 +1 @@
+CREATE UNIQUE INDEX CONCURRENTLY IF NOT EXISTS channel_typing_reaction_quota_idx ON channel_typing_reaction (workspace_id, quota_slot) WHERE cleaned_at IS NULL AND abandoned_at IS NULL;
diff --git a/server/migrations/571_channel_typing_expiry_idx.down.sql b/server/migrations/571_channel_typing_expiry_idx.down.sql
new file mode 100644
index 00000000000..63c1298bdc0
--- /dev/null
+++ b/server/migrations/571_channel_typing_expiry_idx.down.sql
@@ -0,0 +1 @@
+DROP INDEX CONCURRENTLY IF EXISTS channel_typing_reaction_expiry_idx;
diff --git a/server/migrations/571_channel_typing_expiry_idx.up.sql b/server/migrations/571_channel_typing_expiry_idx.up.sql
new file mode 100644
index 00000000000..63d489de3f9
--- /dev/null
+++ b/server/migrations/571_channel_typing_expiry_idx.up.sql
@@ -0,0 +1 @@
+CREATE INDEX CONCURRENTLY IF NOT EXISTS channel_typing_reaction_expiry_idx ON channel_typing_reaction (created_at, id) WHERE cleaned_at IS NULL AND abandoned_at IS NULL;
diff --git a/server/migrations/572_channel_typing_abandoned_idx.down.sql b/server/migrations/572_channel_typing_abandoned_idx.down.sql
new file mode 100644
index 00000000000..a521fff4826
--- /dev/null
+++ b/server/migrations/572_channel_typing_abandoned_idx.down.sql
@@ -0,0 +1 @@
+DROP INDEX CONCURRENTLY IF EXISTS channel_typing_reaction_abandoned_idx;
diff --git a/server/migrations/572_channel_typing_abandoned_idx.up.sql b/server/migrations/572_channel_typing_abandoned_idx.up.sql
new file mode 100644
index 00000000000..64edc5fee36
--- /dev/null
+++ b/server/migrations/572_channel_typing_abandoned_idx.up.sql
@@ -0,0 +1 @@
+CREATE INDEX CONCURRENTLY IF NOT EXISTS channel_typing_reaction_abandoned_idx ON channel_typing_reaction (abandoned_at) WHERE abandoned_at IS NOT NULL;
diff --git a/server/pkg/agent/antigravity.go b/server/pkg/agent/antigravity.go
index 49cfb2f94eb..11bd2a8acbc 100644
--- a/server/pkg/agent/antigravity.go
+++ b/server/pkg/agent/antigravity.go
@@ -185,8 +185,44 @@ type antigravityStreamEvent struct {
Result *antigravityStreamResult `json:"result"`
}
+// antigravityNetworkIssueError is the provider sentence some agy releases
+// report when a trailing round trip fails. It carries no Go error text, so it
+// has to stay a literal match alongside the transport patterns below.
const antigravityNetworkIssueError = "There was a network issue connecting to the server, please try again."
+// antigravityTransportErrorRe matches the causes Go's http client reports when
+// a round trip never produced a response: socket, DNS and TLS failures. These
+// are the strings that appear inside the `*url.Error` agy wraps as
+// `API error (attempt N): request failed: Post "...": `.
+//
+// Two things are deliberately excluded.
+//
+// - agy's own `request failed:` prefix. It is tempting to match it directly
+// since it marks an http.Client failure, but only one spelling has been
+// observed in the field, and nothing rules out agy reusing the same prefix
+// for an HTTP status error. Matching the cause keeps a provider rejection
+// from being read as transport noise.
+// - Provider-side rejections: quota, capacity, overload, policy and auth all
+// arrive as an HTTP response, so they are decisions about the request
+// rather than a failure to deliver it. Those must stay failures the user
+// sees instead of being smoothed over by a complete-looking answer —
+// reportTaskResult documents failing closed for exactly that reason.
+var antigravityTransportErrorRe = regexp.MustCompile(`(?i)(\bEOF\b|connection reset by peer|broken pipe|connection refused|connection timed out|i/o timeout|tls handshake timeout|tls: handshake failure|use of closed network connection|network is unreachable|no such host|server misbehaving|malformed HTTP response|http2: client connection lost|http2: server sent GOAWAY)`)
+
+// antigravityTrailingTransportError reports whether agy's provider error
+// describes a transport-level failure rather than a decision the provider made
+// about the request.
+func antigravityTrailingTransportError(providerError string) bool {
+ trimmed := strings.TrimSpace(providerError)
+ if trimmed == "" {
+ return false
+ }
+ if strings.EqualFold(trimmed, antigravityNetworkIssueError) {
+ return true
+ }
+ return antigravityTransportErrorRe.MatchString(trimmed)
+}
+
func (u antigravityStreamUsage) hasTokens() bool {
return u.InputTokens > 0 || u.OutputTokens > 0 || u.CacheReadTokens > 0 || u.CacheWriteTokens > 0
}
@@ -231,8 +267,15 @@ func antigravityResultStatus(status string) string {
}
}
+// antigravityCompletedDespiteTrailingNetworkError reports whether a turn that
+// agy ended in an error actually delivered a finished answer first. All three
+// conditions are required: the trailing failure has to be transport-level (a
+// provider-side rejection is a real failure), agy has to have handed back a
+// non-empty canonical response, and the latest agent_response step has to have
+// reached DONE — an ACTIVE step means the answer was still being written when
+// the connection went away.
func antigravityCompletedDespiteTrailingNetworkError(providerError, response string, agentResponseDone bool) bool {
- return strings.EqualFold(strings.TrimSpace(providerError), antigravityNetworkIssueError) &&
+ return antigravityTrailingTransportError(providerError) &&
strings.TrimSpace(response) != "" &&
agentResponseDone
}
diff --git a/server/pkg/agent/antigravity_test.go b/server/pkg/agent/antigravity_test.go
index 0fc5cd04d3b..b5527489634 100644
--- a/server/pkg/agent/antigravity_test.go
+++ b/server/pkg/agent/antigravity_test.go
@@ -845,6 +845,33 @@ exit 1
`
}
+// fakeAgyTrailingTransportErrorScript reproduces the real agy 1.2.13/1.2.14
+// sequence observed on 2026-09-30: the agent emits a complete DONE reply, agy
+// then makes one more streamGenerateContent call whose round trip never
+// produces a response, and reports the exhausted retries as
+// `API error (attempt N): request failed: Post "...": EOF`. This is the same
+// shape as the trailing network error above — a finished answer followed by a
+// transport failure — spelled differently, so it must be preserved too.
+func fakeAgyTrailingTransportErrorScript() string {
+ return `#!/bin/sh
+printf '%s\n' '{"event":"step_update","step_update":{"conversation_id":"67a6d8f2-8523-46fc-8fc9-87e630cbe295","step_index":1,"state":"DONE","step_type":"agent_response","text_delta":"Complete answer before the transport failure."}}'
+printf '%s\n' '{"event":"result","result":{"conversation_id":"67a6d8f2-8523-46fc-8fc9-87e630cbe295","status":"ERROR","response":"Complete answer before the transport failure.","error":"API error (attempt 3): request failed: Post \"https://daily-cloudcode-pa.googleapis.com/v1internal:streamGenerateContent?alt=sse\": EOF"}}'
+exit 1
+`
+}
+
+// fakeAgyTrailingNonTransportErrorScript is the negative control for the above:
+// the answer is complete and DONE, but the trailing failure is a provider-side
+// quota/capacity error rather than a transport one. That is a real failure the
+// user must see, so it must stay failed.
+func fakeAgyTrailingNonTransportErrorScript() string {
+ return `#!/bin/sh
+printf '%s\n' '{"event":"step_update","step_update":{"conversation_id":"77a6d8f2-8523-46fc-8fc9-87e630cbe295","step_index":1,"state":"DONE","step_type":"agent_response","text_delta":"Answer produced before quota ran out."}}'
+printf '%s\n' '{"event":"result","result":{"conversation_id":"77a6d8f2-8523-46fc-8fc9-87e630cbe295","status":"ERROR","response":"Answer produced before quota ran out.","error":"API error (attempt 3): model capacity exhausted, retry later"}}'
+exit 1
+`
+}
+
// fakeAgyTrailingNetworkErrorAfterNewPartialResponseScript guards against a
// completed earlier answer making a later, interrupted answer look complete.
func fakeAgyTrailingNetworkErrorAfterNewPartialResponseScript() string {
@@ -964,6 +991,181 @@ func TestAntigravityBackendIgnoresStaleActiveStepAfterDoneResponse(t *testing.T)
}
}
+// A complete DONE answer followed by an exhausted-retry transport error is a
+// finished turn. agy spells such a failure `API error (attempt N): request
+// failed: Post "...": EOF` rather than the network-issue sentence, and matching
+// only the latter discarded an answer that had already been produced and
+// streamed (2026-09-30: every turn of one conversation reported
+// blocked/agent_error.unknown while the reply sat in result.response).
+func TestAntigravityBackendIgnoresTrailingTransportErrorAfterDoneResponse(t *testing.T) {
+ t.Parallel()
+
+ fakePath := filepath.Join(t.TempDir(), "agy")
+ writeTestExecutable(t, fakePath, []byte(fakeAgyTrailingTransportErrorScript()))
+
+ backend, err := New("antigravity", Config{ExecutablePath: fakePath, Logger: quietAntigravityLogger()})
+ if err != nil {
+ t.Fatalf("new antigravity backend: %v", err)
+ }
+ session, err := backend.Execute(context.Background(), "prompt-ignored", ExecOptions{})
+ if err != nil {
+ t.Fatalf("execute: %v", err)
+ }
+ for range session.Messages {
+ }
+ result, ok := <-session.Result
+ if !ok {
+ t.Fatal("result channel closed without a value")
+ }
+ if result.Status != "completed" || result.Output != "Complete answer before the transport failure." || result.Error != "" {
+ t.Fatalf("result = status %q output %q error %q", result.Status, result.Output, result.Error)
+ }
+}
+
+// The negative control: a complete DONE answer does not on its own make a turn
+// successful. When the trailing failure is provider-side (quota, capacity,
+// policy, auth) rather than transport, the turn must stay failed so the user
+// sees the real cause instead of an answer presented as a finished result.
+func TestAntigravityBackendKeepsNonTransportFailureAfterDoneResponse(t *testing.T) {
+ t.Parallel()
+
+ fakePath := filepath.Join(t.TempDir(), "agy")
+ writeTestExecutable(t, fakePath, []byte(fakeAgyTrailingNonTransportErrorScript()))
+
+ backend, err := New("antigravity", Config{ExecutablePath: fakePath, Logger: quietAntigravityLogger()})
+ if err != nil {
+ t.Fatalf("new antigravity backend: %v", err)
+ }
+ session, err := backend.Execute(context.Background(), "prompt-ignored", ExecOptions{})
+ if err != nil {
+ t.Fatalf("execute: %v", err)
+ }
+ for range session.Messages {
+ }
+ result, ok := <-session.Result
+ if !ok {
+ t.Fatal("result channel closed without a value")
+ }
+ if result.Status != "failed" || result.Output != "Answer produced before quota ran out." || !strings.Contains(result.Error, "model capacity exhausted") {
+ t.Fatalf("result = status %q output %q error %q", result.Status, result.Output, result.Error)
+ }
+}
+
+// TestAntigravityTrailingTransportErrorClassification pins both edges of the
+// classifier. The "want true" half is the field-observed spelling plus the
+// other causes Go's http client reports for a round trip that never produced a
+// response. The "want false" half is the one that matters: a provider
+// rejection that happens to travel inside agy's `request failed:` wrapper must
+// stay a failure, otherwise a quota or overload error would be laundered into a
+// successful turn just because the model had already emitted some text.
+func TestAntigravityTrailingTransportErrorClassification(t *testing.T) {
+ t.Parallel()
+
+ tests := []struct {
+ name string
+ err string
+ want bool
+ }{
+ {
+ name: "field observed EOF after exhausted retries",
+ err: `API error (attempt 3): request failed: Post "https://daily-cloudcode-pa.googleapis.com/v1internal:streamGenerateContent?alt=sse": EOF`,
+ want: true,
+ },
+ {
+ name: "legacy provider sentence",
+ err: antigravityNetworkIssueError,
+ want: true,
+ },
+ {
+ name: "legacy provider sentence in other case and padding",
+ err: " THERE WAS A NETWORK ISSUE CONNECTING TO THE SERVER, PLEASE TRY AGAIN. ",
+ want: true,
+ },
+ {name: "unexpected EOF", err: `Post "https://host/v1": unexpected EOF`, want: true},
+ {name: "connection reset", err: `read tcp 1.2.3.4:443: connection reset by peer`, want: true},
+ {name: "broken pipe", err: `write tcp 1.2.3.4:443: broken pipe`, want: true},
+ {name: "connection refused", err: `dial tcp 1.2.3.4:443: connection refused`, want: true},
+ {name: "socket i/o timeout", err: `read tcp 1.2.3.4:443: i/o timeout`, want: true},
+ {name: "tls handshake timeout", err: `net/http: TLS handshake timeout`, want: true},
+ {name: "tls handshake failure", err: `remote error: tls: handshake failure`, want: true},
+ {name: "closed connection", err: `use of closed network connection`, want: true},
+ {name: "dns no such host", err: `dial tcp: lookup host: no such host`, want: true},
+ {name: "dns server misbehaving", err: `lookup host: server misbehaving`, want: true},
+ {name: "network unreachable", err: `connect: network is unreachable`, want: true},
+ {name: "malformed response", err: `malformed HTTP response`, want: true},
+ {name: "http2 connection lost", err: `http2: client connection lost`, want: true},
+ {name: "http2 goaway", err: `http2: server sent GOAWAY and closed the connection`, want: true},
+
+ {name: "empty", err: "", want: false},
+ {name: "whitespace only", err: " ", want: false},
+ {
+ name: "quota rejection inside the request-failed wrapper",
+ err: `API error (attempt 3): request failed: 429 Too Many Requests`,
+ want: false,
+ },
+ {
+ name: "status error inside the request-failed wrapper",
+ err: `API error (attempt 3): request failed: unexpected status code 503`,
+ want: false,
+ },
+ {name: "model capacity exhausted", err: `API error (attempt 3): model capacity exhausted, retry later`, want: false},
+ {name: "prefill queue overloaded", err: `PREFILL_QUEUE_OVERLOADED: Overloaded`, want: false},
+ {name: "resource exhausted", err: `RESOURCE_EXHAUSTED: Quota exceeded for quota group`, want: false},
+ {name: "safety policy", err: `The request was rejected by the safety filter`, want: false},
+ {name: "auth", err: `unauthenticated: invalid API key`, want: false},
+ {name: "invalid argument", err: `INVALID_ARGUMENT: request contains an invalid argument`, want: false},
+ {name: "permission denied", err: `permission denied`, want: false},
+ {name: "tool crash", err: `agent executor error: tool crashed`, want: false},
+ {name: "bare status text", err: `agy returned status ERROR`, want: false},
+ {name: "cancelled", err: `execution cancelled`, want: false},
+ }
+
+ for _, tc := range tests {
+ t.Run(tc.name, func(t *testing.T) {
+ t.Parallel()
+ if got := antigravityTrailingTransportError(tc.err); got != tc.want {
+ t.Fatalf("antigravityTrailingTransportError(%q) = %v, want %v", tc.err, got, tc.want)
+ }
+ })
+ }
+}
+
+// The classifier must not fire on a complete answer alone. Each of these has a
+// non-empty response and a DONE agent_response step, so the error class is the
+// only thing standing between a preserved answer and a laundered failure.
+func TestAntigravityCompletedDespiteTrailingNetworkErrorRequiresAllThree(t *testing.T) {
+ t.Parallel()
+
+ const transport = `API error (attempt 3): request failed: Post "https://host/v1": EOF`
+ const quota = `API error (attempt 3): model capacity exhausted`
+
+ tests := []struct {
+ name string
+ err string
+ response string
+ agentResponseDone bool
+ want bool
+ }{
+ {name: "all three hold", err: transport, response: "answer", agentResponseDone: true, want: true},
+ {name: "response whitespace only", err: transport, response: " \n ", agentResponseDone: true, want: false},
+ {name: "response empty", err: transport, response: "", agentResponseDone: true, want: false},
+ {name: "answer still being written", err: transport, response: "partial", agentResponseDone: false, want: false},
+ {name: "provider rejection", err: quota, response: "answer", agentResponseDone: true, want: false},
+ {name: "no error at all", err: "", response: "answer", agentResponseDone: true, want: false},
+ }
+
+ for _, tc := range tests {
+ t.Run(tc.name, func(t *testing.T) {
+ t.Parallel()
+ got := antigravityCompletedDespiteTrailingNetworkError(tc.err, tc.response, tc.agentResponseDone)
+ if got != tc.want {
+ t.Fatalf("antigravityCompletedDespiteTrailingNetworkError(%q, %q, %v) = %v, want %v",
+ tc.err, tc.response, tc.agentResponseDone, got, tc.want)
+ }
+ })
+ }
+}
+
func TestAntigravityBackendKeepsNetworkFailureForPartialResponse(t *testing.T) {
t.Parallel()
diff --git a/server/pkg/agent/kimi.go b/server/pkg/agent/kimi.go
index 154334a6299..92cb5d66d11 100644
--- a/server/pkg/agent/kimi.go
+++ b/server/pkg/agent/kimi.go
@@ -1,9 +1,11 @@
package agent
import (
+ "bufio"
"bytes"
"context"
"encoding/json"
+ "errors"
"fmt"
"io"
"os"
@@ -520,6 +522,31 @@ func (b *kimiBackend) Execute(ctx context.Context, prompt string, opts ExecOptio
}
}
+ // kimi's own wire log is authoritative on turn outcome: its ACP
+ // adapter can answer session/prompt successfully while the turn's
+ // `agent.turn.ended` record says failed — a bare
+ // "OAuthConnectionError: ..." carries none of the stderr sniffer's
+ // terminal signatures (#9054). Without this flip the run publishes
+ // its partial narration as a successfully delivered result and the
+ // retry/continuation policy never sees the failure. Only a
+ // completed run is promoted: timeout/abort/already-failed keep
+ // their own terminal semantics.
+ if finalStatus == "completed" {
+ if turnErr, failed := scanKimiMainTurnFailure(kimiUsageScan{
+ startTime: startTime,
+ kimiHome: b.cfg.Env["KIMI_CODE_HOME"],
+ sessionID: sessionID,
+ resumed: opts.ResumeSessionID != "",
+ fallbackModel: fallbackModel,
+ }); failed {
+ finalStatus = "failed"
+ finalError = "kimi turn failed"
+ if turnErr != "" {
+ finalError = fmt.Sprintf("kimi turn failed: %s", turnErr)
+ }
+ }
+ }
+
resCh <- Result{
Status: finalStatus,
Output: finalOutput,
@@ -759,6 +786,93 @@ func kimiWireRecordInTurn(recordTimeMillis int64, startTime time.Time, resumed b
return recordTimeMillis >= startTime.UnixMilli()
}
+// kimiTurnEndedRecordType is the wire record kimi-code appends when a turn
+// terminates, with its own verdict in `outcome`.
+const kimiTurnEndedRecordType = "agent.turn.ended"
+
+// kimiWireTurnEnd is the parsed shape of one `agent.turn.ended` record.
+// Turn ids alternate between number and string across kimi builds, so they
+// are deliberately not modeled — only the outcome and diagnostic matter.
+type kimiWireTurnEnd struct {
+ Type string `json:"type"`
+ Outcome string `json:"outcome"`
+ ErrorMessage string `json:"errorMessage"`
+ Time int64 `json:"time"`
+}
+
+// scanKimiMainTurnFailure reports whether the main agent's wire log recorded
+// a failed turn for this run, carrying kimi's own diagnostic message. It
+// reads the same wire logs with the same turn-boundary rules as
+// scanKimiSessionUsage, and only the structured `outcome:"failed"` verdict
+// counts — conversation text merely echoing failure words must not flip a
+// healthy run (see the anchoring lesson in #7920). A delegated agent's
+// failed turn is that agent's own task path; only the main agent's terminal
+// event fails this run.
+func scanKimiMainTurnFailure(scan kimiUsageScan) (string, bool) {
+ root := kimiSessionRoot(scan.kimiHome)
+ if root == "" {
+ return "", false
+ }
+ for _, path := range kimiSessionWireLogs(root, scan.sessionID) {
+ if filepath.Base(filepath.Dir(path)) != "main" {
+ continue
+ }
+ if detail, failed := kimiWireTurnFailure(path, scan); failed {
+ return detail, true
+ }
+ }
+ return "", false
+}
+
+// kimiWireTurnFailure walks one wire log for a failed main turn. A malformed
+// or truncated line is skipped: the log is appended to live, so reporting
+// what is readable beats failing the scan (same policy as the usage scan).
+// A record beyond the scan bound is discarded rather than ending the walk —
+// an oversized context.append ahead of the terminal event must not hide the
+// failed outcome (review on #9057).
+func kimiWireTurnFailure(path string, scan kimiUsageScan) (string, bool) {
+ file, err := os.Open(path)
+ if err != nil {
+ return "", false
+ }
+ defer file.Close()
+
+ detail := ""
+ found := false
+ reader := bufio.NewReaderSize(file, agentStreamInitialBufferBytes)
+ for {
+ line, err := readAgentStreamLine(reader)
+ if err != nil && !errors.Is(err, bufio.ErrTooLong) && !errors.Is(err, io.EOF) {
+ // Any other read failure ends the walk; the records read so
+ // far are the answer (same best-effort policy as a truncated
+ // tail above).
+ break
+ }
+ if line != nil && bytes.Contains(line, []byte(kimiTurnEndedRecordType)) {
+ var record kimiWireTurnEnd
+ if err := json.Unmarshal(line, &record); err != nil || record.Type != kimiTurnEndedRecordType {
+ continue
+ }
+ if !kimiWireRecordInTurn(record.Time, scan.startTime, scan.resumed) {
+ continue
+ }
+ if record.Outcome == "failed" {
+ // kimi's diagnostic is provider-authored text: a JSON
+ // credential in it (request={"api_key":...}) survives both
+ // the raw string and the shared redact.Text, so it takes
+ // the same sanitize pass as every child-process diagnostic
+ // before this reaches Result.Error (review on #9057).
+ detail = sanitizeAgentDiagnostic(record.ErrorMessage)
+ found = true
+ }
+ }
+ if errors.Is(err, io.EOF) {
+ break
+ }
+ }
+ return detail, found
+}
+
// kimiSessionRoot resolves kimi's session directory: an explicitly configured
// home first, then the ambient KIMI_CODE_HOME, then ~/.kimi-code. Mirrors
// codexSessionRoot.
diff --git a/server/pkg/agent/kimi_test.go b/server/pkg/agent/kimi_test.go
index 7bcb365040e..aeed36addad 100644
--- a/server/pkg/agent/kimi_test.go
+++ b/server/pkg/agent/kimi_test.go
@@ -1587,3 +1587,290 @@ func TestKimiSessionWireLogsRejectsTraversal(t *testing.T) {
}
}
}
+
+// kimiWireTurnEndedLine builds an `agent.turn.ended` wire record like the ones
+// kimi-code appends to agents/main/wire.jsonl when a turn terminates.
+func kimiWireTurnEndedLine(outcome, errorMessage string, at time.Time) string {
+ return fmt.Sprintf(`{"turnId":0,"outcome":%q,"errorMessage":%q,"type":"agent.turn.ended","time":%d,"kind":"event"}`,
+ outcome, errorMessage, at.UnixMilli())
+}
+
+// TestScanKimiMainTurnFailureFlipsOnFailedTerminalEvent covers #9054: kimi's
+// ACP adapter answered session/prompt successfully while its own wire log
+// recorded the main turn as failed (a bare OAuthConnectionError no stderr
+// sniffer signature matches). The persisted outcome is authoritative.
+func TestScanKimiMainTurnFailureFlipsOnFailedTerminalEvent(t *testing.T) {
+ t.Parallel()
+
+ home := t.TempDir()
+ now := time.Now()
+ writeKimiWireLog(t, home, "session_kf", "main",
+ `{"type":"metadata","protocol_version":"1"}`,
+ kimiWireTurnEndedLine("failed",
+ "OAuthConnectionError: OAuth request to https://auth.kimi.com/api/oauth/token failed: fetch failed: Connect Timeout Error",
+ now),
+ )
+
+ detail, failed := scanKimiMainTurnFailure(kimiUsageScan{
+ startTime: now.Add(-time.Minute), kimiHome: home,
+ sessionID: "session_kf", fallbackModel: "unknown",
+ })
+ if !failed {
+ t.Fatal("expected the failed main-agent terminal event to be reported")
+ }
+ if !strings.Contains(detail, "OAuthConnectionError") {
+ t.Errorf("detail should carry kimi's diagnostic, got %q", detail)
+ }
+}
+
+// TestScanKimiMainTurnFailureSanitizesWireDiagnostic covers the review on
+// #9057: kimi's diagnostic is provider-authored text, and a JSON credential
+// embedded in it (request={"api_key":...}) survives both the raw string and
+// the shared redact.Text — the diagnostic must take the same
+// sanitizeAgentDiagnostic pass as every child-process diagnostic that
+// reaches Result.Error.
+func TestScanKimiMainTurnFailureSanitizesWireDiagnostic(t *testing.T) {
+ t.Parallel()
+
+ home := t.TempDir()
+ now := time.Now()
+ const fakeSecret = "review-only-fake-secret"
+ writeKimiWireLog(t, home, "session_secret", "main",
+ kimiWireTurnEndedLine("failed",
+ `request={"api_key":"`+fakeSecret+`"} while calling https://api.kimi.com`,
+ now),
+ )
+
+ detail, failed := scanKimiMainTurnFailure(kimiUsageScan{
+ startTime: now.Add(-time.Minute), kimiHome: home,
+ sessionID: "session_secret", fallbackModel: "unknown",
+ })
+ if !failed {
+ t.Fatal("expected the failed main-agent terminal event to be reported")
+ }
+ if strings.Contains(detail, fakeSecret) {
+ t.Errorf("secret leaked through the wire diagnostic: %q", detail)
+ }
+ if !strings.Contains(detail, `api_key":"[REDACTED]"`) {
+ t.Errorf("expected the JSON secret field redacted, got %q", detail)
+ }
+}
+
+// TestScanKimiMainTurnFailureIgnoresHealthyStaleAndDelegated guards the flip
+// against false positives: a successful main turn stays completed, a failed
+// record from a previous task on a resumed session is outside the turn
+// boundary, and a delegated agent's failure is that agent's own task path —
+// only the main agent's terminal event fails the run.
+func TestScanKimiMainTurnFailureIgnoresHealthyStaleAndDelegated(t *testing.T) {
+ t.Parallel()
+
+ home := t.TempDir()
+ now := time.Now()
+
+ writeKimiWireLog(t, home, "session_ok", "main", kimiWireTurnEndedLine("success", "", now))
+ writeKimiWireLog(t, home, "session_del", "main", kimiWireTurnEndedLine("success", "", now))
+ writeKimiWireLog(t, home, "session_del", "researcher", kimiWireTurnEndedLine("failed", "delegated boom", now))
+ writeKimiWireLog(t, home, "session_stale", "main", kimiWireTurnEndedLine("failed", "stale boom", now.Add(-time.Hour)))
+
+ for _, tc := range []struct {
+ sessionID string
+ resumed bool
+ }{
+ {sessionID: "session_ok"},
+ {sessionID: "session_del"},
+ {sessionID: "session_stale", resumed: true},
+ } {
+ if _, failed := scanKimiMainTurnFailure(kimiUsageScan{
+ startTime: now.Add(-time.Minute), kimiHome: home,
+ sessionID: tc.sessionID, resumed: tc.resumed, fallbackModel: "unknown",
+ }); failed {
+ t.Errorf("session %q: expected no main-turn failure", tc.sessionID)
+ }
+ }
+}
+
+// TestScanKimiMainTurnFailureReadsPastOversizedRecord covers the review on
+// #9057: newAgentStreamScanner stops for good at a line beyond
+// agentStreamMaxLineBytes and the old scan loop never checked scanner.Err(),
+// so a valid but oversized context.append record ahead of the terminal event
+// hid the failed outcome and kept the run completed. The scan must discard
+// the oversized record and keep reading.
+func TestScanKimiMainTurnFailureReadsPastOversizedRecord(t *testing.T) {
+ t.Parallel()
+
+ home := t.TempDir()
+ now := time.Now()
+ oversized := `{"type":"context.append","content":"` + strings.Repeat("x", agentStreamMaxLineBytes) + `"}`
+ writeKimiWireLog(t, home, "session_big", "main",
+ oversized,
+ kimiWireTurnEndedLine("failed", "boom after the oversized record", now),
+ )
+
+ detail, failed := scanKimiMainTurnFailure(kimiUsageScan{
+ startTime: now.Add(-time.Minute), kimiHome: home,
+ sessionID: "session_big", fallbackModel: "unknown",
+ })
+ if !failed {
+ t.Fatal("expected the failed terminal event past the oversized record to be reported")
+ }
+ if !strings.Contains(detail, "boom after the oversized record") {
+ t.Errorf("detail should carry the diagnostic, got %q", detail)
+ }
+}
+
+// TestScanKimiMainTurnFailureWithoutWireLog: no session directory (scan ran
+// before kimi wrote anything, or the home is unset) must be a no-op.
+func TestScanKimiMainTurnFailureWithoutWireLog(t *testing.T) {
+ t.Parallel()
+
+ if _, failed := scanKimiMainTurnFailure(kimiUsageScan{
+ startTime: time.Now().Add(-time.Minute), kimiHome: t.TempDir(),
+ sessionID: "session_absent", fallbackModel: "unknown",
+ }); failed {
+ t.Fatal("expected no failure without a wire log")
+ }
+}
+
+// fakeKimiWireTurnFailureScript mimics the #9054 incident shape: the ACP
+// layer answers session/prompt successfully (end_turn) after emitting real
+// narration, while kimi's own wire log records the main turn as failed with
+// a bare OAuthConnectionError that matches no stderr sniffer signature.
+func fakeKimiWireTurnFailureScript(sessionID string) string {
+ script := `#!/bin/sh
+while IFS= read -r line; do
+ id=$(printf '%s' "$line" | sed -n 's/.*"id":\([0-9]*\).*/\1/p')
+ case "$line" in
+ *'"method":"initialize"'*)
+ printf '{"jsonrpc":"2.0","id":%s,"result":{"protocolVersion":1,"agentCapabilities":{}}}\n' "$id"
+ ;;
+ *'"method":"session/new"'*)
+ printf '{"jsonrpc":"2.0","id":%s,"result":{"sessionId":"ses_wirefail"}}\n' "$id"
+ ;;
+ *'"method":"session/prompt"'*)
+ printf '{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"ses_wirefail","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"partial narration before the failure"}}}}\n'
+ dir="$KIMI_CODE_HOME/sessions/wd_test_abc123/ses_wirefail/agents/main"
+ mkdir -p "$dir"
+ # time:0 = untimed record; kimiWireRecordInTurn counts those on fresh
+ # runs, and a shell cannot produce millisecond timestamps portably.
+ printf '{"turnId":0,"outcome":"failed","errorMessage":"OAuthConnectionError: OAuth request to https://auth.kimi.com/api/oauth/token failed: fetch failed: Connect Timeout Error","type":"agent.turn.ended","time":0,"kind":"event"}\n' >> "$dir/wire.jsonl"
+ printf '{"jsonrpc":"2.0","id":%s,"result":{"stopReason":"end_turn"}}\n' "$id"
+ exit 0
+ ;;
+ esac
+done
+`
+ return strings.ReplaceAll(script, "ses_wirefail", sessionID)
+}
+
+// TestKimiBackendWireTurnFailureFailsCompletedRun is the end-to-end guard for
+// #9054: a successful ACP prompt response with nonempty narration must not
+// publish the run as completed when kimi's wire log says the main turn
+// failed. The partial narration stays in the transcript; the status and the
+// sanitized diagnostic must reach the daemon so retry policy sees the truth.
+func TestKimiBackendWireTurnFailureFailsCompletedRun(t *testing.T) {
+ t.Parallel()
+
+ home := t.TempDir()
+ fakePath := filepath.Join(t.TempDir(), "kimi")
+ writeTestExecutable(t, fakePath, []byte(fakeKimiWireTurnFailureScript("ses_wirefail")))
+
+ backend, err := New("kimi", Config{
+ ExecutablePath: fakePath,
+ Logger: slog.Default(),
+ Env: map[string]string{"KIMI_CODE_HOME": home},
+ })
+ if err != nil {
+ t.Fatalf("new kimi backend: %v", err)
+ }
+
+ ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
+ defer cancel()
+
+ session, err := backend.Execute(ctx, "prompt-ignored", ExecOptions{Timeout: 5 * time.Second})
+ if err != nil {
+ t.Fatalf("execute: %v", err)
+ }
+ go func() {
+ for range session.Messages {
+ }
+ }()
+
+ select {
+ case result, ok := <-session.Result:
+ if !ok {
+ t.Fatal("result channel closed without a value")
+ }
+ if result.Status != "failed" {
+ t.Fatalf("expected status=failed from the wire turn outcome, got %q (error=%q)", result.Status, result.Error)
+ }
+ if !strings.Contains(result.Error, "OAuthConnectionError") {
+ t.Errorf("expected kimi's diagnostic in the error, got %q", result.Error)
+ }
+ if !strings.Contains(result.Output, "partial narration") {
+ t.Errorf("partial narration should stay in the transcript, got %q", result.Output)
+ }
+ if result.ResumeRejected {
+ t.Error("an OAuth transport failure is not a poisoned history; ResumeRejected must stay false")
+ }
+ case <-time.After(10 * time.Second):
+ t.Fatal("timeout waiting for result")
+ }
+}
+
+// TestKimiBackendWireTurnFailurePastOversizedRecord reproduces the review's
+// offline repro on #9057: a valid context.append record at the scan's 32 MiB
+// bound sits ahead of the failed agent.turn.ended record. With the old
+// scanner loop — which stops for good at an over-bound line and never checks
+// scanner.Err() — the backend reported Status="completed", Error="".
+func TestKimiBackendWireTurnFailurePastOversizedRecord(t *testing.T) {
+ t.Parallel()
+
+ home := t.TempDir()
+ dir := filepath.Join(home, "sessions", "wd_test_abc123", "ses_big", "agents", "main")
+ if err := os.MkdirAll(dir, 0o755); err != nil {
+ t.Fatalf("mkdir wire log dir: %v", err)
+ }
+ oversized := `{"type":"context.append","content":"` + strings.Repeat("x", agentStreamMaxLineBytes) + `"}`
+ if err := os.WriteFile(filepath.Join(dir, "wire.jsonl"), []byte(oversized+"\n"), 0o644); err != nil {
+ t.Fatalf("write oversized record: %v", err)
+ }
+
+ fakePath := filepath.Join(t.TempDir(), "kimi")
+ writeTestExecutable(t, fakePath, []byte(fakeKimiWireTurnFailureScript("ses_big")))
+
+ backend, err := New("kimi", Config{
+ ExecutablePath: fakePath,
+ Logger: slog.Default(),
+ Env: map[string]string{"KIMI_CODE_HOME": home},
+ })
+ if err != nil {
+ t.Fatalf("new kimi backend: %v", err)
+ }
+
+ ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
+ defer cancel()
+
+ session, err := backend.Execute(ctx, "prompt-ignored", ExecOptions{Timeout: 5 * time.Second})
+ if err != nil {
+ t.Fatalf("execute: %v", err)
+ }
+ go func() {
+ for range session.Messages {
+ }
+ }()
+
+ select {
+ case result, ok := <-session.Result:
+ if !ok {
+ t.Fatal("result channel closed without a value")
+ }
+ if result.Status != "failed" {
+ t.Fatalf("expected status=failed past the oversized record, got %q (error=%q)", result.Status, result.Error)
+ }
+ if !strings.Contains(result.Error, "OAuthConnectionError") {
+ t.Errorf("expected kimi's diagnostic in the error, got %q", result.Error)
+ }
+ case <-time.After(10 * time.Second):
+ t.Fatal("timeout waiting for result")
+ }
+}
diff --git a/server/pkg/agent/omp_session.go b/server/pkg/agent/omp_session.go
new file mode 100644
index 00000000000..dacc9bbd38d
--- /dev/null
+++ b/server/pkg/agent/omp_session.go
@@ -0,0 +1,63 @@
+package agent
+
+import (
+ "fmt"
+ "log/slog"
+ "os"
+ "path/filepath"
+ "strings"
+)
+
+// OMP aliases --session to --resume. All session selection and persistence
+// flags must remain daemon-owned, including on fresh runs using --session-dir.
+var ompSessionArgs = map[string]blockedArgMode{
+ "--session": blockedOptionalValue,
+ "--session-dir": blockedWithValue,
+ "--no-session": blockedStandalone,
+ "--continue": blockedStandalone,
+ "-c": blockedStandalone,
+ "--resume": blockedOptionalValue,
+ "-r": blockedOptionalValue,
+ "--fork": blockedWithValue,
+ "--session-id": blockedWithValue,
+}
+
+func buildOmpArgs(sessionPath, sessionDir string, opts ExecOptions, logger *slog.Logger) []string {
+ opts.CustomArgs = filterCustomArgs(opts.CustomArgs, ompSessionArgs, logger)
+ args := buildPiArgs(sessionPath, opts, logger)
+ if sessionDir != "" {
+ args = append(args, "--session-dir", sessionDir)
+ }
+ return args
+}
+
+// Each fresh execution owns a private directory. OMP chooses the filename;
+// discover it after exit instead of depending on its timestamp/ID naming
+// convention or moving a transcript away from its companion artifacts.
+func findOmpSessionFile(dir string) (string, error) {
+ entries, err := os.ReadDir(dir)
+ if err != nil {
+ return "", err
+ }
+ var path string
+ for _, entry := range entries {
+ if !entry.Type().IsRegular() || !strings.HasSuffix(entry.Name(), ".jsonl") {
+ continue
+ }
+ info, err := entry.Info()
+ if err != nil {
+ return "", err
+ }
+ if info.Size() == 0 {
+ continue
+ }
+ if path != "" {
+ return "", fmt.Errorf("multiple transcripts in fresh session directory %q", dir)
+ }
+ path = filepath.Join(dir, entry.Name())
+ }
+ if path == "" {
+ return "", fmt.Errorf("no persisted transcript in fresh session directory %q", dir)
+ }
+ return path, nil
+}
diff --git a/server/pkg/agent/omp_session_integration_test.go b/server/pkg/agent/omp_session_integration_test.go
new file mode 100644
index 00000000000..cb21198c3d0
--- /dev/null
+++ b/server/pkg/agent/omp_session_integration_test.go
@@ -0,0 +1,86 @@
+//go:build agentintegration
+
+package agent
+
+import (
+ "bytes"
+ "context"
+ "encoding/json"
+ "log/slog"
+ "os"
+ "os/exec"
+ "path/filepath"
+ "strings"
+ "testing"
+ "time"
+)
+
+// TestOmpSessionStartupWithoutModelCall checks the real CLI's session contract
+// without sending a prompt. Use an explicitly selected OMP binary and an empty
+// home/config directory; no user account or provider credential is consulted.
+func TestOmpSessionStartupWithoutModelCall(t *testing.T) {
+ if os.Getenv("MULTICA_RUN_REAL_AGENT_SMOKE") != "1" {
+ t.Skip("set MULTICA_RUN_REAL_AGENT_SMOKE=1 to run real-agent smoke tests")
+ }
+ binary := os.Getenv("MULTICA_OMP_SMOKE_EXECUTABLE")
+ if binary == "" {
+ t.Skip("set MULTICA_OMP_SMOKE_EXECUTABLE to the OMP binary to verify")
+ }
+ home := t.TempDir()
+ cwd := t.TempDir()
+ env := []string{
+ "PATH=" + os.Getenv("PATH"), "HOME=" + home,
+ "USERPROFILE=" + home, "TMPDIR=" + t.TempDir(),
+ "PI_CODING_AGENT_DIR=" + filepath.Join(home, "agent"),
+ // Select a catalog model without requiring a stored account. With no
+ // prompt, print mode never enters session.prompt or calls the provider.
+ "ANTHROPIC_API_KEY=unused-no-model-call",
+ }
+ opts := ExecOptions{Model: "anthropic/claude-sonnet-4-20250514", CustomArgs: []string{"--no-extensions", "--no-skills", "--no-title"}}
+ run := func(args []string) (string, string, error) {
+ t.Helper()
+ ctx, cancel := context.WithTimeout(t.Context(), 45*time.Second)
+ defer cancel()
+ cmd := exec.CommandContext(ctx, binary, args...)
+ cmd.Dir, cmd.Env = cwd, env
+ cmd.Stdin = strings.NewReader("")
+ var stdout, stderr bytes.Buffer
+ cmd.Stdout, cmd.Stderr = &stdout, &stderr
+ err := cmd.Run()
+ return stdout.String(), stderr.String(), err
+ }
+ empty := filepath.Join(cwd, "empty.jsonl")
+ if err := os.WriteFile(empty, nil, 0o600); err != nil {
+ t.Fatal(err)
+ }
+ _, stderr, err := run(buildOmpArgs(empty, "", opts, slog.Default()))
+ if err == nil || !strings.Contains(stderr, "session file holds no entries") {
+ t.Fatalf("expected OMP's empty-resume refusal: %v, %s", err, stderr)
+ }
+ stdout, stderr, err := run(buildOmpArgs("", t.TempDir(), opts, slog.Default()))
+ if err != nil {
+ t.Fatalf("fresh startup: %v, %s", err, stderr)
+ }
+ var header string
+ for _, line := range strings.Split(stdout, "\n") {
+ var event struct {
+ Type string `json:"type"`
+ }
+ if json.Unmarshal([]byte(line), &event) == nil && event.Type == "session" {
+ header = line
+ break
+ }
+ }
+ if header == "" {
+ t.Fatalf("fresh startup did not emit a session header: %s", stdout)
+ }
+ // A prompt-free invocation may discard its draft. Persist its native header
+ // explicitly to verify path-based resume without generating a model turn.
+ valid := filepath.Join(cwd, "valid.jsonl")
+ if err := os.WriteFile(valid, []byte(header+"\n"), 0o600); err != nil {
+ t.Fatal(err)
+ }
+ if _, stderr, err := run(buildOmpArgs(valid, "", opts, slog.Default())); err != nil {
+ t.Fatalf("valid resume startup: %v, %s", err, stderr)
+ }
+}
diff --git a/server/pkg/agent/omp_session_test.go b/server/pkg/agent/omp_session_test.go
new file mode 100644
index 00000000000..c081e17c204
--- /dev/null
+++ b/server/pkg/agent/omp_session_test.go
@@ -0,0 +1,60 @@
+package agent
+
+import (
+ "log/slog"
+ "os"
+ "path/filepath"
+ "slices"
+ "testing"
+)
+
+func TestBuildOmpArgsOwnsSessionSelection(t *testing.T) {
+ for _, resume := range []string{"", "/saved.jsonl"} {
+ dir := ""
+ want := []string{"-p", "--mode", "json", "--tools", "read"}
+ if resume == "" {
+ dir = "/fresh"
+ want = append(want, "--session-dir", dir)
+ } else {
+ want = []string{"-p", "--mode", "json", "--session", resume, "--tools", "read"}
+ }
+ args := buildOmpArgs(resume, dir, ExecOptions{CustomArgs: []string{
+ "--session-dir", "/other", "--session-dir=/other", "--no-session",
+ "--session", "--continue", "-c", "--resume=other", "-r", "other",
+ "--fork", "other", "--session-id", "other", "--tools", "read",
+ }}, slog.Default())
+ if !slices.Equal(args, want) {
+ t.Fatalf("resume %q: args = %v, want %v", resume, args, want)
+ }
+ }
+}
+
+func TestFindOmpSessionFile(t *testing.T) {
+ for _, tc := range []struct {
+ name string
+ files map[string]string
+ want string
+ }{
+ {"missing", nil, ""},
+ {"empty", map[string]string{"empty.jsonl": ""}, ""},
+ {"persisted", map[string]string{"chosen.jsonl": "{}\n", "metadata.json": "{}", "empty.jsonl": ""}, "chosen.jsonl"},
+ {"ambiguous", map[string]string{"a.jsonl": "{}\n", "b.jsonl": "{}\n"}, ""},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ dir := t.TempDir()
+ for name, data := range tc.files {
+ if err := os.WriteFile(filepath.Join(dir, name), []byte(data), 0o600); err != nil {
+ t.Fatal(err)
+ }
+ }
+ path, err := findOmpSessionFile(dir)
+ if tc.want == "" {
+ if err == nil || path != "" {
+ t.Fatalf("expected no resumable path, got %q, %v", path, err)
+ }
+ } else if err != nil || path != filepath.Join(dir, tc.want) {
+ t.Fatalf("got %q, %v", path, err)
+ }
+ })
+ }
+}
diff --git a/server/pkg/agent/omp_session_unix_test.go b/server/pkg/agent/omp_session_unix_test.go
new file mode 100644
index 00000000000..c2848820267
--- /dev/null
+++ b/server/pkg/agent/omp_session_unix_test.go
@@ -0,0 +1,106 @@
+//go:build unix
+
+package agent
+
+import (
+ "log/slog"
+ "os"
+ "path/filepath"
+ "strings"
+ "testing"
+ "time"
+)
+
+func TestOmpFreshSessionThenResume(t *testing.T) {
+ t.Setenv("HOME", t.TempDir())
+ // Model-free reproduction of OMP 18.8.3's contract: --session rejects
+ // missing/empty files; --session-dir creates and persists a new transcript.
+ fake := filepath.Join(t.TempDir(), "omp")
+ writeTestExecutable(t, fake, []byte(`#!/bin/sh
+while [ "$#" -gt 0 ]; do
+ case "$1" in
+ --session) session="$2"; shift ;;
+ --session-dir) dir="$2"; shift ;;
+ esac
+ shift
+done
+if [ -n "$session" ]; then
+ [ -s "$session" ] || exit 1
+else
+ [ -d "$dir" ] || exit 2
+ session="$dir/runtime-chosen.jsonl"
+ printf '%s\n' '{"type":"session","id":"test"}' > "$session"
+fi
+cat >> "$session"
+printf '\n' >> "$session"
+printf '%s\n' '{"type":"agent_end"}'
+`))
+ backend, err := ResolveBackend("omp", Config{ExecutablePath: fake, Logger: slog.Default()})
+ if err != nil {
+ t.Fatal(err)
+ }
+ execute := func(prompt, resume string) Result {
+ t.Helper()
+ session, err := backend.Execute(t.Context(), prompt, ExecOptions{ResumeSessionID: resume, Timeout: 5 * time.Second})
+ if err != nil {
+ t.Fatal(err)
+ }
+ for range session.Messages {
+ }
+ return <-session.Result
+ }
+ var previous string
+ for _, prompt := range []string{"fresh chat", "quick-create task"} {
+ first := execute(prompt, "")
+ if first.Status != "completed" || first.SessionID == "" || first.SessionID == previous {
+ t.Fatalf("fresh result: %+v", first)
+ }
+ previous = first.SessionID
+ second := execute("follow-up", first.SessionID)
+ if second.Status != "completed" || second.SessionID != first.SessionID {
+ t.Fatalf("resume result: %+v", second)
+ }
+ data, err := os.ReadFile(second.SessionID)
+ if err != nil || !strings.Contains(string(data), prompt+"\nfollow-up\n") {
+ t.Fatalf("history not preserved: %q, %v", data, err)
+ }
+ }
+ // Old zero-byte transcripts must not be fabricated into valid resumes.
+ empty := filepath.Join(t.TempDir(), "empty.jsonl")
+ if err := os.WriteFile(empty, nil, 0o600); err != nil {
+ t.Fatal(err)
+ }
+ if result := execute("follow-up", empty); result.Status != "failed" {
+ t.Fatalf("empty resume unexpectedly succeeded: %+v", result)
+ }
+ if info, err := os.Stat(empty); err != nil || info.Size() != 0 {
+ t.Fatalf("empty resume was modified: %v, %v", info, err)
+ }
+ missing := filepath.Join(t.TempDir(), "missing.jsonl")
+ if _, err := backend.Execute(t.Context(), "follow-up", ExecOptions{ResumeSessionID: missing}); err == nil {
+ t.Fatal("missing resume unexpectedly started")
+ }
+ if _, err := os.Stat(missing); !os.IsNotExist(err) {
+ t.Fatalf("missing resume was created: %v", err)
+ }
+}
+
+func TestOmpFreshSessionRequiresPersistedTranscript(t *testing.T) {
+ t.Setenv("HOME", t.TempDir())
+ fake := filepath.Join(t.TempDir(), "omp")
+ writeTestExecutable(t, fake, []byte("#!/bin/sh\ncat >/dev/null\nprintf '%s\\n' '{\"type\":\"agent_end\"}'\n"))
+ backend, err := ResolveBackend("omp", Config{ExecutablePath: fake, Logger: slog.Default()})
+ if err != nil {
+ t.Fatal(err)
+ }
+ session, err := backend.Execute(t.Context(), "prompt", ExecOptions{Timeout: 5 * time.Second})
+ if err != nil {
+ t.Fatal(err)
+ }
+ for range session.Messages {
+ }
+ result := <-session.Result
+ if result.Status != "failed" || result.SessionID != "" || !strings.Contains(result.Error, "no persisted transcript") {
+ t.Fatalf("must not publish a fabricated resume path: %+v", result)
+ }
+}
diff --git a/server/pkg/agent/omp_test.go b/server/pkg/agent/omp_test.go
index 48327b734e9..b31af38a8e0 100644
--- a/server/pkg/agent/omp_test.go
+++ b/server/pkg/agent/omp_test.go
@@ -4,6 +4,7 @@ import (
"context"
"errors"
"log/slog"
+ "os"
"path/filepath"
"runtime"
"slices"
@@ -110,6 +111,9 @@ func TestOmpExecuteDefaultsToOmpBinary(t *testing.T) {
t.Fatalf("New(omp): %v", err)
}
sessionPath := filepath.Join(t.TempDir(), "session.jsonl")
+ if err := os.WriteFile(sessionPath, []byte("{}\n"), 0o600); err != nil {
+ t.Fatal(err)
+ }
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cancel()
session, err := backend.Execute(ctx, "test prompt", ExecOptions{
@@ -201,7 +205,11 @@ func TestOmpExecuteCompletesFromEventStream(t *testing.T) {
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cancel()
- session, err := backend.Execute(ctx, "prompt-ignored", ExecOptions{Timeout: 5 * time.Second})
+ sessionPath := filepath.Join(t.TempDir(), "session.jsonl")
+ if err := os.WriteFile(sessionPath, []byte("{}\n"), 0o600); err != nil {
+ t.Fatal(err)
+ }
+ session, err := backend.Execute(ctx, "prompt-ignored", ExecOptions{Timeout: 5 * time.Second, ResumeSessionID: sessionPath})
if err != nil {
t.Fatalf("execute: %v", err)
}
diff --git a/server/pkg/agent/pi.go b/server/pkg/agent/pi.go
index 2ec0e0a3197..4c00924d330 100644
--- a/server/pkg/agent/pi.go
+++ b/server/pkg/agent/pi.go
@@ -373,35 +373,58 @@ func (b *piBackend) Execute(ctx context.Context, prompt string, opts ExecOptions
timeout := opts.Timeout
- // Pi's --session flag expects a file path where events are appended.
- // The path doubles as our opaque session identifier: we return it as
- // SessionID and expect it back as ResumeSessionID on the next turn.
+ // The persisted JSONL path is our opaque session identifier. OMP's
+ // --session is strictly resume-only; let it create fresh sessions in a
+ // private directory, then return the transcript it actually persisted.
sessionPath := opts.ResumeSessionID
- if sessionPath == "" {
- p, err := newPiSessionPath()
+ var sessionDir string
+ var sessionLock *os.File
+ if label == "omp" && sessionPath == "" {
+ dir, err := piSessionDir()
if err != nil {
- return nil, fmt.Errorf("%s session path: %w", label, err)
+ return nil, fmt.Errorf("%s session directory: %w", label, err)
}
- sessionPath = p
- }
- if err := ensurePiSessionFile(sessionPath); err != nil {
- return nil, fmt.Errorf("%s session file: %w", label, err)
- }
- sessionLock, locked, err := tryLockPiSessionFile(sessionPath)
- if err != nil {
- return nil, fmt.Errorf("%s session lock: %w", label, err)
- }
- if !locked {
- if opts.ResumeSessionID != "" {
- return piSessionBusyResult(label, sessionPath), nil
+ if err := os.MkdirAll(dir, 0o755); err != nil {
+ return nil, fmt.Errorf("%s session directory: %w", label, err)
+ }
+ sessionDir, err = os.MkdirTemp(dir, "omp-")
+ if err != nil {
+ return nil, fmt.Errorf("%s session directory: %w", label, err)
+ }
+ } else {
+ if sessionPath == "" {
+ sessionPath, err = newPiSessionPath()
+ if err != nil {
+ return nil, fmt.Errorf("%s session path: %w", label, err)
+ }
+ }
+ if label != "omp" {
+ if err := ensurePiSessionFile(sessionPath); err != nil {
+ return nil, fmt.Errorf("%s session file: %w", label, err)
+ }
+ }
+ var locked bool
+ sessionLock, locked, err = tryLockPiSessionFile(sessionPath)
+ if err != nil {
+ return nil, fmt.Errorf("%s session lock: %w", label, err)
+ }
+ if !locked {
+ if opts.ResumeSessionID != "" {
+ return piSessionBusyResult(label, sessionPath), nil
+ }
+ return nil, fmt.Errorf("%s session file %q is already in use", label, sessionPath)
}
- return nil, fmt.Errorf("%s session file %q is already in use", label, sessionPath)
}
runCtx, cancel := runContext(ctx, timeout)
processCtx, cancelProcess := context.WithCancel(runCtx)
- args := buildPiArgs(sessionPath, opts, b.cfg.Logger)
+ var args []string
+ if label == "omp" {
+ args = buildOmpArgs(sessionPath, sessionDir, opts, b.cfg.Logger)
+ } else {
+ args = buildPiArgs(sessionPath, opts, b.cfg.Logger)
+ }
cmd, _, _ := b.cfg.commandAt(execName).execVia(processCtx, choosePiInvocation, lookedUp, args, b.cfg.Logger)
hideAgentWindow(cmd)
b.cfg.logAgentCommand(cmd, newAgentCommandLogArgs(args))
@@ -749,6 +772,15 @@ func (b *piBackend) Execute(ctx context.Context, prompt string, opts ExecOptions
authoritativeTerminal = true
}
+ if sessionDir != "" {
+ var sessionErr error
+ sessionPath, sessionErr = findOmpSessionFile(sessionDir)
+ if sessionErr != nil && finalStatus == "completed" {
+ finalStatus = "failed"
+ finalError = fmt.Sprintf("omp session file: %v", sessionErr)
+ }
+ }
+
b.cfg.Logger.Info(label+" finished", "pid", cmd.Process.Pid, "status", finalStatus, "duration", duration.Round(time.Millisecond).String())
// Publish the authoritative terminal boundary before Result. The daemon
// uses this ordering to let an already-observed provider failure outrank
diff --git a/server/pkg/agent/stream_scanner.go b/server/pkg/agent/stream_scanner.go
index ec94e2d79ea..f8f87a47108 100644
--- a/server/pkg/agent/stream_scanner.go
+++ b/server/pkg/agent/stream_scanner.go
@@ -2,6 +2,7 @@ package agent
import (
"bufio"
+ "errors"
"io"
)
@@ -34,3 +35,58 @@ func newAgentStreamScanner(r io.Reader) *bufio.Scanner {
scanner.Buffer(make([]byte, 0, agentStreamInitialBufferBytes), agentStreamMaxLineBytes)
return scanner
}
+
+// readAgentStreamLine reads the next newline-delimited record from r and
+// returns it without the newline (a single trailing '\r' is dropped, like
+// bufio.Scanner's line splitting). It is the log-walking companion to
+// newAgentStreamScanner for scans that read a session's persisted stream
+// after the fact: where the Scanner ends its scan for good once a line
+// passes agentStreamMaxLineBytes, readAgentStreamLine consumes and discards
+// that record whole and returns bufio.ErrTooLong with r positioned after it,
+// so later records stay readable — an oversized record must not hide the
+// records behind it. A final unterminated line (a live log's partial-write
+// tail) is returned whole, like Scanner does. The returned line is non-nil
+// only when err is nil; an exhausted reader returns io.EOF. Any other read
+// error is returned as-is.
+func readAgentStreamLine(r *bufio.Reader) ([]byte, error) {
+ var line []byte
+ discard := false
+ for {
+ chunk, err := r.ReadSlice('\n')
+ if err == bufio.ErrBufferFull {
+ if !discard {
+ line = append(line, chunk...)
+ if discard = len(line) > agentStreamMaxLineBytes; discard {
+ line = nil // one record must not pin the bound in memory
+ }
+ }
+ continue
+ }
+ if err != nil {
+ if !errors.Is(err, io.EOF) {
+ return nil, err
+ }
+ if discard {
+ return nil, bufio.ErrTooLong
+ }
+ if tail := append(line, chunk...); len(tail) > 0 {
+ if len(tail) > agentStreamMaxLineBytes {
+ return nil, bufio.ErrTooLong
+ }
+ return tail, nil
+ }
+ return nil, io.EOF
+ }
+ if discard {
+ return nil, bufio.ErrTooLong
+ }
+ line = append(line, chunk[:len(chunk)-1]...)
+ if len(line) > agentStreamMaxLineBytes {
+ return nil, bufio.ErrTooLong
+ }
+ if n := len(line); n > 0 && line[n-1] == '\r' {
+ line = line[:n-1]
+ }
+ return line, nil
+ }
+}
diff --git a/server/pkg/agent/stream_scanner_test.go b/server/pkg/agent/stream_scanner_test.go
index e073717b03d..584bada394b 100644
--- a/server/pkg/agent/stream_scanner_test.go
+++ b/server/pkg/agent/stream_scanner_test.go
@@ -3,6 +3,7 @@ package agent
import (
"bufio"
"errors"
+ "io"
"strings"
"testing"
)
@@ -51,3 +52,44 @@ func TestAgentStreamScannerStillFailsClosedAboveCap(t *testing.T) {
t.Fatalf("expected bufio.ErrTooLong, got %v", err)
}
}
+
+// TestReadAgentStreamLineSkipsOversizedAndContinues covers the review on
+// #9057: the log-walking companion to newAgentStreamScanner must do what
+// the Scanner cannot — discard a record beyond agentStreamMaxLineBytes and
+// keep the later records readable.
+func TestReadAgentStreamLineSkipsOversizedAndContinues(t *testing.T) {
+ t.Parallel()
+
+ oversized := strings.Repeat("x", agentStreamMaxLineBytes+16)
+ r := bufio.NewReaderSize(strings.NewReader(oversized+"\nafter\n"), agentStreamInitialBufferBytes)
+
+ line, err := readAgentStreamLine(r)
+ if line != nil || !errors.Is(err, bufio.ErrTooLong) {
+ t.Fatalf("expected ErrTooLong with no line, got %d bytes err=%v", len(line), err)
+ }
+ line, err = readAgentStreamLine(r)
+ if err != nil || string(line) != "after" {
+ t.Fatalf("expected the record after the oversized one, got %q err=%v", line, err)
+ }
+ if _, err := readAgentStreamLine(r); !errors.Is(err, io.EOF) {
+ t.Fatalf("expected io.EOF after the last record, got %v", err)
+ }
+}
+
+// TestReadAgentStreamLineReturnsUnterminatedTail: a live log can end
+// mid-write; the final unterminated line is still returned, matching
+// bufio.Scanner's behavior, and an exhausted reader reports io.EOF.
+func TestReadAgentStreamLineReturnsUnterminatedTail(t *testing.T) {
+ t.Parallel()
+
+ r := bufio.NewReaderSize(strings.NewReader("first\npartial-tail"), agentStreamInitialBufferBytes)
+ if line, err := readAgentStreamLine(r); err != nil || string(line) != "first" {
+ t.Fatalf("expected the first line, got %q err=%v", line, err)
+ }
+ if line, err := readAgentStreamLine(r); err != nil || string(line) != "partial-tail" {
+ t.Fatalf("expected the unterminated tail, got %q err=%v", line, err)
+ }
+ if _, err := readAgentStreamLine(r); !errors.Is(err, io.EOF) {
+ t.Fatalf("expected io.EOF, got %v", err)
+ }
+}
diff --git a/server/pkg/agent/zeroclaw_test.go b/server/pkg/agent/zeroclaw_test.go
index b8d7d980772..16db5852ce4 100644
--- a/server/pkg/agent/zeroclaw_test.go
+++ b/server/pkg/agent/zeroclaw_test.go
@@ -73,9 +73,7 @@ func writeFakeZeroclawScript(t *testing.T, script string) string {
t.Helper()
dir := t.TempDir()
bin := filepath.Join(dir, "zeroclaw")
- if err := os.WriteFile(bin, []byte(script), 0755); err != nil {
- t.Fatalf("write fake zeroclaw: %v", err)
- }
+ writeTestExecutable(t, bin, []byte(script))
return bin
}
diff --git a/server/pkg/db/generated/channel.sql.go b/server/pkg/db/generated/channel.sql.go
index df7204eb655..00f9b4c356d 100644
--- a/server/pkg/db/generated/channel.sql.go
+++ b/server/pkg/db/generated/channel.sql.go
@@ -1987,6 +1987,59 @@ func (q *Queries) GetChannelUserBindingByUserID(ctx context.Context, arg GetChan
return i, err
}
+const isChannelMessageTypingActive = `-- name: IsChannelMessageTypingActive :one
+SELECT EXISTS (
+ SELECT 1 FROM chat_message AS message
+ JOIN chat_session AS session ON session.id = message.chat_session_id
+ JOIN channel_chat_session_binding AS binding ON binding.chat_session_id = session.id
+ JOIN channel_installation AS installation ON installation.id = binding.installation_id
+ LEFT JOIN agent_task_queue AS task ON task.id = message.task_id
+ LEFT JOIN agent AS task_agent ON task_agent.id = task.agent_id
+ WHERE message.id = $1
+ AND session.id = $2
+ AND session.workspace_id = $3
+ AND installation.workspace_id = $3
+ AND installation.id = $4
+ AND installation.channel_type = $5
+ AND binding.channel_type = $5
+ AND installation.status = 'active'
+ AND session.status = 'active'
+ AND binding.retired_at IS NULL
+ AND message.role = 'user'
+ AND message.channel_ingested
+ AND NOT message.channel_typing_settled
+ AND (message.task_id IS NULL OR (
+ task_agent.workspace_id = $3
+ AND task.chat_session_id = session.id
+ AND task.status IN ('queued', 'dispatched', 'running', 'waiting_local_directory', 'deferred')
+ ))
+)
+`
+
+type IsChannelMessageTypingActiveParams struct {
+ MessageID pgtype.UUID `json:"message_id"`
+ ChatSessionID pgtype.UUID `json:"chat_session_id"`
+ WorkspaceID pgtype.UUID `json:"workspace_id"`
+ InstallationID pgtype.UUID `json:"installation_id"`
+ ChannelType string `json:"channel_type"`
+}
+
+// Revalidate the exact input after a remote reaction Add. A pending debounce
+// input has no task yet; a sealed input follows its immutable owning task.
+// Do not use the binding's latest message or another turn's active task.
+func (q *Queries) IsChannelMessageTypingActive(ctx context.Context, arg IsChannelMessageTypingActiveParams) (bool, error) {
+ row := q.db.QueryRow(ctx, isChannelMessageTypingActive,
+ arg.MessageID,
+ arg.ChatSessionID,
+ arg.WorkspaceID,
+ arg.InstallationID,
+ arg.ChannelType,
+ )
+ var exists bool
+ err := row.Scan(&exists)
+ return exists, err
+}
+
const listActiveChannelInstallations = `-- name: ListActiveChannelInstallations :many
SELECT ci.id, ci.workspace_id, ci.agent_id, ci.channel_type, ci.config, ci.status, ci.ws_lease_token, ci.ws_lease_expires_at, ci.installer_user_id, ci.installed_at, ci.created_at, ci.updated_at FROM channel_installation ci
JOIN workspace w ON w.id = ci.workspace_id
diff --git a/server/pkg/db/generated/channel_typing.sql.go b/server/pkg/db/generated/channel_typing.sql.go
new file mode 100644
index 00000000000..95f3d0b99d3
--- /dev/null
+++ b/server/pkg/db/generated/channel_typing.sql.go
@@ -0,0 +1,338 @@
+// Code generated by sqlc. DO NOT EDIT.
+// versions:
+// sqlc v1.31.1
+// source: channel_typing.sql
+
+package db
+
+import (
+ "context"
+
+ "github.com/jackc/pgx/v5/pgtype"
+)
+
+const acknowledgeChannelTypingReactionCleanup = `-- name: AcknowledgeChannelTypingReactionCleanup :exec
+UPDATE channel_typing_reaction
+SET cleaned_at = CASE WHEN add_finished AND reaction_id <> '' THEN now() ELSE NULL END,
+ retry_after = GREATEST(retry_after, now() + interval '30 seconds'),
+ installation_snapshot = CASE WHEN add_finished AND reaction_id <> '' THEN '{}'::jsonb ELSE installation_snapshot END
+WHERE id = $1 AND reaction_id = $2 AND cleanup_required AND abandoned_at IS NULL
+`
+
+type AcknowledgeChannelTypingReactionCleanupParams struct {
+ ID pgtype.UUID `json:"id"`
+ ReactionID string `json:"reaction_id"`
+}
+
+// An unfinished Add may still create a reaction after an empty sweep. Retain
+// its anchor; only a known successful Add response can close cleanup forever.
+func (q *Queries) AcknowledgeChannelTypingReactionCleanup(ctx context.Context, arg AcknowledgeChannelTypingReactionCleanupParams) error {
+ _, err := q.db.Exec(ctx, acknowledgeChannelTypingReactionCleanup, arg.ID, arg.ReactionID)
+ return err
+}
+
+const claimChannelTypingReactionCleanup = `-- name: ClaimChannelTypingReactionCleanup :many
+WITH candidates AS (
+ SELECT r.id FROM channel_typing_reaction r
+ WHERE r.cleaned_at IS NULL AND r.abandoned_at IS NULL AND r.retry_after <= now()
+ AND r.created_at > now() - interval '7 days'
+ AND ($1::uuid IS NULL OR r.chat_session_id = $1)
+ AND (r.cleanup_required OR (NOT r.add_finished AND r.created_at < now() - interval '1 minute') OR NOT EXISTS (
+ SELECT 1 FROM chat_message m
+ JOIN chat_session s ON s.id = m.chat_session_id
+ JOIN channel_installation i ON i.id = r.installation_id
+ JOIN channel_chat_session_binding b ON b.chat_session_id = s.id AND b.installation_id = i.id
+ LEFT JOIN agent_task_queue t ON t.id = m.task_id
+ LEFT JOIN agent a ON a.id = t.agent_id
+ WHERE m.id = r.chat_message_id AND s.id = r.chat_session_id
+ AND s.workspace_id = r.workspace_id AND i.workspace_id = r.workspace_id
+ AND i.channel_type = 'feishu' AND b.channel_type = 'feishu'
+ AND s.status = 'active' AND i.status = 'active' AND b.retired_at IS NULL
+ AND m.channel_ingested AND m.role = 'user' AND NOT m.channel_typing_settled
+ -- A taskless badge has a bounded visual lifetime even if a crashed
+ -- debouncer never flushes. Do not settle the input or cancel a future run.
+ AND ((m.task_id IS NULL AND r.created_at > now() - interval '2 minutes') OR (t.chat_session_id = s.id AND a.workspace_id = r.workspace_id
+ AND t.status IN ('queued','dispatched','running','waiting_local_directory','deferred')))
+ ))
+ ORDER BY r.retry_after, r.id LIMIT 100
+ FOR UPDATE OF r SKIP LOCKED
+)
+UPDATE channel_typing_reaction r SET cleanup_required = true,
+ attempts = attempts + 1,
+ retry_after = now() + LEAST(3600, 30 * power(2, LEAST(r.attempts, 7))) * interval '1 second'
+FROM candidates c WHERE r.id = c.id RETURNING r.id, r.workspace_id, r.chat_session_id, r.chat_message_id, r.installation_id, r.channel_message_id, r.installation_snapshot, r.reaction_id, r.add_finished, r.cleanup_required, r.cleaned_at, r.retry_after, r.attempts, r.created_at, r.abandoned_at, r.quota_slot
+`
+
+// The DB is the shared source of eligibility. Missing source rows after a
+// committed deletion are terminal too. No transaction takes remote I/O locks.
+func (q *Queries) ClaimChannelTypingReactionCleanup(ctx context.Context, chatSessionID pgtype.UUID) ([]ChannelTypingReaction, error) {
+ rows, err := q.db.Query(ctx, claimChannelTypingReactionCleanup, chatSessionID)
+ if err != nil {
+ return nil, err
+ }
+ defer rows.Close()
+ items := []ChannelTypingReaction{}
+ for rows.Next() {
+ var i ChannelTypingReaction
+ if err := rows.Scan(
+ &i.ID,
+ &i.WorkspaceID,
+ &i.ChatSessionID,
+ &i.ChatMessageID,
+ &i.InstallationID,
+ &i.ChannelMessageID,
+ &i.InstallationSnapshot,
+ &i.ReactionID,
+ &i.AddFinished,
+ &i.CleanupRequired,
+ &i.CleanedAt,
+ &i.RetryAfter,
+ &i.Attempts,
+ &i.CreatedAt,
+ &i.AbandonedAt,
+ &i.QuotaSlot,
+ ); err != nil {
+ return nil, err
+ }
+ items = append(items, i)
+ }
+ if err := rows.Err(); err != nil {
+ return nil, err
+ }
+ return items, nil
+}
+
+const expireChannelTypingReactionCleanup = `-- name: ExpireChannelTypingReactionCleanup :many
+WITH candidates AS (
+ SELECT id FROM channel_typing_reaction
+ WHERE cleaned_at IS NULL AND abandoned_at IS NULL AND created_at <= now() - interval '7 days'
+ ORDER BY created_at, id LIMIT 1000 FOR UPDATE SKIP LOCKED
+)
+UPDATE channel_typing_reaction r SET abandoned_at = now(), cleanup_required = true,
+ installation_snapshot = '{}'::jsonb
+FROM candidates c WHERE r.id = c.id
+RETURNING r.id, r.workspace_id
+`
+
+type ExpireChannelTypingReactionCleanupRow struct {
+ ID pgtype.UUID `json:"id"`
+ WorkspaceID pgtype.UUID `json:"workspace_id"`
+}
+
+// Independent maintenance, including active/uncertain Adds: the cosmetic badge
+// has a seven-day maximum lifetime. Remote failure is NOT reported as success.
+func (q *Queries) ExpireChannelTypingReactionCleanup(ctx context.Context) ([]ExpireChannelTypingReactionCleanupRow, error) {
+ rows, err := q.db.Query(ctx, expireChannelTypingReactionCleanup)
+ if err != nil {
+ return nil, err
+ }
+ defer rows.Close()
+ items := []ExpireChannelTypingReactionCleanupRow{}
+ for rows.Next() {
+ var i ExpireChannelTypingReactionCleanupRow
+ if err := rows.Scan(&i.ID, &i.WorkspaceID); err != nil {
+ return nil, err
+ }
+ items = append(items, i)
+ }
+ if err := rows.Err(); err != nil {
+ return nil, err
+ }
+ return items, nil
+}
+
+const finishChannelTypingReactionAdd = `-- name: FinishChannelTypingReactionAdd :one
+UPDATE channel_typing_reaction SET reaction_id = $1, add_finished = true,
+ cleanup_required = cleanup_required OR $2::boolean,
+ cleaned_at = NULL, retry_after = now()
+WHERE id = $3 AND abandoned_at IS NULL AND created_at > now() - interval '7 days'
+RETURNING id, workspace_id, chat_session_id, chat_message_id, installation_id, channel_message_id, installation_snapshot, reaction_id, add_finished, cleanup_required, cleaned_at, retry_after, attempts, created_at, abandoned_at, quota_slot
+`
+
+type FinishChannelTypingReactionAddParams struct {
+ ReactionID string `json:"reaction_id"`
+ CleanupRequired bool `json:"cleanup_required"`
+ ID pgtype.UUID `json:"id"`
+}
+
+// cleanup_required is monotonic. A stale active read cannot undo another
+// replica's terminal decision; late HTTP completion re-arms compensation.
+func (q *Queries) FinishChannelTypingReactionAdd(ctx context.Context, arg FinishChannelTypingReactionAddParams) (ChannelTypingReaction, error) {
+ row := q.db.QueryRow(ctx, finishChannelTypingReactionAdd, arg.ReactionID, arg.CleanupRequired, arg.ID)
+ var i ChannelTypingReaction
+ err := row.Scan(
+ &i.ID,
+ &i.WorkspaceID,
+ &i.ChatSessionID,
+ &i.ChatMessageID,
+ &i.InstallationID,
+ &i.ChannelMessageID,
+ &i.InstallationSnapshot,
+ &i.ReactionID,
+ &i.AddFinished,
+ &i.CleanupRequired,
+ &i.CleanedAt,
+ &i.RetryAfter,
+ &i.Attempts,
+ &i.CreatedAt,
+ &i.AbandonedAt,
+ &i.QuotaSlot,
+ )
+ return i, err
+}
+
+const listRequestedChannelTypingReactions = `-- name: ListRequestedChannelTypingReactions :many
+SELECT id FROM channel_typing_reaction WHERE id = ANY($1::uuid[]) AND cleanup_required
+`
+
+func (q *Queries) ListRequestedChannelTypingReactions(ctx context.Context, ids []pgtype.UUID) ([]pgtype.UUID, error) {
+ rows, err := q.db.Query(ctx, listRequestedChannelTypingReactions, ids)
+ if err != nil {
+ return nil, err
+ }
+ defer rows.Close()
+ items := []pgtype.UUID{}
+ for rows.Next() {
+ var id pgtype.UUID
+ if err := rows.Scan(&id); err != nil {
+ return nil, err
+ }
+ items = append(items, id)
+ }
+ if err := rows.Err(); err != nil {
+ return nil, err
+ }
+ return items, nil
+}
+
+const pruneChannelTypingReactionCleanup = `-- name: PruneChannelTypingReactionCleanup :exec
+WITH candidates AS (
+ SELECT id FROM channel_typing_reaction
+ WHERE cleaned_at < now() - interval '7 days' OR abandoned_at < now() - interval '7 days'
+ LIMIT 1000 FOR UPDATE SKIP LOCKED
+)
+DELETE FROM channel_typing_reaction r USING candidates c WHERE r.id = c.id
+`
+
+// Each terminal outcome remains inspectable for seven days without credentials.
+func (q *Queries) PruneChannelTypingReactionCleanup(ctx context.Context) error {
+ _, err := q.db.Exec(ctx, pruneChannelTypingReactionCleanup)
+ return err
+}
+
+const registerChannelTypingReaction = `-- name: RegisterChannelTypingReaction :one
+INSERT INTO channel_typing_reaction (
+ id, workspace_id, chat_session_id, chat_message_id, installation_id,
+ channel_message_id, installation_snapshot, cleanup_required, quota_slot
+)
+SELECT $1, $2, $3, $4, $5,
+ $6, $7, message.channel_typing_settled, slot.value
+FROM chat_message AS message
+JOIN chat_session AS session ON session.id = message.chat_session_id
+JOIN channel_installation AS installation ON installation.id = $5
+CROSS JOIN LATERAL (
+ SELECT value FROM generate_series(1, 1000) AS value
+ WHERE NOT EXISTS (SELECT 1 FROM channel_typing_reaction r
+ WHERE r.workspace_id = $2 AND r.quota_slot = value
+ AND r.cleaned_at IS NULL AND r.abandoned_at IS NULL)
+ ORDER BY value LIMIT 1
+) slot
+WHERE message.id = $4
+ AND session.id = $3 AND session.workspace_id = $2
+ AND installation.workspace_id = $2 AND installation.channel_type = 'feishu'
+ AND message.channel_ingested AND message.role = 'user'
+ON CONFLICT DO NOTHING RETURNING id, workspace_id, chat_session_id, chat_message_id, installation_id, channel_message_id, installation_snapshot, reaction_id, add_finished, cleanup_required, cleaned_at, retry_after, attempts, created_at, abandoned_at, quota_slot
+`
+
+type RegisterChannelTypingReactionParams struct {
+ ID pgtype.UUID `json:"id"`
+ WorkspaceID pgtype.UUID `json:"workspace_id"`
+ ChatSessionID pgtype.UUID `json:"chat_session_id"`
+ ChatMessageID pgtype.UUID `json:"chat_message_id"`
+ InstallationID pgtype.UUID `json:"installation_id"`
+ ChannelMessageID string `json:"channel_message_id"`
+ InstallationSnapshot []byte `json:"installation_snapshot"`
+}
+
+// Commit the complete external anchor BEFORE issuing Add. A terminal observer
+// can enumerate every input, independently of the one delivery trigger.
+// A concurrent registration can take the same slot. Fail closed: a cosmetic
+// badge may be skipped, but an untracked Add or a quota overrun is never safe.
+func (q *Queries) RegisterChannelTypingReaction(ctx context.Context, arg RegisterChannelTypingReactionParams) (ChannelTypingReaction, error) {
+ row := q.db.QueryRow(ctx, registerChannelTypingReaction,
+ arg.ID,
+ arg.WorkspaceID,
+ arg.ChatSessionID,
+ arg.ChatMessageID,
+ arg.InstallationID,
+ arg.ChannelMessageID,
+ arg.InstallationSnapshot,
+ )
+ var i ChannelTypingReaction
+ err := row.Scan(
+ &i.ID,
+ &i.WorkspaceID,
+ &i.ChatSessionID,
+ &i.ChatMessageID,
+ &i.InstallationID,
+ &i.ChannelMessageID,
+ &i.InstallationSnapshot,
+ &i.ReactionID,
+ &i.AddFinished,
+ &i.CleanupRequired,
+ &i.CleanedAt,
+ &i.RetryAfter,
+ &i.Attempts,
+ &i.CreatedAt,
+ &i.AbandonedAt,
+ &i.QuotaSlot,
+ )
+ return i, err
+}
+
+const settleChannelTypingInputs = `-- name: SettleChannelTypingInputs :exec
+UPDATE chat_message AS message SET channel_typing_settled = true
+FROM chat_message AS cutoff, chat_session AS session, channel_installation AS installation
+WHERE cutoff.id = $1 AND cutoff.chat_session_id = $2
+ AND session.id = $2 AND session.workspace_id = $3
+ AND installation.id = $4 AND installation.workspace_id = $3
+ AND installation.channel_type = 'feishu'
+ AND EXISTS (SELECT 1 FROM channel_chat_session_binding b WHERE b.chat_session_id = session.id
+ AND b.installation_id = installation.id AND b.channel_type = 'feishu')
+ AND message.chat_session_id = session.id AND message.task_id IS NULL
+ AND message.channel_ingested AND message.role = 'user'
+ AND COALESCE(message.channel_context_revision, 1) = $5
+ AND (message.created_at, message.id) <= (cutoff.created_at, cutoff.id)
+`
+
+type SettleChannelTypingInputsParams struct {
+ ThroughMessageID pgtype.UUID `json:"through_message_id"`
+ ChatSessionID pgtype.UUID `json:"chat_session_id"`
+ WorkspaceID pgtype.UUID `json:"workspace_id"`
+ InstallationID pgtype.UUID `json:"installation_id"`
+ ContextRevision pgtype.Int8 `json:"context_revision"`
+}
+
+// The flush owns only this context's unowned inputs up to its captured cutoff.
+// Later inputs, even in the same session, are not settled by this failed run.
+func (q *Queries) SettleChannelTypingInputs(ctx context.Context, arg SettleChannelTypingInputsParams) error {
+ _, err := q.db.Exec(ctx, settleChannelTypingInputs,
+ arg.ThroughMessageID,
+ arg.ChatSessionID,
+ arg.WorkspaceID,
+ arg.InstallationID,
+ arg.ContextRevision,
+ )
+ return err
+}
+
+const skipChannelTypingReactionAdd = `-- name: SkipChannelTypingReactionAdd :exec
+UPDATE channel_typing_reaction SET add_finished = true, cleanup_required = true,
+ cleaned_at = now(), installation_snapshot = '{}'::jsonb WHERE id = $1 AND abandoned_at IS NULL
+`
+
+// No HTTP Add was issued for an already-settled input.
+func (q *Queries) SkipChannelTypingReactionAdd(ctx context.Context, id pgtype.UUID) error {
+ _, err := q.db.Exec(ctx, skipChannelTypingReactionAdd, id)
+ return err
+}
diff --git a/server/pkg/db/generated/chat.sql.go b/server/pkg/db/generated/chat.sql.go
index 94ccc385a73..05f8710e1d8 100644
--- a/server/pkg/db/generated/chat.sql.go
+++ b/server/pkg/db/generated/chat.sql.go
@@ -262,7 +262,7 @@ VALUES (
$11,
COALESCE($12::uuid, gen_random_uuid())
)
-RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids
+RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled
`
type CreateChatMessageParams struct {
@@ -317,6 +317,7 @@ func (q *Queries) CreateChatMessage(ctx context.Context, arg CreateChatMessagePa
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
)
return i, err
}
@@ -516,7 +517,7 @@ VALUES (
$3::timestamptz + interval '1 microsecond',
COALESCE($4::uuid, gen_random_uuid())
)
-RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids
+RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled
`
type CreateMikaOnboardingOpeningParams struct {
@@ -564,6 +565,7 @@ func (q *Queries) CreateMikaOnboardingOpening(ctx context.Context, arg CreateMik
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
)
return i, err
}
@@ -737,7 +739,7 @@ DELETE FROM chat_message
WHERE task_id = $1
AND role = 'user'
AND message_kind <> 'onboarding_kickoff'
-RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids
+RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled
`
// Deletes the MEMBER-TYPED input of a cancelled/edited turn.
@@ -772,6 +774,7 @@ func (q *Queries) DeleteUserChatMessageByTask(ctx context.Context, taskID pgtype
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
)
return i, err
}
@@ -814,7 +817,7 @@ func (q *Queries) GetChannelMediaPendingUntil(ctx context.Context, arg GetChanne
}
const getChatMessage = `-- name: GetChatMessage :one
-SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids FROM chat_message
+SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled FROM chat_message
WHERE id = $1
`
@@ -839,12 +842,13 @@ func (q *Queries) GetChatMessage(ctx context.Context, id pgtype.UUID) (ChatMessa
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
)
return i, err
}
const getChatMessageByTaskAssistant = `-- name: GetChatMessageByTaskAssistant :one
-SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids FROM chat_message
+SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled FROM chat_message
WHERE task_id = $1 AND role = 'assistant'
ORDER BY created_at DESC
LIMIT 1
@@ -873,6 +877,7 @@ func (q *Queries) GetChatMessageByTaskAssistant(ctx context.Context, taskID pgty
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
)
return i, err
}
@@ -1077,7 +1082,7 @@ func (q *Queries) GetLastChatTaskSession(ctx context.Context, arg GetLastChatTas
}
const getLatestAssistantChatMessageForSession = `-- name: GetLatestAssistantChatMessageForSession :one
-SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids FROM chat_message
+SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled FROM chat_message
WHERE chat_session_id = $1 AND role = 'assistant' AND task_id IS NOT NULL
ORDER BY created_at DESC
LIMIT 1
@@ -1108,6 +1113,7 @@ func (q *Queries) GetLatestAssistantChatMessageForSession(ctx context.Context, c
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
)
return i, err
}
@@ -1895,7 +1901,7 @@ func (q *Queries) ListChatDraftRestoresBySession(ctx context.Context, chatSessio
}
const listChatInputMessages = `-- name: ListChatInputMessages :many
-SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids FROM chat_message
+SELECT id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled FROM chat_message
WHERE task_id = $1 AND role = 'user'
ORDER BY created_at ASC, id ASC
`
@@ -1934,6 +1940,7 @@ func (q *Queries) ListChatInputMessages(ctx context.Context, taskID pgtype.UUID)
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
); err != nil {
return nil, err
}
@@ -1946,7 +1953,7 @@ func (q *Queries) ListChatInputMessages(ctx context.Context, taskID pgtype.UUID)
}
const listChatMessages = `-- name: ListChatMessages :many
-SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids FROM chat_message AS message
+SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids, message.channel_typing_settled FROM chat_message AS message
WHERE message.chat_session_id = $1
AND message.message_kind != 'channel_command'
AND NOT (
@@ -2018,6 +2025,7 @@ func (q *Queries) ListChatMessages(ctx context.Context, chatSessionID pgtype.UUI
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
); err != nil {
return nil, err
}
@@ -2030,7 +2038,7 @@ func (q *Queries) ListChatMessages(ctx context.Context, chatSessionID pgtype.UUI
}
const listChatMessagesForLegacyTask = `-- name: ListChatMessagesForLegacyTask :many
-SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids FROM chat_message AS message
+SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids, message.channel_typing_settled FROM chat_message AS message
WHERE message.chat_session_id = $1
AND NOT (
message.role = 'user'
@@ -2091,6 +2099,7 @@ func (q *Queries) ListChatMessagesForLegacyTask(ctx context.Context, chatSession
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
); err != nil {
return nil, err
}
@@ -2103,7 +2112,7 @@ func (q *Queries) ListChatMessagesForLegacyTask(ctx context.Context, chatSession
}
const listChatMessagesPage = `-- name: ListChatMessagesPage :many
-SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids FROM chat_message AS message
+SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids, message.channel_typing_settled FROM chat_message AS message
WHERE message.chat_session_id = $1
AND message.message_kind != 'channel_command'
AND NOT (
@@ -2180,6 +2189,7 @@ func (q *Queries) ListChatMessagesPage(ctx context.Context, arg ListChatMessages
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
); err != nil {
return nil, err
}
@@ -2192,7 +2202,7 @@ func (q *Queries) ListChatMessagesPage(ctx context.Context, arg ListChatMessages
}
const listChatMessagesPageForChannelContext = `-- name: ListChatMessagesPageForChannelContext :many
-SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids
+SELECT message.id, message.chat_session_id, message.role, message.content, message.task_id, message.created_at, message.failure_reason, message.elapsed_ms, message.message_kind, message.channel_media_pending_until, message.channel_ingested, message.quick_actions, message.channel_context_revision, message.channel_outbound_type, message.channel_outbound_installation_id, message.channel_outbound_chat_id, message.channel_outbound_message_ids, message.channel_typing_settled
FROM chat_message AS message
LEFT JOIN agent_task_queue AS owner ON owner.id = message.task_id
WHERE message.chat_session_id = $1
@@ -2267,6 +2277,7 @@ func (q *Queries) ListChatMessagesPageForChannelContext(ctx context.Context, arg
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
); err != nil {
return nil, err
}
@@ -3363,7 +3374,7 @@ WHERE id = (
ORDER BY inner_msg.created_at DESC
LIMIT 1
)
-RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids
+RETURNING id, chat_session_id, role, content, task_id, created_at, failure_reason, elapsed_ms, message_kind, channel_media_pending_until, channel_ingested, quick_actions, channel_context_revision, channel_outbound_type, channel_outbound_installation_id, channel_outbound_chat_id, channel_outbound_message_ids, channel_typing_settled
`
type SetChatMessageQuickActionsByTaskParams struct {
@@ -3392,6 +3403,7 @@ func (q *Queries) SetChatMessageQuickActionsByTask(ctx context.Context, arg SetC
&i.ChannelOutboundInstallationID,
&i.ChannelOutboundChatID,
&i.ChannelOutboundMessageIds,
+ &i.ChannelTypingSettled,
)
return i, err
}
diff --git a/server/pkg/db/generated/models.go b/server/pkg/db/generated/models.go
index 2cd4893996e..87a89e51a58 100644
--- a/server/pkg/db/generated/models.go
+++ b/server/pkg/db/generated/models.go
@@ -475,6 +475,25 @@ type ChannelTaskDelivery struct {
ChannelSenderID pgtype.Text `json:"channel_sender_id"`
}
+type ChannelTypingReaction struct {
+ ID pgtype.UUID `json:"id"`
+ WorkspaceID pgtype.UUID `json:"workspace_id"`
+ ChatSessionID pgtype.UUID `json:"chat_session_id"`
+ ChatMessageID pgtype.UUID `json:"chat_message_id"`
+ InstallationID pgtype.UUID `json:"installation_id"`
+ ChannelMessageID string `json:"channel_message_id"`
+ InstallationSnapshot []byte `json:"installation_snapshot"`
+ ReactionID string `json:"reaction_id"`
+ AddFinished bool `json:"add_finished"`
+ CleanupRequired bool `json:"cleanup_required"`
+ CleanedAt pgtype.Timestamptz `json:"cleaned_at"`
+ RetryAfter pgtype.Timestamptz `json:"retry_after"`
+ Attempts int32 `json:"attempts"`
+ CreatedAt pgtype.Timestamptz `json:"created_at"`
+ AbandonedAt pgtype.Timestamptz `json:"abandoned_at"`
+ QuotaSlot pgtype.Int4 `json:"quota_slot"`
+}
+
type ChannelUserBinding struct {
ID pgtype.UUID `json:"id"`
WorkspaceID pgtype.UUID `json:"workspace_id"`
@@ -513,6 +532,7 @@ type ChatMessage struct {
ChannelOutboundInstallationID pgtype.UUID `json:"channel_outbound_installation_id"`
ChannelOutboundChatID pgtype.Text `json:"channel_outbound_chat_id"`
ChannelOutboundMessageIds []string `json:"channel_outbound_message_ids"`
+ ChannelTypingSettled bool `json:"channel_typing_settled"`
}
type ChatPinnedAgent struct {
diff --git a/server/pkg/db/queries/channel.sql b/server/pkg/db/queries/channel.sql
index 775844130a3..7457f3f5d6b 100644
--- a/server/pkg/db/queries/channel.sql
+++ b/server/pkg/db/queries/channel.sql
@@ -1518,3 +1518,34 @@ WHERE channel_reply_delivery.phase <> 'settled'
-- would be dropped as "already settled".
AND EXCLUDED.attempt_depth >= channel_reply_delivery.attempt_depth
RETURNING *;
+
+-- name: IsChannelMessageTypingActive :one
+-- Revalidate the exact input after a remote reaction Add. A pending debounce
+-- input has no task yet; a sealed input follows its immutable owning task.
+-- Do not use the binding's latest message or another turn's active task.
+SELECT EXISTS (
+ SELECT 1 FROM chat_message AS message
+ JOIN chat_session AS session ON session.id = message.chat_session_id
+ JOIN channel_chat_session_binding AS binding ON binding.chat_session_id = session.id
+ JOIN channel_installation AS installation ON installation.id = binding.installation_id
+ LEFT JOIN agent_task_queue AS task ON task.id = message.task_id
+ LEFT JOIN agent AS task_agent ON task_agent.id = task.agent_id
+ WHERE message.id = @message_id
+ AND session.id = @chat_session_id
+ AND session.workspace_id = @workspace_id
+ AND installation.workspace_id = @workspace_id
+ AND installation.id = @installation_id
+ AND installation.channel_type = @channel_type
+ AND binding.channel_type = @channel_type
+ AND installation.status = 'active'
+ AND session.status = 'active'
+ AND binding.retired_at IS NULL
+ AND message.role = 'user'
+ AND message.channel_ingested
+ AND NOT message.channel_typing_settled
+ AND (message.task_id IS NULL OR (
+ task_agent.workspace_id = @workspace_id
+ AND task.chat_session_id = session.id
+ AND task.status IN ('queued', 'dispatched', 'running', 'waiting_local_directory', 'deferred')
+ ))
+);
diff --git a/server/pkg/db/queries/channel_typing.sql b/server/pkg/db/queries/channel_typing.sql
new file mode 100644
index 00000000000..6504cb6ca78
--- /dev/null
+++ b/server/pkg/db/queries/channel_typing.sql
@@ -0,0 +1,123 @@
+-- name: RegisterChannelTypingReaction :one
+-- Commit the complete external anchor BEFORE issuing Add. A terminal observer
+-- can enumerate every input, independently of the one delivery trigger.
+INSERT INTO channel_typing_reaction (
+ id, workspace_id, chat_session_id, chat_message_id, installation_id,
+ channel_message_id, installation_snapshot, cleanup_required, quota_slot
+)
+SELECT @id, @workspace_id, @chat_session_id, @chat_message_id, @installation_id,
+ @channel_message_id, @installation_snapshot, message.channel_typing_settled, slot.value
+FROM chat_message AS message
+JOIN chat_session AS session ON session.id = message.chat_session_id
+JOIN channel_installation AS installation ON installation.id = @installation_id
+CROSS JOIN LATERAL (
+ SELECT value FROM generate_series(1, 1000) AS value
+ WHERE NOT EXISTS (SELECT 1 FROM channel_typing_reaction r
+ WHERE r.workspace_id = @workspace_id AND r.quota_slot = value
+ AND r.cleaned_at IS NULL AND r.abandoned_at IS NULL)
+ ORDER BY value LIMIT 1
+) slot
+WHERE message.id = @chat_message_id
+ AND session.id = @chat_session_id AND session.workspace_id = @workspace_id
+ AND installation.workspace_id = @workspace_id AND installation.channel_type = 'feishu'
+ AND message.channel_ingested AND message.role = 'user'
+-- A concurrent registration can take the same slot. Fail closed: a cosmetic
+-- badge may be skipped, but an untracked Add or a quota overrun is never safe.
+ON CONFLICT DO NOTHING RETURNING *;
+
+-- name: FinishChannelTypingReactionAdd :one
+-- cleanup_required is monotonic. A stale active read cannot undo another
+-- replica's terminal decision; late HTTP completion re-arms compensation.
+UPDATE channel_typing_reaction SET reaction_id = @reaction_id, add_finished = true,
+ cleanup_required = cleanup_required OR @cleanup_required::boolean,
+ cleaned_at = NULL, retry_after = now()
+WHERE id = @id AND abandoned_at IS NULL AND created_at > now() - interval '7 days'
+RETURNING *;
+
+-- name: AcknowledgeChannelTypingReactionCleanup :exec
+-- An unfinished Add may still create a reaction after an empty sweep. Retain
+-- its anchor; only a known successful Add response can close cleanup forever.
+UPDATE channel_typing_reaction
+SET cleaned_at = CASE WHEN add_finished AND reaction_id <> '' THEN now() ELSE NULL END,
+ retry_after = GREATEST(retry_after, now() + interval '30 seconds'),
+ installation_snapshot = CASE WHEN add_finished AND reaction_id <> '' THEN '{}'::jsonb ELSE installation_snapshot END
+WHERE id = @id AND reaction_id = @reaction_id AND cleanup_required AND abandoned_at IS NULL;
+
+-- name: ClaimChannelTypingReactionCleanup :many
+-- The DB is the shared source of eligibility. Missing source rows after a
+-- committed deletion are terminal too. No transaction takes remote I/O locks.
+WITH candidates AS (
+ SELECT r.id FROM channel_typing_reaction r
+ WHERE r.cleaned_at IS NULL AND r.abandoned_at IS NULL AND r.retry_after <= now()
+ AND r.created_at > now() - interval '7 days'
+ AND (sqlc.narg('chat_session_id')::uuid IS NULL OR r.chat_session_id = sqlc.narg('chat_session_id'))
+ AND (r.cleanup_required OR (NOT r.add_finished AND r.created_at < now() - interval '1 minute') OR NOT EXISTS (
+ SELECT 1 FROM chat_message m
+ JOIN chat_session s ON s.id = m.chat_session_id
+ JOIN channel_installation i ON i.id = r.installation_id
+ JOIN channel_chat_session_binding b ON b.chat_session_id = s.id AND b.installation_id = i.id
+ LEFT JOIN agent_task_queue t ON t.id = m.task_id
+ LEFT JOIN agent a ON a.id = t.agent_id
+ WHERE m.id = r.chat_message_id AND s.id = r.chat_session_id
+ AND s.workspace_id = r.workspace_id AND i.workspace_id = r.workspace_id
+ AND i.channel_type = 'feishu' AND b.channel_type = 'feishu'
+ AND s.status = 'active' AND i.status = 'active' AND b.retired_at IS NULL
+ AND m.channel_ingested AND m.role = 'user' AND NOT m.channel_typing_settled
+ -- A taskless badge has a bounded visual lifetime even if a crashed
+ -- debouncer never flushes. Do not settle the input or cancel a future run.
+ AND ((m.task_id IS NULL AND r.created_at > now() - interval '2 minutes') OR (t.chat_session_id = s.id AND a.workspace_id = r.workspace_id
+ AND t.status IN ('queued','dispatched','running','waiting_local_directory','deferred')))
+ ))
+ ORDER BY r.retry_after, r.id LIMIT 100
+ FOR UPDATE OF r SKIP LOCKED
+)
+UPDATE channel_typing_reaction r SET cleanup_required = true,
+ attempts = attempts + 1,
+ retry_after = now() + LEAST(3600, 30 * power(2, LEAST(r.attempts, 7))) * interval '1 second'
+FROM candidates c WHERE r.id = c.id RETURNING r.*;
+
+-- name: SettleChannelTypingInputs :exec
+-- The flush owns only this context's unowned inputs up to its captured cutoff.
+-- Later inputs, even in the same session, are not settled by this failed run.
+UPDATE chat_message AS message SET channel_typing_settled = true
+FROM chat_message AS cutoff, chat_session AS session, channel_installation AS installation
+WHERE cutoff.id = @through_message_id AND cutoff.chat_session_id = @chat_session_id
+ AND session.id = @chat_session_id AND session.workspace_id = @workspace_id
+ AND installation.id = @installation_id AND installation.workspace_id = @workspace_id
+ AND installation.channel_type = 'feishu'
+ AND EXISTS (SELECT 1 FROM channel_chat_session_binding b WHERE b.chat_session_id = session.id
+ AND b.installation_id = installation.id AND b.channel_type = 'feishu')
+ AND message.chat_session_id = session.id AND message.task_id IS NULL
+ AND message.channel_ingested AND message.role = 'user'
+ AND COALESCE(message.channel_context_revision, 1) = @context_revision
+ AND (message.created_at, message.id) <= (cutoff.created_at, cutoff.id);
+
+-- name: PruneChannelTypingReactionCleanup :exec
+-- Each terminal outcome remains inspectable for seven days without credentials.
+WITH candidates AS (
+ SELECT id FROM channel_typing_reaction
+ WHERE cleaned_at < now() - interval '7 days' OR abandoned_at < now() - interval '7 days'
+ LIMIT 1000 FOR UPDATE SKIP LOCKED
+)
+DELETE FROM channel_typing_reaction r USING candidates c WHERE r.id = c.id;
+
+-- name: ExpireChannelTypingReactionCleanup :many
+-- Independent maintenance, including active/uncertain Adds: the cosmetic badge
+-- has a seven-day maximum lifetime. Remote failure is NOT reported as success.
+WITH candidates AS (
+ SELECT id FROM channel_typing_reaction
+ WHERE cleaned_at IS NULL AND abandoned_at IS NULL AND created_at <= now() - interval '7 days'
+ ORDER BY created_at, id LIMIT 1000 FOR UPDATE SKIP LOCKED
+)
+UPDATE channel_typing_reaction r SET abandoned_at = now(), cleanup_required = true,
+ installation_snapshot = '{}'::jsonb
+FROM candidates c WHERE r.id = c.id
+RETURNING r.id, r.workspace_id;
+
+-- name: SkipChannelTypingReactionAdd :exec
+-- No HTTP Add was issued for an already-settled input.
+UPDATE channel_typing_reaction SET add_finished = true, cleanup_required = true,
+ cleaned_at = now(), installation_snapshot = '{}'::jsonb WHERE id = $1 AND abandoned_at IS NULL;
+
+-- name: ListRequestedChannelTypingReactions :many
+SELECT id FROM channel_typing_reaction WHERE id = ANY(@ids::uuid[]) AND cleanup_required;