feat(chat): per-thread todo list strip above the composer (#4017)

This commit is contained in:
Steven Enamakel
2026-06-23 17:31:29 -07:00
committed by GitHub
parent f05b971fb2
commit 39fd83e866
21 changed files with 668 additions and 245 deletions
+1
View File
@@ -2942,6 +2942,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'نقل لليسار',
'conversations.taskKanban.moveRight': 'نقل لليمين',
'conversations.taskKanban.title': 'المهام',
'conversations.threadTodo.title': 'الخطة',
'conversations.taskKanban.approval.default': 'التقصير',
'conversations.taskKanban.approval.notRequired': 'غير مطلوب',
'conversations.taskKanban.approval.notRequiredBadge': 'عدم الموافقة',
+1
View File
@@ -3007,6 +3007,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'বামে সরান',
'conversations.taskKanban.moveRight': 'ডানে সরান',
'conversations.taskKanban.title': 'টাস্ক',
'conversations.threadTodo.title': 'পরিকল্পনা',
'conversations.taskKanban.approval.default': 'ডিফল্ট',
'conversations.taskKanban.approval.notRequired': 'প্রয়োজন নেই',
'conversations.taskKanban.approval.notRequiredBadge': 'অনুমোদন করা হবে না',
+1
View File
@@ -3077,6 +3077,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Bewege dich nach links',
'conversations.taskKanban.moveRight': 'Bewege dich nach rechts',
'conversations.taskKanban.title': 'Aufgaben',
'conversations.threadTodo.title': 'Plan',
'conversations.taskKanban.approval.default': 'Standard',
'conversations.taskKanban.approval.notRequired': 'Nicht erforderlich',
'conversations.taskKanban.approval.notRequiredBadge': 'keine Genehmigung',
+1
View File
@@ -3536,6 +3536,7 @@ const en: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Move left',
'conversations.taskKanban.moveRight': 'Move right',
'conversations.taskKanban.title': 'Tasks',
'conversations.threadTodo.title': 'Plan',
'conversations.taskKanban.approval.default': 'Default',
'conversations.taskKanban.approval.notRequired': 'Not required',
'conversations.taskKanban.approval.notRequiredBadge': 'no approval',
+1
View File
@@ -3056,6 +3056,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Mover a la izquierda',
'conversations.taskKanban.moveRight': 'Mover a la derecha',
'conversations.taskKanban.title': 'Tareas',
'conversations.threadTodo.title': 'Plan',
'conversations.taskKanban.approval.default': 'Por defecto',
'conversations.taskKanban.approval.notRequired': 'No requerido',
'conversations.taskKanban.approval.notRequiredBadge': 'sin aprobación',
+1
View File
@@ -3071,6 +3071,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Déplacer à gauche',
'conversations.taskKanban.moveRight': 'Déplacer à droite',
'conversations.taskKanban.title': 'Tâches',
'conversations.threadTodo.title': 'Plan',
'conversations.taskKanban.approval.default': 'Par défaut',
'conversations.taskKanban.approval.notRequired': 'Non requis',
'conversations.taskKanban.approval.notRequiredBadge': 'aucune approbation',
+1
View File
@@ -3007,6 +3007,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'बाएं ले जाएं',
'conversations.taskKanban.moveRight': 'दाएं ले जाएं',
'conversations.taskKanban.title': 'टास्क',
'conversations.threadTodo.title': 'योजना',
'conversations.taskKanban.approval.default': 'डिफ़ॉल्ट',
'conversations.taskKanban.approval.notRequired': 'आवश्यकता नहीं',
'conversations.taskKanban.approval.notRequiredBadge': 'कोई अनुमोदन नहीं',
+1
View File
@@ -3010,6 +3010,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Pindah ke kiri',
'conversations.taskKanban.moveRight': 'Pindah ke kanan',
'conversations.taskKanban.title': 'Tugas',
'conversations.threadTodo.title': 'Rencana',
'conversations.taskKanban.approval.default': 'Default',
'conversations.taskKanban.approval.notRequired': 'Tidak diperlukan',
'conversations.taskKanban.approval.notRequiredBadge': 'tidak ada persetujuan',
+1
View File
@@ -3050,6 +3050,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Sposta a sinistra',
'conversations.taskKanban.moveRight': 'Sposta a destra',
'conversations.taskKanban.title': 'Attività',
'conversations.threadTodo.title': 'Piano',
'conversations.taskKanban.approval.default': 'Predefinito',
'conversations.taskKanban.approval.notRequired': 'Non richiesto',
'conversations.taskKanban.approval.notRequiredBadge': 'nessuna approvazione',
+1
View File
@@ -2981,6 +2981,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': '왼쪽으로 이동',
'conversations.taskKanban.moveRight': '오른쪽으로 이동',
'conversations.taskKanban.title': '작업',
'conversations.threadTodo.title': '계획',
'conversations.taskKanban.approval.default': '기본값',
'conversations.taskKanban.approval.notRequired': '필요 없음',
'conversations.taskKanban.approval.notRequiredBadge': '승인 없음',
+1
View File
@@ -3037,6 +3037,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Przesuń w lewo',
'conversations.taskKanban.moveRight': 'Przesuń w prawo',
'conversations.taskKanban.title': 'Zadania',
'conversations.threadTodo.title': 'Plan',
'conversations.taskKanban.approval.default': 'Domyślne',
'conversations.taskKanban.approval.notRequired': 'Niewymagane',
'conversations.taskKanban.approval.notRequiredBadge': 'bez zatwierdzenia',
+1
View File
@@ -3055,6 +3055,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Mover para esquerda',
'conversations.taskKanban.moveRight': 'Mover para direita',
'conversations.taskKanban.title': 'Tarefas',
'conversations.threadTodo.title': 'Plano',
'conversations.taskKanban.approval.default': 'Padrão',
'conversations.taskKanban.approval.notRequired': 'Não obrigatório',
'conversations.taskKanban.approval.notRequiredBadge': 'sem aprovação',
+1
View File
@@ -3029,6 +3029,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': 'Переместить влево',
'conversations.taskKanban.moveRight': 'Переместить вправо',
'conversations.taskKanban.title': 'Задачи',
'conversations.threadTodo.title': 'План',
'conversations.taskKanban.approval.default': 'По умолчанию',
'conversations.taskKanban.approval.notRequired': 'Не требуется',
'conversations.taskKanban.approval.notRequiredBadge': 'нет одобрения',
+1
View File
@@ -2860,6 +2860,7 @@ const messages: TranslationMap = {
'conversations.taskKanban.moveLeft': '向左移动',
'conversations.taskKanban.moveRight': '向右移动',
'conversations.taskKanban.title': '任务',
'conversations.threadTodo.title': '计划',
'conversations.taskKanban.approval.default': '默认',
'conversations.taskKanban.approval.notRequired': '无需审批',
'conversations.taskKanban.approval.notRequiredBadge': '无需审批',
+33 -90
View File
@@ -66,7 +66,6 @@ import {
} from '../store/threadSlice';
import type { ConfirmationModal as ConfirmationModalType } from '../types/intelligence';
import type { ThreadMessage } from '../types/thread';
import type { TaskBoardCard, TaskBoardCardStatus } from '../types/turnState';
import { splitAgentMessageIntoBubbles } from '../utils/agentMessageBubbles';
import { chatThreadPath } from '../utils/chatRoutes';
import { CHAT_ATTACHMENTS_ENABLED } from '../utils/config';
@@ -94,7 +93,7 @@ import {
} from './conversations/components/BackgroundProcessesPanel';
import { CitationChips, type MessageCitation } from './conversations/components/CitationChips';
import { SubagentDrawer } from './conversations/components/SubagentDrawer';
import { TaskKanbanBoard } from './conversations/components/TaskKanbanBoard';
import { ThreadTodoStrip } from './conversations/components/ThreadTodoStrip';
import { ToolTimelineBlock } from './conversations/components/ToolTimelineBlock';
import {
evaluateComposerSend,
@@ -1412,60 +1411,6 @@ const Conversations = ({
const shouldRenderTimelineBeforeLatestAgentMessage =
selectedThreadToolTimeline.length > 0 && !isSending && Boolean(latestVisibleAgentMessage);
const handleMoveTaskCard = async (
card: TaskBoardCard,
nextStatus: TaskBoardCardStatus
): Promise<void> => {
if (!selectedThreadId || !selectedTaskBoard) return;
const now = new Date().toISOString();
const nextBoard = {
...selectedTaskBoard,
cards: selectedTaskBoard.cards.map(existing =>
existing.id === card.id ? { ...existing, status: nextStatus, updatedAt: now } : existing
),
updatedAt: now,
};
dispatch(setTaskBoardForThread({ threadId: selectedThreadId, board: nextBoard }));
try {
const saved = await threadApi.putTaskBoard(selectedThreadId, nextBoard.cards);
if (!saved) {
throw new Error('Task board update returned no board');
}
dispatch(setTaskBoardForThread({ threadId: selectedThreadId, board: saved }));
} catch (error) {
debug('putTaskBoard failed: %o', error);
setSendAdvisory(t('conversations.taskKanban.updateFailed'));
dispatch(setTaskBoardForThread({ threadId: selectedThreadId, board: selectedTaskBoard }));
}
};
const handleUpdateTaskCard = async (
card: TaskBoardCard,
nextCard: TaskBoardCard
): Promise<void> => {
if (!selectedThreadId || !selectedTaskBoard) return;
const now = new Date().toISOString();
const nextBoard = {
...selectedTaskBoard,
cards: selectedTaskBoard.cards.map(existing =>
existing.id === card.id ? { ...nextCard, updatedAt: now } : existing
),
updatedAt: now,
};
dispatch(setTaskBoardForThread({ threadId: selectedThreadId, board: nextBoard }));
try {
const saved = await threadApi.putTaskBoard(selectedThreadId, nextBoard.cards);
if (!saved) {
throw new Error('Task board update returned no board');
}
dispatch(setTaskBoardForThread({ threadId: selectedThreadId, board: saved }));
} catch (error) {
debug('putTaskBoard failed: %o', error);
setSendAdvisory(t('conversations.taskKanban.updateFailed'));
dispatch(setTaskBoardForThread({ threadId: selectedThreadId, board: selectedTaskBoard }));
}
};
const filteredThreads = useMemo(() => {
return threads.filter(t => isThreadVisibleInTab(t, selectedLabel));
}, [threads, selectedLabel]);
@@ -1805,40 +1750,6 @@ const Conversations = ({
className={`mx-auto w-full max-w-[48.75rem] space-y-3 px-5 pt-4 ${
isSidebar ? 'pb-4' : 'pb-32'
}`}>
{selectedTaskBoard && hasTaskBoard && (
<TaskKanbanBoard
board={selectedTaskBoard}
disabled={!selectedThreadId}
onMove={(card, status) => {
void handleMoveTaskCard(card, status);
}}
onUpdateCard={(card, nextCard) => {
void handleUpdateTaskCard(card, nextCard);
}}
onDecidePlan={(card, approve) => {
void runDecidePlan({
threadId: selectedThreadId,
card,
approve,
dispatch,
notify: setSendAdvisory,
t,
});
}}
onViewSession={card => {
if (!card.sessionThreadId) return;
// Navigation only — do NOT mark the thread active. activeThreadId
// tracks a true in-flight turn (set on send, cleared on
// done/error). A completed session never emits that lifecycle
// event, so forcing it active would wedge the composer.
dispatch(setSelectedThread(card.sessionThreadId));
void dispatch(loadThreadMessages(card.sessionThreadId));
if (shouldSyncChatRoute) {
navigate(chatThreadPath(card.sessionThreadId));
}
}}
/>
)}
{visibleMessages.map(msg => {
const isAgentTextMode = msg.sender === 'agent' && agentMessageViewMode === 'text';
// Parsed once per message: for current messages (extraMetadata
@@ -2534,6 +2445,38 @@ const Conversations = ({
);
})()}
{/* Thread-scoped todo list the agent maintains as it works — read-only,
pinned above the composer. Distinct from the Intelligence-tab kanban
(global `user-tasks`). Renders nothing when the thread has no active
cards. */}
{selectedThreadId && (
<ThreadTodoStrip
board={selectedTaskBoard}
disabled={!selectedThreadId}
onDecidePlan={(card, approve) => {
void runDecidePlan({
threadId: selectedThreadId,
card,
approve,
dispatch,
notify: setSendAdvisory,
t,
});
}}
onViewSession={card => {
if (!card.sessionThreadId) return;
// Navigation only — do NOT mark the thread active. activeThreadId
// tracks a true in-flight turn; forcing a completed session active
// would wedge the composer.
dispatch(setSelectedThread(card.sessionThreadId));
void dispatch(loadThreadMessages(card.sessionThreadId));
if (shouldSyncChatRoute) {
navigate(chatThreadPath(card.sessionThreadId));
}
}}
/>
)}
{composer === 'mic-cloud' ? (
<div className="flex flex-col items-center gap-3 py-1">
<MicComposer
@@ -77,6 +77,9 @@ vi.mock('../../services/api/threadApi', () => ({
putTaskBoard: vi
.fn()
.mockResolvedValue({ threadId: 't-1', cards: [], updatedAt: '2026-05-04T10:00:00Z' }),
decidePlan: vi
.fn()
.mockResolvedValue({ threadId: 't-1', cards: [], updatedAt: '2026-05-04T10:00:00Z' }),
appendMessage: vi.fn().mockResolvedValue({}),
deleteThread: vi.fn().mockResolvedValue({ deleted: true }),
generateTitleIfNeeded: vi.fn().mockResolvedValue({}),
@@ -1296,109 +1299,6 @@ describe('Conversations — smoke render (#1123 welcome-lock removal)', () => {
});
});
it('rolls back and shows feedback when task board move persistence fails', async () => {
const thread = makeThread({ id: 'board-thread', title: 'Board Thread' });
const board = {
threadId: 'board-thread',
updatedAt: '2026-05-04T10:00:00Z',
cards: [
{
id: 'task-1',
title: 'Plan rollout',
status: 'todo' as const,
order: 0,
updatedAt: '2026-05-04T10:00:00Z',
},
],
};
mockGetThreads.mockResolvedValue({ threads: [thread], count: 1 });
mockGetThreadMessages.mockResolvedValue({ messages: [], count: 0 });
vi.mocked(threadApi.getTaskBoard).mockResolvedValueOnce(board);
vi.mocked(threadApi.putTaskBoard).mockRejectedValueOnce(new Error('write failed'));
await act(async () => {
await renderConversations({
thread: selectedThreadState(thread),
socket: socketState('connected'),
});
});
expect(await screen.findByText('Plan rollout')).toBeInTheDocument();
fireEvent.click(screen.getByLabelText('Move right'));
await waitFor(() => {
expect(
screen.getByText('Could not update task; changes were not saved.')
).toBeInTheDocument();
});
// With the 5-column model, todo → right → awaiting_approval (not in_progress)
expect(threadApi.putTaskBoard).toHaveBeenCalledWith(
'board-thread',
expect.arrayContaining([
expect.objectContaining({ id: 'task-1', status: 'awaiting_approval' }),
])
);
});
it('rolls back and shows feedback when task board edit persistence fails', async () => {
const thread = makeThread({ id: 'edit-board-thread', title: 'Edit Board Thread' });
const board = {
threadId: 'edit-board-thread',
updatedAt: '2026-05-04T10:00:00Z',
cards: [
{
id: 'task-1',
title: 'Plan rollout',
status: 'todo' as const,
objective: 'Draft the launch task brief',
assignedAgent: 'planner',
approvalMode: 'required' as const,
plan: ['Read docs'],
allowedTools: ['todo'],
acceptanceCriteria: ['Saved board round-trips'],
evidence: [],
order: 0,
updatedAt: '2026-05-04T10:00:00Z',
},
],
};
mockGetThreads.mockResolvedValue({ threads: [thread], count: 1 });
mockGetThreadMessages.mockResolvedValue({ messages: [], count: 0 });
vi.mocked(threadApi.getTaskBoard).mockResolvedValueOnce(board);
vi.mocked(threadApi.putTaskBoard).mockRejectedValueOnce(new Error('write failed'));
await act(async () => {
await renderConversations({
thread: selectedThreadState(thread),
socket: socketState('connected'),
});
});
expect(await screen.findByText('Plan rollout')).toBeInTheDocument();
fireEvent.click(screen.getByText('Task brief'));
fireEvent.change(screen.getByLabelText('Title'), { target: { value: 'Updated rollout' } });
fireEvent.change(screen.getByLabelText('Assigned agent'), {
target: { value: 'code_executor' },
});
fireEvent.click(screen.getByText('Save changes'));
await waitFor(() => {
expect(
screen.getByText('Could not update task; changes were not saved.')
).toBeInTheDocument();
});
expect(threadApi.putTaskBoard).toHaveBeenCalledWith(
'edit-board-thread',
expect.arrayContaining([
expect.objectContaining({
id: 'task-1',
title: 'Updated rollout',
assignedAgent: 'code_executor',
}),
])
);
});
it('sends with Enter when the composer is not composing text', async () => {
const { textarea, thread } = await renderSelectedConversation();
@@ -2086,4 +1986,41 @@ describe('Conversations — open-session resume (View work)', () => {
await waitFor(() => expect(store.getState().thread.selectedThreadId).toBe('sess-99'));
expect(screen.getByTestId('route-path')).toHaveTextContent('/human');
});
it('approves a parked plan card from the thread todo strip', async () => {
const thread = makeThread({ id: 'approve-thread', title: 'Approve thread' });
mockGetThreads.mockResolvedValue({ threads: [thread], count: 1 });
const store = await renderConversations({ thread: selectedThreadState(thread) });
const selectedId = store.getState().thread.selectedThreadId ?? 'approve-thread';
await act(async () => {
store.dispatch(
setTaskBoardForThread({
threadId: selectedId,
board: {
threadId: selectedId,
updatedAt: '',
cards: [
{
id: 'pc1',
title: 'Needs sign-off',
status: 'awaiting_approval',
order: 0,
updatedAt: '',
},
],
},
})
);
});
// The strip surfaces Approve/Reject only for parked cards; approving routes
// through onDecidePlan → runDecidePlan → threadApi.decidePlan.
const approveBtn = await screen.findByTitle('Approve');
await act(async () => {
fireEvent.click(approveBtn);
});
await waitFor(() => expect(threadApi.decidePlan).toHaveBeenCalledWith(selectedId, 'pc1', true));
});
});
@@ -0,0 +1,176 @@
import { fireEvent, render, screen } from '@testing-library/react';
import { describe, expect, it, vi } from 'vitest';
import type { TaskBoard, TaskBoardCard } from '../../../types/turnState';
import { ThreadTodoStrip } from './ThreadTodoStrip';
// Echo i18n keys so we can assert on the stable key string.
vi.mock('../../../lib/i18n/I18nContext', () => ({ useT: () => ({ t: (key: string) => key }) }));
function card(partial: Partial<TaskBoardCard>): TaskBoardCard {
return {
id: 'c1',
title: 'Do thing',
status: 'todo',
order: 0,
updatedAt: '',
...partial,
} as TaskBoardCard;
}
function board(cards: TaskBoardCard[]): TaskBoard {
return { threadId: 't1', cards, updatedAt: '' };
}
describe('ThreadTodoStrip', () => {
it('renders nothing when board is null', () => {
const { container } = render(<ThreadTodoStrip board={null} />);
expect(container).toBeEmptyDOMElement();
});
it('renders nothing when there are no active cards', () => {
const { container } = render(
<ThreadTodoStrip
board={board([card({ id: 'a', status: 'done' }), card({ id: 'b', status: 'rejected' })])}
/>
);
expect(container).toBeEmptyDOMElement();
});
it('lists active cards and hides done/rejected ones', () => {
render(
<ThreadTodoStrip
board={board([
card({ id: 'a', title: 'Active work', status: 'in_progress' }),
card({ id: 'b', title: 'Queued work', status: 'todo' }),
card({ id: 'c', title: 'Finished work', status: 'done' }),
card({ id: 'd', title: 'Dropped work', status: 'rejected' }),
])}
/>
);
expect(screen.getByText('Active work')).toBeInTheDocument();
expect(screen.getByText('Queued work')).toBeInTheDocument();
expect(screen.queryByText('Finished work')).not.toBeInTheDocument();
expect(screen.queryByText('Dropped work')).not.toBeInTheDocument();
});
it('shows a done/total progress count excluding rejected cards', () => {
render(
<ThreadTodoStrip
board={board([
card({ id: 'a', status: 'in_progress' }),
card({ id: 'b', status: 'done' }),
card({ id: 'c', status: 'rejected' }),
])}
/>
);
// 1 done out of 2 tracked (rejected excluded entirely).
expect(screen.getByText('1/2')).toBeInTheDocument();
});
it('orders active cards by their `order` field', () => {
render(
<ThreadTodoStrip
board={board([
card({ id: 'a', title: 'Second', status: 'todo', order: 5 }),
card({ id: 'b', title: 'First', status: 'todo', order: 1 }),
])}
/>
);
const items = screen.getAllByRole('listitem').map(li => li.textContent);
expect(items[0]).toContain('First');
expect(items[1]).toContain('Second');
});
it('falls back to objective then id when title is blank', () => {
render(
<ThreadTodoStrip
board={board([
card({ id: 'a', title: ' ', objective: 'Ship it', status: 'todo' }),
card({ id: 'only-id', title: '', objective: null, status: 'todo' }),
])}
/>
);
expect(screen.getByText('Ship it')).toBeInTheDocument();
expect(screen.getByText('only-id')).toBeInTheDocument();
});
it('collapses and expands the list when the header is clicked', () => {
render(
<ThreadTodoStrip board={board([card({ id: 'a', title: 'Active work', status: 'todo' })])} />
);
expect(screen.getByText('Active work')).toBeInTheDocument();
const header = screen.getByRole('button', { expanded: true });
fireEvent.click(header);
expect(screen.queryByText('Active work')).not.toBeInTheDocument();
fireEvent.click(screen.getByRole('button', { expanded: false }));
expect(screen.getByText('Active work')).toBeInTheDocument();
});
it('renders Approve/Reject only for awaiting_approval cards when onDecidePlan is provided', () => {
const onDecidePlan = vi.fn();
render(
<ThreadTodoStrip
board={board([
card({ id: 'parked', title: 'Needs sign-off', status: 'awaiting_approval' }),
card({ id: 'plain', title: 'Just working', status: 'in_progress' }),
])}
onDecidePlan={onDecidePlan}
/>
);
// Exactly one Approve and one Reject — only the parked card has them.
expect(screen.getAllByText('chat.approval.approve')).toHaveLength(1);
expect(screen.getAllByText('chat.approval.deny')).toHaveLength(1);
fireEvent.click(screen.getByText('chat.approval.approve'));
expect(onDecidePlan).toHaveBeenCalledWith(expect.objectContaining({ id: 'parked' }), true);
fireEvent.click(screen.getByText('chat.approval.deny'));
expect(onDecidePlan).toHaveBeenCalledWith(expect.objectContaining({ id: 'parked' }), false);
});
it('surfaces the blocker reason for blocked cards', () => {
render(
<ThreadTodoStrip
board={board([
card({ id: 'b', title: 'Stuck step', status: 'blocked', blocker: 'needs API key' }),
])}
/>
);
expect(screen.getByText('Stuck step')).toBeInTheDocument();
expect(screen.getByText('needs API key')).toBeInTheDocument();
});
it('renders a View work jump only for cards with a session when onViewSession is provided', () => {
const onViewSession = vi.fn();
render(
<ThreadTodoStrip
board={board([
card({
id: 'linked',
title: 'Has session',
status: 'in_progress',
sessionThreadId: 's1',
}),
card({ id: 'plain', title: 'No session', status: 'in_progress' }),
])}
onViewSession={onViewSession}
/>
);
expect(screen.getAllByText('conversations.taskKanban.viewWork')).toHaveLength(1);
fireEvent.click(screen.getByText('conversations.taskKanban.viewWork'));
expect(onViewSession).toHaveBeenCalledWith(expect.objectContaining({ id: 'linked' }));
});
it('stays fully read-only (no approve/reject) when onDecidePlan is omitted', () => {
render(
<ThreadTodoStrip
board={board([
card({ id: 'parked', title: 'Needs sign-off', status: 'awaiting_approval' }),
])}
/>
);
expect(screen.getByText('Needs sign-off')).toBeInTheDocument();
expect(screen.queryByText('chat.approval.approve')).not.toBeInTheDocument();
expect(screen.queryByText('chat.approval.deny')).not.toBeInTheDocument();
});
});
@@ -0,0 +1,195 @@
import React, { useMemo, useState } from 'react';
import { useT } from '../../../lib/i18n/I18nContext';
import type { TaskBoard, TaskBoardCard, TaskBoardCardStatus } from '../../../types/turnState';
/**
* Compact, read-only strip that surfaces the *current conversation thread's*
* task board ("todo list") directly above the composer. It mirrors hermes-agent's
* pinned todo panel: a one-glance plan the agent maintains as it works through a
* multi-step task.
*
* Read-only by design — the agent owns the board via its `todo` tool (mutations
* arrive live through the `task_board_updated` socket event into
* `taskBoardByThread`). Direct user editing lives in the full board UI, not here.
*
* This is NOT the Intelligence-tab kanban (that board is the global `user-tasks`
* list). This strip is strictly scoped to the selected chat thread.
*/
/** Statuses that represent live, actionable work — shown in the strip. */
const ACTIVE_STATUSES: readonly TaskBoardCardStatus[] = [
'in_progress',
'todo',
'ready',
'awaiting_approval',
'blocked',
];
/** Terminal "finished" status — counted toward progress, not listed. */
const DONE_STATUS: TaskBoardCardStatus = 'done';
/** Plain-text glyph per status (monochrome, terminal-style — like hermes). */
function statusGlyph(status: TaskBoardCardStatus): string {
switch (status) {
case 'in_progress':
return '[~]';
case 'blocked':
return '[!]';
case 'awaiting_approval':
return '[?]';
case 'done':
return '[x]';
case 'rejected':
return '[-]';
case 'todo':
case 'ready':
default:
return '[ ]';
}
}
/** Tailwind text-color token per status, matching the app's semantic palette. */
function statusColorClass(status: TaskBoardCardStatus): string {
switch (status) {
case 'in_progress':
return 'text-primary-600 dark:text-primary-300 font-medium';
case 'blocked':
return 'text-coral dark:text-coral';
case 'awaiting_approval':
return 'text-amber-700 dark:text-amber-300';
default:
return 'text-stone-600 dark:text-neutral-300';
}
}
interface Props {
board: TaskBoard | null;
/**
* Decide a parked plan (`awaiting_approval` card). When provided, those cards —
* and only those — gain inline Approve/Reject controls; every other card stays
* read-only. Omit to keep the strip fully read-only.
*/
onDecidePlan?: (card: TaskBoardCard, approve: boolean) => void;
/**
* Jump to a card's linked agent session. When provided, cards that carry a
* `sessionThreadId` (stamped by the autonomous/manual task-session flow) gain a
* "View work" affordance. Omit to hide it.
*/
onViewSession?: (card: TaskBoardCard) => void;
/** Disable the approve/reject controls (e.g. no thread selected). */
disabled?: boolean;
}
export const ThreadTodoStrip: React.FC<Props> = ({
board,
onDecidePlan,
onViewSession,
disabled = false,
}) => {
const { t } = useT();
const [collapsed, setCollapsed] = useState(false);
const { activeCards, doneCount, total } = useMemo(() => {
const cards = board?.cards ?? [];
// Exclude `rejected` cards entirely; they're neither active nor progress.
const tracked = cards.filter(c => c.status !== 'rejected');
const active = tracked
.filter(c => ACTIVE_STATUSES.includes(c.status))
.sort((a, b) => a.order - b.order);
return {
activeCards: active,
doneCount: tracked.filter(c => c.status === DONE_STATUS).length,
total: tracked.length,
};
}, [board]);
// Nothing to plan → render nothing (no empty chrome above the composer).
if (activeCards.length === 0) return null;
return (
<div
className="mb-2 rounded-xl border border-stone-200 bg-stone-50 text-xs shadow-sm dark:border-neutral-700 dark:bg-neutral-900"
data-testid="thread-todo-strip">
<button
type="button"
onClick={() => setCollapsed(prev => !prev)}
aria-expanded={!collapsed}
className="flex w-full items-center gap-1.5 px-3 py-2 text-left text-stone-500 transition-colors hover:text-stone-700 dark:text-neutral-400 dark:hover:text-neutral-200">
<span aria-hidden className="text-primary-500">
{collapsed ? '▸' : '▾'}
</span>
<span className="font-semibold text-stone-700 dark:text-neutral-200">
{t('conversations.threadTodo.title')}
</span>
<span className="text-stone-400 dark:text-neutral-500">
{doneCount}/{total}
</span>
</button>
{!collapsed && (
// Cap the expanded list so a long plan (one card per step) can't cover
// the latest messages/controls above the composer — scroll instead.
<ul className="flex max-h-48 flex-col gap-0.5 overflow-y-auto px-3 pb-2 pl-5">
{activeCards.map(card => (
<li
key={card.id}
className={`flex items-start gap-1.5 break-words ${statusColorClass(card.status)}`}>
<span aria-hidden className="font-mono">
{statusGlyph(card.status)}
</span>
<span className="flex min-w-0 flex-1 flex-col">
<span className="min-w-0">{cardLabel(card)}</span>
{card.status === 'blocked' && card.blocker?.trim() && (
// Surface why a step is stuck + what's needed next, matching
// the todo-tool guidance to set `blocked` with a `blocker`.
<span className="min-w-0 text-[11px] text-coral/80">{card.blocker.trim()}</span>
)}
</span>
{card.sessionThreadId && onViewSession && (
<button
type="button"
title={t('conversations.taskKanban.viewWork')}
onClick={() => onViewSession(card)}
className="flex-shrink-0 rounded-md border border-stone-200 px-1.5 py-0.5 text-[10px] font-medium text-stone-600 transition-colors hover:bg-stone-100 dark:border-neutral-700 dark:text-neutral-300 dark:hover:bg-neutral-800">
{t('conversations.taskKanban.viewWork')}
</button>
)}
{card.status === 'awaiting_approval' && onDecidePlan && (
<span className="flex flex-shrink-0 items-center gap-1">
<button
type="button"
title={t('chat.approval.approve')}
disabled={disabled}
onClick={() => onDecidePlan(card, true)}
className="rounded-md bg-ocean-600 px-1.5 py-0.5 text-[10px] font-medium text-white transition-colors hover:bg-ocean-700 disabled:opacity-40">
{t('chat.approval.approve')}
</button>
<button
type="button"
title={t('chat.approval.deny')}
disabled={disabled}
onClick={() => onDecidePlan(card, false)}
className="rounded-md border border-stone-200 px-1.5 py-0.5 text-[10px] font-medium text-stone-600 transition-colors hover:bg-stone-100 disabled:opacity-40 dark:border-neutral-700 dark:text-neutral-300 dark:hover:bg-neutral-800">
{t('chat.approval.deny')}
</button>
</span>
)}
</li>
))}
</ul>
)}
</div>
);
};
/** Prefer the card title; fall back to its objective, then a generic label. */
function cardLabel(card: TaskBoardCard): string {
const title = card.title?.trim();
if (title) return title;
const objective = card.objective?.trim();
if (objective) return objective;
return card.id;
}
export default ThreadTodoStrip;
@@ -0,0 +1,128 @@
// @ts-nocheck
/**
* Chat thread todo strip — end-to-end.
*
* Exercises the per-conversation-thread todo list, top to bottom:
* - The mock LLM emits a `todo` tool call (op:add) on turn 1, then a final
* answer on turn 2.
* - The Rust core executes the thread-bound `todo` tool against the selected
* thread's task board and emits a `task_board_updated` progress event.
* - The frontend records the board into `chatRuntime.taskBoardByThread[tid]`.
* - The read-only `ThreadTodoStrip` (`[data-testid="thread-todo-strip"]`)
* renders above the composer with the agent-authored card.
*
* This is the feature E2E for the "todo list per thread" surface: it proves the
* agent can author its plan AND that the plan renders in the chat view — the
* full path the unit tests can only stub.
*/
import { waitForApp } from '../helpers/app-helpers';
import {
chatMounted,
clickByTitle,
clickSend,
getSelectedThreadId,
typeIntoComposer,
waitForSocketConnected,
} from '../helpers/chat-harness';
import { textExists } from '../helpers/element-helpers';
import { resetApp } from '../helpers/reset-app';
import { navigateViaHash } from '../helpers/shared-flows';
import { clearRequestLog, setMockBehavior, startMockServer, stopMockServer } from '../mock-server';
const LOG_PREFIX = '[chat-thread-todo-strip]';
const USER_ID = 'e2e-chat-thread-todo-strip';
const PROMPT = 'Plan the multi-step refactor before you start.';
const CARD_TITLE = 'canary-todo-card-7e2a9d';
const FINAL_REPLY = 'Plan recorded — starting now. canary-final-3f1c';
// Turn 1: the LLM writes a card to the thread board via the `todo` tool.
// Turn 2: it answers now that the plan exists.
const FORCED_RESPONSES = [
{
content: '',
toolCalls: [
{
id: 'call_todo_add_1',
name: 'todo',
arguments: JSON.stringify({ op: 'add', content: CARD_TITLE, status: 'in_progress' }),
},
],
},
{ content: FINAL_REPLY },
];
/** Read the selected thread's board cards straight out of the redux store. */
async function boardCardTitles(threadId: string): Promise<string[]> {
return (await browser.execute((tid: string) => {
const winAny = window as unknown as { __OPENHUMAN_STORE__?: { getState: () => unknown } };
const state = winAny.__OPENHUMAN_STORE__?.getState() as
| {
chatRuntime?: {
taskBoardByThread?: Record<string, { cards?: Array<{ title?: string }> }>;
};
}
| undefined;
const board = state?.chatRuntime?.taskBoardByThread?.[tid];
return (board?.cards ?? []).map(c => c?.title ?? '');
}, threadId)) as string[];
}
describe('Chat thread todo strip', () => {
before(async function beforeSuite() {
this.timeout(90_000);
await startMockServer();
await waitForApp();
await resetApp(USER_ID);
setMockBehavior('llmForcedResponses', JSON.stringify(FORCED_RESPONSES));
setMockBehavior('llmStreamChunkDelayMs', '10');
clearRequestLog();
console.log(`${LOG_PREFIX} setup complete — forced todo tool-call configured`);
});
after(async () => {
setMockBehavior('llmForcedResponses', '');
setMockBehavior('llmStreamChunkDelayMs', '');
await stopMockServer();
});
it('renders the agent-authored card in the thread todo strip above the composer', async () => {
await navigateViaHash('/chat');
await browser.waitUntil(async () => await chatMounted(), {
timeout: 15_000,
timeoutMsg: 'Conversations panel did not mount',
});
expect(await clickByTitle('New thread', 8_000)).toBe(true);
const threadId = (await browser.waitUntil(async () => await getSelectedThreadId(), {
timeout: 8_000,
timeoutMsg: 'thread.selectedThreadId never populated',
})) as string;
expect(typeof threadId).toBe('string');
await typeIntoComposer(PROMPT);
if (!(await waitForSocketConnected(30_000))) {
console.warn(`${LOG_PREFIX} socket did not connect within 30 s — send may fail`);
}
expect(
await browser.waitUntil(async () => await clickSend(), {
timeout: 5_000,
timeoutMsg: 'Send button never enabled',
})
).toBe(true);
// The board lands in redux once the core processes the `todo` tool call and
// emits task_board_updated.
await browser.waitUntil(async () => (await boardCardTitles(threadId)).includes(CARD_TITLE), {
timeout: 45_000,
timeoutMsg: 'thread board never received the agent-authored card',
});
// The read-only strip mounts above the composer and lists the active card.
const strip = await $('[data-testid="thread-todo-strip"]');
await strip.waitForExist({ timeout: 15_000 });
expect(await strip.isDisplayed()).toBe(true);
expect(await textExists(CARD_TITLE)).toBe(true);
console.log(`${LOG_PREFIX} passed — strip rendered the agent-authored card`);
});
});
+51 -50
View File
@@ -176,12 +176,13 @@ Canonical mapping of every product feature to its test source(s). Drives gap-fil
### 4.2 Messaging
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ----- | -------------------------------------------------------------- | ----- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------- |
| 4.2.1 | User Message Handling | WD+RI | `conversations-web-channel-flow.spec.ts`, `tests/json_rpc_e2e.rs` | ✅ | |
| 4.2.2 | AI Response Generation | WD | `agent-review.spec.ts` | ✅ | Mock LLM |
| 4.2.3 | Streaming Responses | RI | `tests/json_rpc_e2e.rs`, `tests/agent_harness_e2e.rs` | ✅ | `tests/agent_harness_e2e.rs` adds provider-level SSE tool-arg accumulation (chunked args reassembled + parsed) and engine-level delta forwarding (#3471) |
| 4.2.4 | Parallel inference (cross-thread + within-thread forked turns) | RU+VU | `src/openhuman/channels/providers/web_tests.rs`, `app/src/store/__tests__/chatRuntimeSlice.test.ts`, `app/src/providers/__tests__/ChatRuntimeProvider.test.tsx` | 🟡 | Concurrent same-/cross-thread dispatch, cooperative `CancellationToken` teardown, and parallel-lane stream routing covered; dedicated WD E2E is a follow-up |
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ----- | -------------------------------------------------------------- | -------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| 4.2.1 | User Message Handling | WD+RI | `conversations-web-channel-flow.spec.ts`, `tests/json_rpc_e2e.rs` | ✅ | |
| 4.2.2 | AI Response Generation | WD | `agent-review.spec.ts` | ✅ | Mock LLM |
| 4.2.3 | Streaming Responses | RI | `tests/json_rpc_e2e.rs`, `tests/agent_harness_e2e.rs` | ✅ | `tests/agent_harness_e2e.rs` adds provider-level SSE tool-arg accumulation (chunked args reassembled + parsed) and engine-level delta forwarding (#3471) |
| 4.2.4 | Parallel inference (cross-thread + within-thread forked turns) | RU+VU | `src/openhuman/channels/providers/web_tests.rs`, `app/src/store/__tests__/chatRuntimeSlice.test.ts`, `app/src/providers/__tests__/ChatRuntimeProvider.test.tsx` | 🟡 | Concurrent same-/cross-thread dispatch, cooperative `CancellationToken` teardown, and parallel-lane stream routing covered; dedicated WD E2E is a follow-up |
| 4.2.5 | Per-thread todo list (plan strip above composer) | RU+VU+WD | `src/openhuman/agent/tools/todo.rs`, `app/src/pages/conversations/components/ThreadTodoStrip.test.tsx`, `app/test/e2e/specs/chat-thread-todo-strip.spec.ts` | ✅ | Read-only thread-scoped todo strip fed by `task_board_updated`; agent `todo` tool guidance + thread binding; E2E drives a `todo` tool call and asserts the card renders |
### 4.3 Tool Invocation
@@ -199,20 +200,20 @@ Canonical mapping of every product feature to its test source(s). Drives gap-fil
End-to-end coverage of the agent harness via the web-chat RPC surface against an in-test scripted upstream (`tests/agent_harness_e2e.rs`) plus browser-level Redux/DOM assertions (`app/test/e2e/specs/agent-harness-behaviors.spec.ts`). Added for #3471.
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ------ | --------------------------------------- | ----- | --------------------------------------------------------------------------------------------------------------------------- | ------ | -------------------------------------------------------------------------------------------------------------- |
| 4.4.1 | Subagent Delegation | RI+WD | `tests/agent_harness_e2e.rs`, `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Orchestrator→researcher delegation asserts ≥3 upstream calls, distinct researcher persona, no "Unknown tool:" |
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ------ | --------------------------------------- | ----- | ---------------------------------------------------------------------------------------------------------------------------- | ------ | --------------------------------------------------------------------------------------------------------------- |
| 4.4.1 | Subagent Delegation | RI+WD | `tests/agent_harness_e2e.rs`, `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Orchestrator→researcher delegation asserts ≥3 upstream calls, distinct researcher persona, no "Unknown tool:" |
| 4.4.2 | Subagent Clarification (input-required) | RI+WD | `tests/agent_harness_e2e.rs`, `app/test/e2e/specs/agent-harness-behaviors.spec.ts`, `chat-harness-subagent-continue.spec.ts` | ✅ | Subagent question surfaced + reply completes turn; full `continue_subagent` resume covered by the continue spec |
| 4.4.3 | Approval Gate — Approve | RI+WD | `tests/agent_harness_e2e.rs`, `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | `approval_request` event → approve → tool executes (timeline `success`, file written) |
| 4.4.4 | Approval Gate — Deny | RI+WD | `tests/agent_harness_e2e.rs`, `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Deny → tool not executed (timeline `error`, file unchanged), agent explains gracefully |
| 4.4.5 | Approval Gate — TTL Timeout | RI | `tests/agent_harness_e2e.rs` | ✅ | `OPENHUMAN_APPROVAL_TTL_SECS` (debug-only) → no decision → auto-deny |
| 4.4.6 | Subagent + Approval Gate | RI | `tests/agent_harness_e2e.rs` | ✅ | Approval fires inside subagent context; decision propagates; artifact content asserted |
| 4.4.7 | Multi-turn State Persistence | RI | `tests/agent_harness_e2e.rs` | ✅ | Turn-2 upstream request carries turn-1 user + assistant messages |
| 4.4.8 | Parallel Subagent Fan-out | RI | `tests/agent_harness_e2e.rs` | ✅ | `spawn_parallel_agents` 2-task fan-out; both child results reach synthesis call |
| 4.4.9 | Max Iterations / Empty Response | RI | `tests/agent_harness_e2e.rs` | ✅ | Tool-call limit surface + empty-completion graceful terminal (both skip Sentry) |
| 4.4.10 | Provider Error Retry | RI | `tests/agent_harness_e2e.rs` | ✅ | First upstream 500 retried by ReliableProvider; second succeeds |
| 4.4.11 | Inference Phase Transitions | WD | `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Redux `inferenceStatusByThread` observes `subagent` phase then clears to idle |
| 4.4.12 | Tool Timeline Completeness | WD | `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Timeline entries carry id/name/status/round; subagent row reaches `success`; rounds non-decreasing |
| 4.4.3 | Approval Gate — Approve | RI+WD | `tests/agent_harness_e2e.rs`, `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | `approval_request` event → approve → tool executes (timeline `success`, file written) |
| 4.4.4 | Approval Gate — Deny | RI+WD | `tests/agent_harness_e2e.rs`, `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Deny → tool not executed (timeline `error`, file unchanged), agent explains gracefully |
| 4.4.5 | Approval Gate — TTL Timeout | RI | `tests/agent_harness_e2e.rs` | ✅ | `OPENHUMAN_APPROVAL_TTL_SECS` (debug-only) → no decision → auto-deny |
| 4.4.6 | Subagent + Approval Gate | RI | `tests/agent_harness_e2e.rs` | ✅ | Approval fires inside subagent context; decision propagates; artifact content asserted |
| 4.4.7 | Multi-turn State Persistence | RI | `tests/agent_harness_e2e.rs` | ✅ | Turn-2 upstream request carries turn-1 user + assistant messages |
| 4.4.8 | Parallel Subagent Fan-out | RI | `tests/agent_harness_e2e.rs` | ✅ | `spawn_parallel_agents` 2-task fan-out; both child results reach synthesis call |
| 4.4.9 | Max Iterations / Empty Response | RI | `tests/agent_harness_e2e.rs` | ✅ | Tool-call limit surface + empty-completion graceful terminal (both skip Sentry) |
| 4.4.10 | Provider Error Retry | RI | `tests/agent_harness_e2e.rs` | ✅ | First upstream 500 retried by ReliableProvider; second succeeds |
| 4.4.11 | Inference Phase Transitions | WD | `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Redux `inferenceStatusByThread` observes `subagent` phase then clears to idle |
| 4.4.12 | Tool Timeline Completeness | WD | `app/test/e2e/specs/agent-harness-behaviors.spec.ts` | ✅ | Timeline entries carry id/name/status/round; subagent row reaches `success`; rounds non-decreasing |
---
@@ -274,17 +275,17 @@ End-to-end coverage of the agent harness via the web-chat RPC surface against an
### 6.3 Sub-agent Orchestration
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ----- | ---------------------------- | ----- | --------------------------------------------------------------------------------------------------------------------- | ------ | -------------------------------------------------------------------------------------------------------------------------------------------------- |
| 6.3.1 | Steer a running sub-agent | RU | `src/openhuman/agent_orchestration/running_subagents.rs`, `src/openhuman/agent_orchestration/tools/steer_subagent.rs` | ✅ | `steer_subagent` injects a steer/collect message into a running async sub-agent's run-queue; registry enforces parent ownership + terminal guard. |
| 6.3.2 | Wait for a sub-agent result | RU | `src/openhuman/agent_orchestration/running_subagents.rs`, `src/openhuman/agent_orchestration/tools/wait_subagent.rs` | ✅ | `wait_subagent` blocks on the completion `watch` with a timeout; prunes terminal entries, leaves entries intact on timeout. |
| 6.3.3 | Steer lands in child history | RU | `src/openhuman/agent/harness/subagent_runner/ops_tests.rs::run_queue_steer_lands_in_subagent_history` | ✅ | End-to-end: a queued steer is drained by the child `run_turn_engine` and appears as a `[User steering message]` user turn in the provider request. |
| 6.3.4 | Subconscious trigger pipeline (normalize → dedupe/rate → gate → queue) | RU+RI | `src/openhuman/subconscious_triggers/`, `tests/subconscious_triggers_e2e.rs` | ✅ | Event→Trigger normalization for cron/user/composio/sub-agent, dedupe TTL + per-source rate limit, LLM gate over `agent::triage`, priority queue with overflow eviction. |
| 6.3.5 | Long-lived subconscious orchestrator session | RU | `src/openhuman/subconscious/session.rs`, `src/openhuman/subconscious/user_thread.rs` | ✅ | Persistent compressed session backed by a reserved thread; `notify_user` handoff to the user-facing thread; mode→autonomy config parity. |
| 6.3.6 | Multi-party human↔subconscious↔sub-agent conversation | RI | `tests/subconscious_conversation_e2e.rs` | ✅ | Scripted Gate/SessionExecutor seam drives delegate→sub-agent→merge, failure/retry, interleaving, dedupe, and rate-limit scenarios through the real orchestrator. |
| 6.3.7 | Full-stack trigger pipeline with mocked LLM | RI | `tests/subconscious_fullstack_e2e.rs` (feature `e2e-test-support`) | ✅ | Real `GatePass`+`LongLivedSession`+`Agent`+sub-agent run against a provider-layer mock (no network); promote/drop, persistence, real `spawn_subagent`. |
| 6.3.8 | Subconscious Triggers debug/manage panel (Brain) | WD | `app/test/playwright/specs/subconscious-triggers.spec.ts` | ✅ | Brain→Subconscious panel: renders disabled baseline + hint + reserved thread ids; enable toggle → Pipeline Enabled + event_driven + orchestrator running; disable; refresh re-fetches. |
| 6.3.9 | Vision sub-agent reads attached images | RU | `src/openhuman/agent_registry/agents/loader.rs::vision_agent_loads_on_vision_hint`, `src/openhuman/inference/provider/factory_tests.rs::vision_tier_is_vision_capable`, `src/openhuman/agent/harness/engine/core.rs::gate_tests`, `src/openhuman/agent/multimodal_tests.rs::extract_image_placeholders_pulls_att_tokens_in_order` | ✅ | Orchestrator (non-vision `chat-v1`) keeps the image as a placeholder, delegates to `vision_agent` on the `vision-v1` tier, which rehydrates the on-disk attachment and reads it. Engine gate prefers per-tier `current_model_vision`; turn placeholders forwarded into the sub-agent prompt. |
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ----- | ---------------------------------------------------------------------- | ----- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| 6.3.1 | Steer a running sub-agent | RU | `src/openhuman/agent_orchestration/running_subagents.rs`, `src/openhuman/agent_orchestration/tools/steer_subagent.rs` | ✅ | `steer_subagent` injects a steer/collect message into a running async sub-agent's run-queue; registry enforces parent ownership + terminal guard. |
| 6.3.2 | Wait for a sub-agent result | RU | `src/openhuman/agent_orchestration/running_subagents.rs`, `src/openhuman/agent_orchestration/tools/wait_subagent.rs` | ✅ | `wait_subagent` blocks on the completion `watch` with a timeout; prunes terminal entries, leaves entries intact on timeout. |
| 6.3.3 | Steer lands in child history | RU | `src/openhuman/agent/harness/subagent_runner/ops_tests.rs::run_queue_steer_lands_in_subagent_history` | ✅ | End-to-end: a queued steer is drained by the child `run_turn_engine` and appears as a `[User steering message]` user turn in the provider request. |
| 6.3.4 | Subconscious trigger pipeline (normalize → dedupe/rate → gate → queue) | RU+RI | `src/openhuman/subconscious_triggers/`, `tests/subconscious_triggers_e2e.rs` | ✅ | Event→Trigger normalization for cron/user/composio/sub-agent, dedupe TTL + per-source rate limit, LLM gate over `agent::triage`, priority queue with overflow eviction. |
| 6.3.5 | Long-lived subconscious orchestrator session | RU | `src/openhuman/subconscious/session.rs`, `src/openhuman/subconscious/user_thread.rs` | ✅ | Persistent compressed session backed by a reserved thread; `notify_user` handoff to the user-facing thread; mode→autonomy config parity. |
| 6.3.6 | Multi-party human↔subconscious↔sub-agent conversation | RI | `tests/subconscious_conversation_e2e.rs` | ✅ | Scripted Gate/SessionExecutor seam drives delegate→sub-agent→merge, failure/retry, interleaving, dedupe, and rate-limit scenarios through the real orchestrator. |
| 6.3.7 | Full-stack trigger pipeline with mocked LLM | RI | `tests/subconscious_fullstack_e2e.rs` (feature `e2e-test-support`) | ✅ | Real `GatePass`+`LongLivedSession`+`Agent`+sub-agent run against a provider-layer mock (no network); promote/drop, persistence, real `spawn_subagent`. |
| 6.3.8 | Subconscious Triggers debug/manage panel (Brain) | WD | `app/test/playwright/specs/subconscious-triggers.spec.ts` | ✅ | Brain→Subconscious panel: renders disabled baseline + hint + reserved thread ids; enable toggle → Pipeline Enabled + event_driven + orchestrator running; disable; refresh re-fetches. |
| 6.3.9 | Vision sub-agent reads attached images | RU | `src/openhuman/agent_registry/agents/loader.rs::vision_agent_loads_on_vision_hint`, `src/openhuman/inference/provider/factory_tests.rs::vision_tier_is_vision_capable`, `src/openhuman/agent/harness/engine/core.rs::gate_tests`, `src/openhuman/agent/multimodal_tests.rs::extract_image_placeholders_pulls_att_tokens_in_order` | ✅ | Orchestrator (non-vision `chat-v1`) keeps the image as a placeholder, delegates to `vision_agent` on the `vision-v1` tier, which rehydrates the on-disk attachment and reads it. Engine gate prefers per-tier `current_model_vision`; turn placeholders forwarded into the sub-agent prompt. |
---
@@ -349,10 +350,10 @@ End-to-end coverage of the agent harness via the web-chat RPC surface against an
### 8.5 Long-term Goals
| ID | Feature | Test | Source / Test File | Status | Notes |
| ----- | ----------------------------- | ----- | ------------------------------------------------------------------------------------------------------------------------------------------------- | ------ | --------------------------------------------------------------------------------------------- |
| 8.5.1 | Goals CRUD (list/add/edit/delete) | RU+VU | `src/openhuman/memory_goals/store.rs`, `src/openhuman/memory_goals/ops.rs`, `src/openhuman/memory_goals/tools.rs`, `app/src/services/api/goalsApi.test.ts`, `app/src/components/intelligence/GoalsPanel.test.tsx` | ✅ | Editable `MEMORY_GOALS.md` list over `memory_goals_*` RPC + Brain > Goals UI |
| 8.5.2 | Goals enrichment (reflect) | RU+VU | `src/openhuman/memory_goals/enrich.rs`, `src/openhuman/memory_goals/schemas.rs`, `app/src/components/intelligence/GoalsPanel.test.tsx` | 🟡 | Turn-based `goals_agent` enrichment; prompt/registry/error paths unit-tested, live LLM run manual |
| ID | Feature | Test | Source / Test File | Status | Notes |
| ----- | --------------------------------- | ----- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------ | ------------------------------------------------------------------------------------------------- |
| 8.5.1 | Goals CRUD (list/add/edit/delete) | RU+VU | `src/openhuman/memory_goals/store.rs`, `src/openhuman/memory_goals/ops.rs`, `src/openhuman/memory_goals/tools.rs`, `app/src/services/api/goalsApi.test.ts`, `app/src/components/intelligence/GoalsPanel.test.tsx` | ✅ | Editable `MEMORY_GOALS.md` list over `memory_goals_*` RPC + Brain > Goals UI |
| 8.5.2 | Goals enrichment (reflect) | RU+VU | `src/openhuman/memory_goals/enrich.rs`, `src/openhuman/memory_goals/schemas.rs`, `app/src/components/intelligence/GoalsPanel.test.tsx` | 🟡 | Turn-based `goals_agent` enrichment; prompt/registry/error paths unit-tested, live LLM run manual |
---
@@ -452,22 +453,22 @@ End-to-end coverage of the agent harness via the web-chat RPC surface against an
### 11.1 Analysis Engine
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ------- | -------------------------------------- | ----- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| 11.1.1 | Multi-Source Analysis | RI | `tests/memory_graph_sync_e2e.rs` | 🟡 | Frontend trigger untested |
| 11.1.2 | Actionable Item Extraction | VU | `app/src/components/intelligence/__tests__/utils.test.ts` | ✅ | Was ❌ |
| 11.1.3 | Analyze Trigger | WD | `app/test/e2e/specs/insights-dashboard.spec.ts` mounts the route; explicit analyze-handler invocation TBD | 🟡 | Route mounts and search/filter UI assert — full analyze trigger flow tracked as follow-up |
| 11.1.4 | MCP server (stdio + HTTP) | RU | `src/openhuman/mcp_server/` | ✅ | Stdio framing plus Streamable HTTP/SSE session lifecycle; `McpHttpClient` round-trip tests |
| 11.1.5 | Global tool registry | RI | `src/openhuman/tool_registry/`, `tests/json_rpc_e2e.rs`, `tests/domain_modules_e2e.rs`, `tests/worker_b_domain_e2e.rs` | ✅ | Read-only MCP/controller discovery with routes, schemas, version, allowed agents, and health |
| 11.1.6 | SearXNG MCP search | RU | `src/openhuman/integrations/searxng.rs`, `src/openhuman/mcp_server/tools.rs`, `src/openhuman/tools/schemas.rs` | ✅ | Self-hosted search config, normalized results, MCP argument validation, and mocked HTTP execution |
| 11.1.7 | Bundled prompt resources | RU | `src/openhuman/mcp_server/resources.rs`, `src/openhuman/mcp_server/protocol.rs` | ✅ | `resources/list` catalog + `resources/read` happy path, -32002 unknown URI, -32602 missing param, catalog-mirrors-BUILTINS parity test |
| 11.1.8 | Resource templates list | RU | `src/openhuman/mcp_server/resources.rs`, `src/openhuman/mcp_server/protocol.rs` | ✅ | `resources/templates/list` returns `{resourceTemplates: []}` (static catalog), tolerates unknown/cursor params |
| 11.1.10 | MCP registry install→connect→tool_call | RI | `tests/json_rpc_e2e.rs` (`mcp_clients_install_connect_tool_call_happy_path`), `tests/mcp_registry_e2e.rs`, `src/openhuman/mcp_registry/setup_ops.rs` (#3039) | ✅ | HTTP-RPC happy path install→connect→tool_call→update_env against `test-mcp-stub`; transport-aware install (stdio + http_remote) via `build_install_transport` |
| 11.1.11 | MCP env reconfigure + registry creds | RI/VU | `tests/json_rpc_e2e.rs` (`mcp_clients_registry_settings_roundtrip`), `src/openhuman/mcp_registry/registries/mcp_official.rs`, `app/src/components/channels/mcp/InstalledServerDetail.test.tsx` (#3039) | ✅ | `update_env` persist+reconnect; `registry_settings` get/set with secrets write-only (config-first, env-fallback); reconfigure form validation |
| 11.1.12 | MCP UI surface + setup-agent client | VU | `app/src/components/channels/mcp/InstallDialog.test.tsx`, `app/src/services/api/mcpClientsApi.test.ts`, `app/src/services/api/mcpSetupApi.test.ts` (#3039) | ✅ | Skills `?tab=mcp` renders `McpServersTab` (not Coming Soon); auto-connect on install (best-effort); typed `mcpSetupApi` wrapper |
| 11.1.13 | MCP HTTP-remote auth (token / Bearer / OAuth) + redirect resolution | RU/VU | `src/openhuman/mcp_registry/connections.rs` (`build_http_auth*`, `resolve_final_url`), `src/openhuman/mcp_registry/oauth.rs` (PKCE/token/bundle/callback port), `app/src/components/channels/mcp/ConnectAuthModal.test.tsx` (#3495) | ✅ | Bearer/raw scheme + custom headers; redirect-final-URL resolved before auth; OAuth dynamic client registration + PKCE + refresh; tokens MERGED into stored env; credentials stored encrypted locally, never sent to backend |
| 11.1.14 | MCP "Help & configure" assistant | VU/RU | `app/src/components/channels/mcp/ConfigAssistantPanel.test.tsx`, `app/src/components/channels/mcp/ConfigHelpModal.test.tsx`, `src/openhuman/mcp_registry/ops.rs` (`invoke_config_assist_agent`) (#3495) | ✅ | Fixed server-specific prompt runs an agentic turn scoped to web_search_tool/web_fetch/curl only; markdown-rendered reply; per-MCP chat persisted while on the detail page |
| 11.1.15 | Agent uses connected MCP servers in chat | RU | `src/openhuman/agent_registry/agents/loader.rs` (`orchestrator_subagents_include_mcp_agent`, `mcp_agent_drives_connected_servers_without_install_or_shell`, `planner_has_readonly_mcp_discovery_not_execute`), `src/openhuman/agent_registry/agents/orchestrator/prompt.rs` (`connected_mcp_block_*`), `src/openhuman/agent/harness/session/turn_tests.rs` (`mcp_announcement_fires_once_for_new_server`), `src/openhuman/mcp_registry/{tools,connections}.rs` (#3495) | ✅ | `use_mcp_server` delegate → `mcp_agent` worker (discover→list→call); `mcp_registry_list_tools` read-only discovery; orchestrator `## Connected MCP Servers` prompt block + mid-session connect announcement on the user turn; planner read-only MCP discovery (no `tool_call`) |
| ID | Feature | Layer | Test path(s) | Status | Notes |
| ------- | ------------------------------------------------------------------- | ----- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
| 11.1.1 | Multi-Source Analysis | RI | `tests/memory_graph_sync_e2e.rs` | 🟡 | Frontend trigger untested |
| 11.1.2 | Actionable Item Extraction | VU | `app/src/components/intelligence/__tests__/utils.test.ts` | ✅ | Was ❌ |
| 11.1.3 | Analyze Trigger | WD | `app/test/e2e/specs/insights-dashboard.spec.ts` mounts the route; explicit analyze-handler invocation TBD | 🟡 | Route mounts and search/filter UI assert — full analyze trigger flow tracked as follow-up |
| 11.1.4 | MCP server (stdio + HTTP) | RU | `src/openhuman/mcp_server/` | ✅ | Stdio framing plus Streamable HTTP/SSE session lifecycle; `McpHttpClient` round-trip tests |
| 11.1.5 | Global tool registry | RI | `src/openhuman/tool_registry/`, `tests/json_rpc_e2e.rs`, `tests/domain_modules_e2e.rs`, `tests/worker_b_domain_e2e.rs` | ✅ | Read-only MCP/controller discovery with routes, schemas, version, allowed agents, and health |
| 11.1.6 | SearXNG MCP search | RU | `src/openhuman/integrations/searxng.rs`, `src/openhuman/mcp_server/tools.rs`, `src/openhuman/tools/schemas.rs` | ✅ | Self-hosted search config, normalized results, MCP argument validation, and mocked HTTP execution |
| 11.1.7 | Bundled prompt resources | RU | `src/openhuman/mcp_server/resources.rs`, `src/openhuman/mcp_server/protocol.rs` | ✅ | `resources/list` catalog + `resources/read` happy path, -32002 unknown URI, -32602 missing param, catalog-mirrors-BUILTINS parity test |
| 11.1.8 | Resource templates list | RU | `src/openhuman/mcp_server/resources.rs`, `src/openhuman/mcp_server/protocol.rs` | ✅ | `resources/templates/list` returns `{resourceTemplates: []}` (static catalog), tolerates unknown/cursor params |
| 11.1.10 | MCP registry install→connect→tool_call | RI | `tests/json_rpc_e2e.rs` (`mcp_clients_install_connect_tool_call_happy_path`), `tests/mcp_registry_e2e.rs`, `src/openhuman/mcp_registry/setup_ops.rs` (#3039) | ✅ | HTTP-RPC happy path install→connect→tool_call→update_env against `test-mcp-stub`; transport-aware install (stdio + http_remote) via `build_install_transport` |
| 11.1.11 | MCP env reconfigure + registry creds | RI/VU | `tests/json_rpc_e2e.rs` (`mcp_clients_registry_settings_roundtrip`), `src/openhuman/mcp_registry/registries/mcp_official.rs`, `app/src/components/channels/mcp/InstalledServerDetail.test.tsx` (#3039) | ✅ | `update_env` persist+reconnect; `registry_settings` get/set with secrets write-only (config-first, env-fallback); reconfigure form validation |
| 11.1.12 | MCP UI surface + setup-agent client | VU | `app/src/components/channels/mcp/InstallDialog.test.tsx`, `app/src/services/api/mcpClientsApi.test.ts`, `app/src/services/api/mcpSetupApi.test.ts` (#3039) | ✅ | Skills `?tab=mcp` renders `McpServersTab` (not Coming Soon); auto-connect on install (best-effort); typed `mcpSetupApi` wrapper |
| 11.1.13 | MCP HTTP-remote auth (token / Bearer / OAuth) + redirect resolution | RU/VU | `src/openhuman/mcp_registry/connections.rs` (`build_http_auth*`, `resolve_final_url`), `src/openhuman/mcp_registry/oauth.rs` (PKCE/token/bundle/callback port), `app/src/components/channels/mcp/ConnectAuthModal.test.tsx` (#3495) | ✅ | Bearer/raw scheme + custom headers; redirect-final-URL resolved before auth; OAuth dynamic client registration + PKCE + refresh; tokens MERGED into stored env; credentials stored encrypted locally, never sent to backend |
| 11.1.14 | MCP "Help & configure" assistant | VU/RU | `app/src/components/channels/mcp/ConfigAssistantPanel.test.tsx`, `app/src/components/channels/mcp/ConfigHelpModal.test.tsx`, `src/openhuman/mcp_registry/ops.rs` (`invoke_config_assist_agent`) (#3495) | ✅ | Fixed server-specific prompt runs an agentic turn scoped to web_search_tool/web_fetch/curl only; markdown-rendered reply; per-MCP chat persisted while on the detail page |
| 11.1.15 | Agent uses connected MCP servers in chat | RU | `src/openhuman/agent_registry/agents/loader.rs` (`orchestrator_subagents_include_mcp_agent`, `mcp_agent_drives_connected_servers_without_install_or_shell`, `planner_has_readonly_mcp_discovery_not_execute`), `src/openhuman/agent_registry/agents/orchestrator/prompt.rs` (`connected_mcp_block_*`), `src/openhuman/agent/harness/session/turn_tests.rs` (`mcp_announcement_fires_once_for_new_server`), `src/openhuman/mcp_registry/{tools,connections}.rs` (#3495) | ✅ | `use_mcp_server` delegate → `mcp_agent` worker (discover→list→call); `mcp_registry_list_tools` read-only discovery; orchestrator `## Connected MCP Servers` prompt block + mid-session connect announcement on the user turn; planner read-only MCP discovery (no `tool_call`) |
<!-- 11.1.9 Vault Markdown Writes — removed: Knowledge Vaults dropped (vault domain + VaultPanel deleted). -->
+31 -2
View File
@@ -35,7 +35,17 @@ impl Tool for TodoTool {
}
fn description(&self) -> &str {
"Manage the agent's per-thread task board. Dispatch via the `op` field: \
"Maintain a visible plan for THIS conversation thread (the cards render \
above the user's composer and survive across turns). \
Use it for any request with 3+ steps or several distinct tasks: at the \
start, `add` one card per step up front; keep exactly ONE card \
`in_progress` at a time; mark a card `done` the moment it is finished \
(do not batch completions); if a step fails or is abandoned, set it \
`blocked` with a `blocker`, or revise it. `list` to re-read the current \
plan when resuming. Skip this tool for trivial single-step requests. \
The board is bound automatically to the current thread — do not pass a \
thread id. \
Dispatch via the `op` field: \
`add` (content, status?, objective?, plan?, assignedAgent?, allowedTools?, \
approvalMode?, acceptanceCriteria?, evidence?, notes?, blocker?), \
`edit` (id, content?, status?, objective?, plan?, assignedAgent?, allowedTools?, \
@@ -45,7 +55,6 @@ impl Tool for TodoTool {
`replace` (cards: full list — wholesale replace), \
`clear`, or `list`. \
`status` is one of `todo` / `in_progress` / `blocked` / `done`. \
At most one card may be `in_progress` at a time. \
Returns the updated list as cards plus a markdown rendering."
}
@@ -347,6 +356,26 @@ mod tests {
assert!(err.to_string().contains("content"));
}
#[test]
fn description_carries_planning_guidance() {
// The `todo` tool steers the live orchestrator purely through its static
// (prompt-cache-stable) schema description — there is no per-thread prompt
// injection. Lock in the behavioural contract so the guidance can't be
// silently dropped: when-to-use, single-in_progress discipline, and the
// "bound to the current thread, don't pass a thread id" rule.
let tool = TodoTool::new();
let desc = tool.description();
assert!(desc.contains("3+ steps"), "missing when-to-use guidance");
assert!(
desc.contains("ONE card `in_progress`"),
"missing single-in_progress discipline"
);
assert!(
desc.contains("do not pass a thread id"),
"missing explicit 'do not pass a thread id' note"
);
}
#[tokio::test]
async fn edit_rejects_unknown_id() {
let _guard = scratch_lock();