-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathtemporal-tools.ts
More file actions
401 lines (351 loc) · 10.9 KB
/
Copy pathtemporal-tools.ts
File metadata and controls
401 lines (351 loc) · 10.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
/**
* EP09 Temporal Analysis - Tool Contracts
*
* Defines the interface for temporal analysis tools following
* the Claude Agent SDK `tool()` pattern per ADR-0005.
*
* @module specs/ep09-temporal-analysis/contracts
*/
import { z } from 'zod';
import type {
BaselineDelta,
TrendAnalysis,
QualitativeReview,
ReviewDimensionName,
} from './temporal-types';
// =============================================================================
// Zod Schemas for Tool Inputs
// =============================================================================
/**
* Schema for store_baseline tool input.
*/
export const StoreBaselineInputSchema = z.object({
label: z
.string()
.optional()
.describe('Optional label for this baseline (e.g., "Post-CLAUDE.md rewrite")'),
notes: z.string().optional().describe('Optional notes about this baseline'),
includeSessionMetrics: z
.boolean()
.optional()
.default(true)
.describe('Include session metrics from EP06 analysis'),
});
export type StoreBaselineInput = z.infer<typeof StoreBaselineInputSchema>;
/**
* Schema for query_baseline tool input.
*/
export const QueryBaselineInputSchema = z.object({
id: z.string().optional().describe('Baseline UUID, or omit for "latest"'),
includeFindings: z
.boolean()
.optional()
.default(false)
.describe('Include full findings array (increases response size)'),
});
export type QueryBaselineInput = z.infer<typeof QueryBaselineInputSchema>;
/**
* Schema for list_baselines tool input.
*/
export const ListBaselinesInputSchema = z.object({
limit: z.number().optional().default(20).describe('Maximum number of baselines to return'),
after: z.string().optional().describe('Only return baselines after this ISO-8601 date'),
before: z.string().optional().describe('Only return baselines before this ISO-8601 date'),
label: z.string().optional().describe('Filter by label (partial match)'),
orderBy: z
.enum(['createdAt', 'findingsCount'])
.optional()
.default('createdAt')
.describe('Sort field'),
order: z.enum(['asc', 'desc']).optional().default('desc').describe('Sort direction'),
});
export type ListBaselinesInput = z.infer<typeof ListBaselinesInputSchema>;
/**
* Schema for calculate_delta tool input.
*/
export const CalculateDeltaInputSchema = z.object({
fromId: z.string().describe('UUID of the older baseline'),
toId: z.string().describe('UUID of the newer baseline (or "latest")'),
includeGitCommits: z
.boolean()
.optional()
.default(true)
.describe('Include git commits between baselines'),
detailedDiff: z
.boolean()
.optional()
.default(false)
.describe('Include full jsondiffpatch delta (verbose)'),
});
export type CalculateDeltaInput = z.infer<typeof CalculateDeltaInputSchema>;
/**
* Schema for query_trends tool input.
*/
export const QueryTrendsInputSchema = z.object({
metrics: z
.array(z.string())
.optional()
.describe('Specific metrics to analyze (default: all)'),
afterDate: z.string().optional().describe('Analyze baselines after this ISO-8601 date'),
beforeDate: z.string().optional().describe('Analyze baselines before this ISO-8601 date'),
minBaselines: z
.number()
.optional()
.default(3)
.describe('Minimum baselines required for trend analysis'),
includeQualitative: z
.boolean()
.optional()
.default(true)
.describe('Include qualitative review trends'),
includeCorrelations: z
.boolean()
.optional()
.default(true)
.describe('Correlate trends with git commits'),
});
export type QueryTrendsInput = z.infer<typeof QueryTrendsInputSchema>;
/**
* Schema for conduct_review tool input.
*/
export const ConductReviewInputSchema = z.object({
baselineId: z
.string()
.optional()
.describe('Baseline to attach review to (default: latest)'),
dimensions: z
.array(
z.enum([
'perceivedFriction',
'trustCalibration',
'taskFit',
'configurationConfidence',
'improvementAttribution',
'workflowSatisfaction',
])
)
.optional()
.describe('Specific dimensions to cover (default: all)'),
triggerReason: z
.enum(['scheduled', 'triggered', 'manual'])
.optional()
.default('manual')
.describe('Why this review is being conducted'),
});
export type ConductReviewInput = z.infer<typeof ConductReviewInputSchema>;
/**
* Schema for get_review_history tool input.
*/
export const GetReviewHistoryInputSchema = z.object({
baselineId: z.string().optional().describe('Filter reviews by baseline'),
afterDate: z.string().optional().describe('Reviews after this ISO-8601 date'),
beforeDate: z.string().optional().describe('Reviews before this ISO-8601 date'),
dimension: z
.enum([
'perceivedFriction',
'trustCalibration',
'taskFit',
'configurationConfidence',
'improvementAttribution',
'workflowSatisfaction',
])
.optional()
.describe('Filter by dimension'),
limit: z.number().optional().default(20).describe('Maximum reviews to return'),
});
export type GetReviewHistoryInput = z.infer<typeof GetReviewHistoryInputSchema>;
// =============================================================================
// Tool Result Types
// =============================================================================
/**
* Result from store_baseline tool.
*/
export interface StoreBaselineResult {
success: boolean;
baselineId: string;
createdAt: string;
label?: string;
metrics: {
findingsCount: number;
warningCount: number;
avgTokensPerSession?: number;
};
message: string;
}
/**
* Result from query_baseline tool.
*/
export interface QueryBaselineResult {
found: boolean;
baseline?: {
id: string;
createdAt: string;
projectPath: string;
actType: string;
gitCommit?: string;
label?: string;
metrics: Record<string, number>;
findingsCount: number;
findings?: unknown[];
};
message?: string;
}
/**
* Result from list_baselines tool.
*/
export interface ListBaselinesResult {
baselines: Array<{
id: string;
createdAt: string;
label?: string;
gitCommit?: string;
findingsCount: number;
criticalCount: number;
highCount: number;
}>;
total: number;
hasMore: boolean;
}
/**
* Result from calculate_delta tool.
*/
export interface CalculateDeltaResult {
success: boolean;
delta?: BaselineDelta;
error?: string;
}
/**
* Result from query_trends tool.
*/
export interface QueryTrendsResult {
success: boolean;
analysis?: TrendAnalysis;
error?: string;
insufficientData?: {
baselines: number;
required: number;
message: string;
};
}
/**
* Partial result during conduct_review - represents a single dimension prompt.
*/
export interface ReviewPrompt {
dimension: ReviewDimensionName;
promptText: string;
previousResponse?: string;
sentimentScale: string;
}
/**
* Result from conduct_review tool.
*/
export interface ConductReviewResult {
success: boolean;
review?: QualitativeReview;
prompts?: ReviewPrompt[];
status: 'prompting' | 'complete' | 'error';
error?: string;
}
/**
* Result from get_review_history tool.
*/
export interface GetReviewHistoryResult {
reviews: Array<{
id: string;
baselineId: string;
createdAt: string;
overallSentiment: number;
themes: string[];
triggerReason?: string;
}>;
total: number;
hasMore: boolean;
}
// =============================================================================
// Tool Descriptions (for SDK registration)
// =============================================================================
/**
* Rich tool descriptions following ADR-0005 guidelines.
* These are used when registering tools with the Claude Agent SDK.
*/
export const TOOL_DESCRIPTIONS = {
store_baseline: `
Store a baseline snapshot of the current project state.
Use this tool when you need to:
- Capture a point-in-time snapshot for future comparison
- Mark a milestone (e.g., "before refactoring", "after config update")
- Establish a baseline for tracking improvement over time
The baseline includes:
- Analysis findings and metrics
- Configuration analysis results
- Session statistics (if available)
- Git commit reference for correlation
Returns the baseline ID for future reference.
`.trim(),
query_baseline: `
Retrieve a baseline snapshot by ID or get the latest baseline.
Use this tool when you need to:
- Get the most recent baseline for comparison
- Retrieve a specific historical baseline by ID
- Access baseline metrics for trend analysis
Returns structured baseline data with metrics.
Set includeFindings=true for full findings array (larger response).
`.trim(),
list_baselines: `
List available baseline snapshots with filtering and sorting.
Use this tool when you need to:
- See the history of baselines for a project
- Find baselines in a date range
- Look up baselines by label
Returns summary information without full findings.
Use query_baseline with specific ID for full details.
`.trim(),
calculate_delta: `
Calculate the difference between two baselines.
Use this tool when you need to:
- Compare current state to a previous baseline
- Identify what improved or regressed
- See the magnitude and direction of changes
Returns:
- Metric changes with direction indicators (↑ ↓ →)
- Warnings added/resolved
- Recommendations added/resolved
- Overall trend classification
Set includeGitCommits=true to see commits in the date range.
`.trim(),
query_trends: `
Analyze trends across multiple baselines over time.
Use this tool when you need to:
- See improvement trajectory over time
- Identify when trends changed direction (inflection points)
- Correlate changes with git commits
- Compare quantitative and qualitative trends
Requires at least 3 baselines (configurable via minBaselines).
Returns insufficient data message if not enough baselines exist.
`.trim(),
conduct_review: `
Facilitate a structured qualitative review session.
Use this tool when you need to:
- Capture user's subjective experience with AI-assisted workflow
- Gather context that metrics alone cannot reveal
- Build qualitative trend data over time
The review covers 6 dimensions:
1. Perceived Friction - Where friction occurs
2. Trust Calibration - How user verifies AI suggestions
3. Task Fit - What tasks work well/poorly
4. Configuration Confidence - Confidence in CLAUDE.md
5. Improvement Attribution - What changes made difference
6. Workflow Satisfaction - Overall satisfaction
Each dimension uses a Likert scale (-2 to +2).
Review is attached to a baseline for correlation.
`.trim(),
get_review_history: `
Retrieve qualitative review history for trend analysis.
Use this tool when you need to:
- See how sentiment has evolved over time
- Compare qualitative trends with quantitative metrics
- Find reviews for specific baselines or time ranges
Returns review summaries with sentiment and themes.
Use query_trends for integrated quantitative/qualitative analysis.
`.trim(),
} as const;