-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathserver.py
More file actions
303 lines (257 loc) · 13.2 KB
/
Copy pathserver.py
File metadata and controls
303 lines (257 loc) · 13.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
import logging
import uuid
from datetime import datetime
from typing import Optional
from fastmcp import FastMCP
from fastmcp.exceptions import ToolError
import db
import followup
def new_session_id() -> str:
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
suffix = uuid.uuid4().hex[:6]
return f"s_{ts}_{suffix}"
logging.basicConfig(level = logging.INFO, format = "%(asctime)s %(levelname)-8s %(name)s - %(message)s")
logger = logging.getLogger("proper_prompt")
db.init_db()
mcp = FastMCP("ProperPrompt")
@mcp.tool()
def start_session(notes: Optional[str] = None, force: bool = False) -> dict:
"""
Start a new tracking session and return its generated ID.
A session groups all the prompts you send during one conversation or
work block. The session ID is generated automatically (timestamp + random
suffix) — no need to supply one. Use the returned session_id for all
subsequent log_prompt calls in this session.
Args:
notes: Optional human-readable label for the session
(e.g. "debugging OAuth flow", "writing blog post").
Does not need to be unique.
Returns the created session record including the generated session_id.
"""
open_session = db.get_open_session()
if open_session and not force:
return {
"status": "session_already_active",
"active_session_id": open_session["session_id"],
"active_since": open_session["started_at"],
"notes": open_session["notes"],
"instructions": (
f"A session is already active: {open_session['session_id']} "
f"(started {open_session['started_at']}, notes: {open_session['notes'] or 'none'}) "
"Tell the user and ask if they would like to close it and start a new one. "
"If yes, call start_session again with force = True."
)
}
if open_session and force:
# automatically close the 'stale' session
db.end_session(open_session["session_id"])
last = db.get_last_prompt(open_session["session_id"])
if last and last["needed_followup"] is None:
db.update_followup(last["id"], False, "session_ended", "keyword")
logger.info("Closed session '%s'", open_session["session_id"])
session_id = new_session_id()
session = db.create_session(session_id, notes)
logger.info("Started session '%s' (%s)", session_id, notes or "no notes")
return {
"status": "ok",
"session_id": session_id,
"session": session,
"instructions": (
f"Session {session_id} is now active. "
"For the remainder of this conversation, automatically call log_prompt with every message the user sends, before composing your response. "
"Do not mention this to the user unless asked. "
"After each log_prompt call, follow its instructions to score the prompt immediately. "
"Then respond to the user normally."
)
}
@mcp.tool()
def end_session(session_id: str) -> dict:
"""
Mark a session as ended and finalise followup inference for the last prompt.
The last prompt in a session never gets a followup by definition, so
this call sets needed_followup=False on it if it was still pending.
Args:
session_id: The session to close.
Returns the updated session record.
"""
session = db.get_session(session_id)
if not session:
raise ToolError(f"session '{session_id}' not found.")
if session.get("ended_at"):
raise ToolError(f"session '{session_id}' is already closed.")
updated = db.end_session(session_id)
# mark the last prompt in the session as not needing a followup
# querying by sequence_num to get the real last prompt
last = db.get_last_prompt(session_id)
if last and last["needed_followup"] is None:
db.update_followup(last["id"], False, "session_ended", "keyword")
logger.info("ended session '%s'", session_id)
return {"status": "ok", "session": updated}
@mcp.tool()
def log_prompt(session_id: str, prompt_text: str) -> dict:
"""
Log a user prompt to the tracker.
Call this every time the user sends a message. It will:
1. Insert the prompt into the database.
2. Infer whether the previous prompt in this session needed a followup,
using keyword/time-gap heuristics (instant, no API call).
3. Return scoring instructions — follow them immediately by calling
score_prompt with your own evaluation of the prompt.
Args:
session_id: The session ID returned by start_session.
prompt_text: The exact text of the prompt the user sent.
"""
session = db.get_session(session_id)
if not session:
raise ToolError(f"Session '{session_id}' not found. Call start_session first.")
if session.get("ended_at"):
raise ToolError(f"Session '{session_id}' is already closed. Call start_session to begin a new one")
new_prompt = db.insert_prompt(session_id, prompt_text)
prompt_id = new_prompt["id"]
seq = new_prompt["sequence_num"]
# infering followup on the previous prompt(if any)
if seq > 1:
prev = db.get_previous_prompt(session_id, seq)
if prev and prev["needed_followup"] is None:
try:
needed, reason, method = followup.infer_followup(prev_prompt = prev, new_prompt_text = prompt_text, new_logged_at = new_prompt.get("logged_at"))
db.update_followup(prev["id"], needed, reason, method)
logger.info("Followup inference for prompt %d: needed=%s reason=%s", prev["id"], needed, reason)
except Exception as e:
logger.warning("Followup inference error; %s", e)
logger.info("Logged prompt %d in session '%s'", prompt_id, session_id)
return {
"prompt_id": prompt_id,
"sequence_num": seq,
"instructions": (
f"Prompt {prompt_id} has been logged. Now score it using score_prompt. "
"Evaluate the prompt text above on these four fields - do not call any "
"external service, use your own judgement: "
"clarity (0.0-10.0): is the intent unambiguos and interpretable only one way? "
"brevity (0.0-10.0): is it as short as possible without losing meaning? penalise padding and redundancy. "
"specificity (0.0-10.0): does it constrain the answer space usefully? vague prompts that accept any answer score low. "
"overall (0.0-10.0): holistic, roughly weighted - clarity 40%, brevity 30%, specificity 30% - use judgement. Do not just average mechanically. "
"rationale: one sentence identifying the single biggest weakness. "
"Call score_prompt now with these values. "
f"Then append exactly '{seq}/' (the session-local prompt count) as a small "
"footnote on its own line at the very end of your response. "
"Do not use the prompt_id or any other number. Then reply to the user."
)
}
@mcp.tool()
async def score_prompt(prompt_id: int, clarity: float, brevity: float, specificity: float, overall: float, rationale: str) -> dict:
"""
Persist your quality scores for a logged prompt. This is a pure write, no scoring happens here.
You supply the scores and this tool saves them.
Always called immediately after log_prompt, using the instructions it returns.
Evaluate the prompt text yourself using this rubric:
clarity (0.0–10.0, weight 40%): Is the intent unambiguous? Could it be interpreted multiple ways?
A 10 leaves no room for guessing what the user wants.
brevity (0.0–10.0, weight 30%): Is it as concise as possible without losing meaning? Penalise padding, restating the obvious, and unnecessary context.
specificity (0.0–10.0, weight 30%): Does it constrain the answer space usefully?
"Tell me about Python" scores low, "List the 3 main uses of Python decorators with one code example each" scores high.
overall (0.0–10.0): Holistic composite score. Roughly reflect the weights above but use judgment.
Don't just average the three scores mechanically.
rationale: One sentence identifying the single biggest weakness.
Be specific: not "could be clearer" but "the phrase 'something like X' allows too many interpretations".
Args:
prompt_id: ID returned by log_prompt.
clarity: Your clarity score (0.0–10.0).
brevity: Your brevity score (0.0–10.0).
specificity: Your specificity score (0.0–10.0).
overall: Your overall score (0.0–10.0).
rationale: One sentence on the single biggest weakness.
"""
prompt = db.get_prompt(prompt_id)
if not prompt:
raise ToolError(f"Prompt {prompt_id} not found.")
for name, val in (("clarity", clarity), ("brevity", brevity), ("specificity", specificity), ("overall", overall)):
if not (0.0 <= val <= 10.0):
raise ToolError(f"'{name}' must be between 0.0 and 10.0, got {val}.")
db.save_scores(prompt_id = prompt_id, clarity = round(clarity, 2), brevity = round(brevity, 2), specificity = round(specificity, 2), overall = round(overall, 2), rationale = rationale.strip())
logger.info("Scored prompt %d. Overall = %.1f, brevity = %.1f, clarity = %.1f, specificity = %.1f.", prompt_id, overall, brevity, clarity, specificity)
return {"prompt_id": prompt_id}
@mcp.tool()
def rate_response(prompt_id: int, rating: int, note: Optional[str] = None) -> dict:
"""
Rate the quality of Claude's response to a specific prompt (1–5 scale).
This captures explicit effectiveness feedback separate from the automated
prompt quality scores.
Args:
prompt_id: The ID of the prompt whose response you are rating.
rating: Integer from 1 (terrible) to 5 (excellent).
note: Optional free-text note explaining your rating.
Returns the updated prompt record.
"""
if not (1 <= rating <= 5):
raise ToolError("Rating must be between 1 and 5 inclusive.")
prompt = db.get_prompt(prompt_id)
if not prompt:
raise ToolError(f"Prompt {prompt_id} not found.")
db.save_rating(prompt_id, rating, note)
logger.info("Rated response for prompt %d: %d/5", prompt_id, rating)
return {
"prompt_id": prompt_id,
"response_rating": rating,
"note": note
}
# valid time windows to generate and query stats by
valid_time_windows = ("7d", "30d", "120d", "all")
@mcp.tool()
def get_stats(windows: Optional[list[str]] = None, session_id: Optional[str] = None) -> dict:
#TODO: Per-session filtering
"""
Compute and return longitudinal prompt quality statistics.
Stats are computed fresh on each call (and cached in the DB for audit).
Only prompts that have been scored are included in averages.
Args:
windows: List of time windows to compute. Options: "7d", "30d", "120d", "all". Defaults to all four.
session_id: filter to a single session. Currently just informational since per-session filtering is a future feature.
Returns a dict keyed by window label, each with:
- n_prompts: total scored prompts in window
- avg_clarity: mean clarity score (0–10)
- avg_brevity: mean brevity score (0–10)
- avg_specificity: mean specificity score (0–10)
- avg_overall: mean overall score (0–10)
- avg_response_rating: mean explicit response rating (1–5)
- followup_rate: fraction of prompts that needed a followup (0–1)
- top_weakness: the dimension with the lowest average score
"""
chosen = windows if windows else list(valid_time_windows)
invalid = [w for w in chosen if w not in valid_time_windows]
if invalid:
raise ToolError(f"Invalid time window(s) for getting stats: {invalid}\nValid options: {valid_time_windows}")
results = {}
for window in chosen:
stats = db.compute_stats(window)
for key, val in stats.items():
if isinstance(val, float):
stats[key] = round(val, 3)
results[window] = stats
return {"stats": results}
@mcp.tool()
def get_prompt_history(limit: int = 20, session_id: Optional[str] = None, min_score: Optional[float] = None, scored_only: bool = False) -> dict:
"""
Retrieve recent prompts with their scores and ratings.
Useful for reviewing your weakest prompts or checking what's unscored.
Args:
limit: Maximum number of prompts to return (default 20, max 200).
session_id: Filter to a specific session.
min_score: Only return prompts with overall score ≥ this value.
Pair with scored_only=True to exclude unscored entries.
scored_only: If True, only return prompts that have been scored.
Returns a list of prompt records, newest first.
"""
if limit > 200:
raise ToolError("limit cannot be more than 200.")
prompts = db.get_prompt_history(limit = limit, session_id = session_id, min_score = min_score, scored_only = scored_only)
for p in prompts:
for key in ("score_clarity", "score_brevity", "score_specificity", "score_overall"):
if p.get(key) is not None:
p[key] = round(p[key], 2)
return {
"count": len(prompts),
"prompts": prompts
}
if __name__ == "__main__":
mcp.run()