-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathsummary_task.py
More file actions
253 lines (209 loc) · 9.16 KB
/
Copy pathsummary_task.py
File metadata and controls
253 lines (209 loc) · 9.16 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
#!/usr/bin/env python3
"""
Keyword Extraction Task - Self-Evolving Framework
This script demonstrates using the self-evolving framework to optimize
a prompt for automatic keyword extraction from text paragraphs.
Uses two LLM-based scorers:
- coverage_score: 指关键词占原有句子中概念的比例(LLM 评估)
- precision_score: 指关键词在原文关键字的比例(LLM 评估)
Run with:
python summary_task.py
"""
import asyncio
import os
from typing import Optional
from openai import OpenAI
# Import from the modular package
from self_evo import (
SelfEvolvingAgent,
Evaluator,
PromptVersioner,
MetaPromptAgent,
LLMAsJudgeScorer,
create_default_scorers,
generate_evaluation_report,
)
def create_keyword_dataset():
"""Create a sample dataset for keyword extraction testing.
Each sample contains the text content and a list of key terms
that represent the main concepts in the text.
"""
return [
{
"section_number": "1",
"content": """Python is a high-level, interpreted programming language known for its
clean syntax and readability. It supports multiple programming paradigms including
procedural, object-oriented, and functional programming. Python has a large standard
library and is widely used for web development, data science, automation, and more.""",
"key_terms": ["python", "programming", "language", "syntax", "library", "web", "data"],
},
{
"section_number": "2",
"content": """Machine learning is a subset of artificial intelligence that enables
systems to learn and improve from experience. Machine learning focuses on the
development of computer programs that can access data and use it to learn for
themselves. The main focus is on the development of algorithms that can modify
themselves as they encounter new data.""",
"key_terms": ["machine", "learning", "artificial", "intelligence", "algorithm", "data", "program"],
},
{
"section_number": "3",
"content": """The quick brown fox jumps over the lazy dog. This sentence contains every
letter of the English alphabet and is commonly used for typing practice and font
display testing. It demonstrates how pangrams work in the English language.""",
"key_terms": ["quick", "brown", "fox", "lazy", "dog", "pangram", "alphabet"],
},
{
"section_number": "4",
"content": """Deep learning is a subset of machine learning based on artificial neural networks
with representation learning. Learning can be supervised, semi-supervised or
unsupervised. Deep learning architectures include convolutional neural networks,
recurrent neural networks, and transformers. These have revolutionized AI in areas
like computer vision and natural language processing.""",
"key_terms": ["deep", "learning", "neural", "network", "transformer", "convolutional", "vision"],
},
{
"section_number": "5",
"content": """Python decorators are a special syntax that allows you to modify the behavior
of a function or class. A decorator is a function that takes another function and
returns a new function. They are often used for logging, access control, caching,
and other cross-cutting concerns in applications.""",
"key_terms": ["python", "decorator", "function", "class", "syntax", "logging", "control"],
},
]
def create_keyword_agent(client: OpenAI, captured_prompt: str):
"""Create a keyword extraction agent with a specific prompt.
The agent uses closure to capture the prompt, so each agent instance
is bound to its prompt and uses it for all requests.
This agent also supports being called with an explicit `prompt` keyword
argument, allowing the caller to override the captured prompt.
"""
def agent(input_text: str, prompt: Optional[str] = None) -> str:
system_prompt = prompt if prompt is not None else captured_prompt
response = client.chat.completions.create(
model="Qwen3Coder",
messages=[
{"role": "system", "content": system_prompt},
{"role": "user", "content": input_text},
],
temperature=0.7,
)
return response.choices[0].message.content
return agent
async def run_keyword_task():
"""Run the keyword extraction task with self-evolution."""
api_key = os.getenv("OPENAI_API_KEY", "not-needed")
base_url = "http://192.168.1.159:19000/v1"
# Set environment variable for scorers to use
os.environ["OPENAI_BASE_URL"] = base_url
client = OpenAI(
api_key=api_key,
base_url=base_url,
)
# Get initial prompt from versioner
initial_prompt = """You are a helpful assistant that extracts keywords from text.
Provide 2-6 most important keywords from the given text. Output only the keywords."""
agent = create_keyword_agent(client, initial_prompt)
# Create evaluator with LLM-based scorers
evaluator = Evaluator(
scorers=create_default_scorers(),
lenient_pass_ratio=0.75,
lenient_avg_threshold=0.70,
)
# Coverage Score - LLM evaluates how many concepts from the text are captured
# coverage_score = extracted_keywords / original_concepts_in_text
coverage_scorer = LLMAsJudgeScorer(
name="coverage",
system_prompt="""You are an expert at evaluating keyword extraction quality.
Task: Evaluate the coverage score - how well the extracted keywords represent
the main concepts in the original text.
Instructions:
1. Read the original text and identify its main concepts/keywords
2. Read the extracted keywords provided by the system
3. Calculate: coverage = (extracted keywords that match original concepts) / (total original concepts)
4. Return a score between 0.0 and 1.0
Scoring:
- 1.0: All original concepts are captured
- 0.8: Most concepts captured (80%+)
- 0.6: Moderate coverage (60%+)
- 0.4: Partial coverage (40%+)
- 0.2: Low coverage (20%+)
- 0.0: Very few or no concepts captured
Respond with ONLY the numeric score (e.g., 0.8, 1.0, 0.5).""",
input_keys=["content"],
output_key="output_text",
pass_threshold=0.5,
model="Qwen3Coder",
)
evaluator.scorers.insert(0, coverage_scorer)
# Precision Score - LLM evaluates how many extracted keywords are truly relevant
# precision_score = relevant_keywords / total_extracted_keywords
precision_scorer = LLMAsJudgeScorer(
name="precision",
system_prompt="""You are an expert at evaluating keyword extraction quality.
Task: Evaluate the precision score - how many of the extracted keywords are
actually relevant and meaningful for the given text.
Instructions:
1. Read the original text and understand its main topic
2. Read the extracted keywords provided by the system
3. Determine how many extracted keywords are truly relevant to the text
4. Calculate: precision = (relevant keywords) / (total extracted keywords)
5. Return a score between 0.0 and 1.0
Scoring:
- 1.0: All extracted keywords are highly relevant
- 0.8: Most keywords are relevant (80%+)
- 0.6: Moderate precision (60%+)
- 0.4: Some irrelevant keywords (40%+)
- 0.2: Many irrelevant keywords (20%+)
- 0.0: Most or all keywords are irrelevant
Relevant keywords are nouns, technical terms, and concepts that directly relate
to the text's main topic. Avoid counting generic words like "the", "and", "for".
Respond with ONLY the numeric score (e.g., 0.8, 1.0, 0.5).""",
input_keys=["content"],
output_key="output_text",
pass_threshold=0.6,
model="Qwen3Coder",
)
evaluator.scorers.insert(0, precision_scorer)
# Initial prompt for keyword extraction
initial_prompt = """You are a helpful assistant that extracts keywords from text.
Provide 2-6 most important keywords from the given text. Output only the keywords."""
prompter = PromptVersioner(initial_prompt=initial_prompt, model="Qwen3Coder")
meta_agent = MetaPromptAgent(
model="Qwen3Coder",
client=client,
)
evolver = SelfEvolvingAgent(
base_agent=agent,
evaluator=evaluator,
prompter=prompter,
meta_agent=meta_agent,
max_retries=3,
use_async=False,
)
dataset = create_keyword_dataset()
print("=" * 80)
print("Keyword Extraction Task - Self-Evolving Framework")
print("=" * 80)
print(f"\nInitial prompt:\n{initial_prompt}\n")
print("=" * 80)
# Run evolution
best_prompt = await evolver.evolve(dataset, verbose=True)
print("\n" + "=" * 80)
print("OPTIMIZATION COMPLETE")
print("=" * 80)
print(f"\nBest prompt (version {best_prompt.version}):")
print(best_prompt.prompt)
# Save results
evolver.save_results("keyword_results.json")
print("\nResults saved to keyword_results.json")
# Generate and save evaluation report
with open("keyword_results.json", "r") as f:
import json
results = json.load(f)
report = generate_evaluation_report(results, format="markdown")
with open("keyword_report.md", "w") as f:
f.write(report)
print("Report saved to keyword_report.md")
if __name__ == "__main__":
asyncio.run(run_keyword_task())