2026-05-31 23:58:26 +09:00
|
|
|
# src/goal_based_extractor.py
|
|
|
|
|
"""
|
|
|
|
|
Goal-based content extraction prompt inspired by Alibaba Tongyi DeepResearch.
|
|
|
|
|
"""
|
|
|
|
|
|
2026-06-06 11:37:10 +02:00
|
|
|
EXTRACTOR_SYSTEM = """Extract relevant information from a webpage for a given research goal.
|
2026-05-31 23:58:26 +09:00
|
|
|
|
2026-06-06 11:37:10 +02:00
|
|
|
Goal: {goal}
|
2026-05-31 23:58:26 +09:00
|
|
|
|
2026-06-06 11:37:10 +02:00
|
|
|
Task guidelines:
|
|
|
|
|
1. Locate the specific sections directly related to the goal within the provided webpage content.
|
|
|
|
|
2. Identify and extract the most relevant information; output full original context where possible, up to three or more paragraphs.
|
|
|
|
|
3. Organize into a concise paragraph with logical flow, judging each piece of information's contribution to the goal.
|
2026-05-31 23:58:26 +09:00
|
|
|
|
2026-06-06 11:37:10 +02:00
|
|
|
Respond in JSON with exactly these fields: "rational", "evidence", "summary".
|
2026-05-31 23:58:26 +09:00
|
|
|
|
2026-06-06 11:37:10 +02:00
|
|
|
Example:
|
2026-05-31 23:58:26 +09:00
|
|
|
{{
|
|
|
|
|
"rational": "This section discusses X which directly relates to the goal of understanding Y",
|
|
|
|
|
"evidence": "Full quotes and context from the page...",
|
|
|
|
|
"summary": "Concise summary of how this information answers the goal"
|
|
|
|
|
}}
|
|
|
|
|
"""
|