Skip to content

Commit 2f8a4a3

Browse files
Hashwanth SutharapuHashwanth Sutharapu
authored andcommitted
Add comprehensive Scrapy-based content extraction system
- Implement full-scale Scrapy research service with exhaustive extraction - Add progress logging in ChatGPT style for real-time updates - Support multiple search sources (DuckDuckGo, Google, Bing) - Add content extraction choice (Firecrawl/Scrapy/Hybrid) - Include Newspaper3k for article content extraction - Add BeautifulSoup fallback for general content - Implement relevance scoring and content filtering - Add comprehensive error handling and logging - Update requirements.txt with Scrapy dependencies - Add test script for Scrapy functionality - Maintain backward compatibility with Firecrawl
1 parent 6e49a3c commit 2f8a4a3

6 files changed

Lines changed: 959 additions & 1046 deletions

File tree

‎advanced-agent/main.py‎

Lines changed: 44 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -83,6 +83,41 @@ def show_relevant_pdfs(workflow, query: str) -> List[str]:
8383
print(f"✅ Selected {len(selected_keys)} PDF(s) for research")
8484
return selected_keys
8585

86+
def show_content_extraction_choice() -> str:
87+
"""Show content extraction options to user"""
88+
print("\n🔍 Content Extraction Options:")
89+
print("=" * 50)
90+
print("1. 🚀 Firecrawl (Fast & Reliable)")
91+
print(" - Uses Firecrawl API")
92+
print(" - High-quality content extraction")
93+
print(" - Limited by API credits")
94+
print()
95+
print("2. 🕷️ Scrapy (Comprehensive & Free)")
96+
print(" - Multi-source web scraping")
97+
print(" - Exhaustive content extraction")
98+
print(" - No API limits, completely free")
99+
print(" - Shows real-time progress")
100+
print()
101+
print("3. 🔄 Hybrid (Both)")
102+
print(" - Combines Firecrawl + Scrapy")
103+
print(" - Maximum coverage")
104+
print(" - Uses API credits + free scraping")
105+
print()
106+
107+
while True:
108+
choice = input("Choose extraction method (1-3): ").strip()
109+
if choice == "1":
110+
return "firecrawl"
111+
elif choice == "2":
112+
return "scrapy"
113+
elif choice == "3":
114+
return "hybrid"
115+
else:
116+
print("❌ Invalid choice. Please enter 1, 2, or 3.")
117+
118+
def show_progress(message: str):
119+
"""Display progress updates in ChatGPT style"""
120+
print(f"📊 {message}")
86121

87122
def main():
88123
workflow = Workflow()
@@ -107,6 +142,9 @@ def main():
107142
if not query:
108143
continue
109144

145+
# Get content extraction choice
146+
extraction_method = show_content_extraction_choice()
147+
110148
# Handle PDF selection based on choice
111149
selected_pdf_keys = []
112150
if choice == "2":
@@ -116,13 +154,16 @@ def main():
116154
# Find relevant PDFs for query
117155
selected_pdf_keys = show_relevant_pdfs(workflow, query)
118156

119-
# Run research with selected PDFs
157+
# Set progress callback
158+
workflow.set_progress_callback(show_progress)
159+
160+
# Run research with selected extraction method and PDFs
120161
if selected_pdf_keys:
121162
# Use user-selected PDFs
122-
result = workflow.run_with_selected_pdfs(query, selected_pdf_keys)
163+
result = workflow.run_with_selected_pdfs_and_extraction(query, selected_pdf_keys, extraction_method)
123164
else:
124165
# Use regular research (auto-select PDFs)
125-
result = workflow.run(query)
166+
result = workflow.run_with_extraction_method(query, extraction_method)
126167

127168
# Display results
128169
print(f"\n📊 Results for: {query}")

‎advanced-agent/requirements.txt‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -6,8 +6,10 @@ anthropic==0.18.1
66

77
# Web scraping and content extraction
88
firecrawl==0.1.0
9+
scrapy==2.11.0
910
requests==2.31.0
1011
beautifulsoup4==4.12.2
12+
newspaper3k==0.2.8
1113

1214
# PDF processing
1315
pdfplumber==0.10.3

0 commit comments

Comments
 (0)