May 17, 2026, 11:40 PM
This commit is contained in:
@@ -0,0 +1,134 @@
|
||||
import asyncio
|
||||
import json
|
||||
from bs4 import BeautifulSoup
|
||||
from crawl4ai import AsyncWebCrawler
|
||||
|
||||
async def get_latest_news_futurepedia():
|
||||
# Simplified and escaped CSS selector for the table
|
||||
table_selector = "div#data tbody.PaginatedTable_tbody__NkLBr"
|
||||
|
||||
async with AsyncWebCrawler(verbose=True) as crawler:
|
||||
result = await crawler.arun(
|
||||
url="https://www.futurepedia.io/ai-innovations",
|
||||
css_selector=table_selector, # Target the table
|
||||
bypass_cache=True, # Always fetch fresh content
|
||||
wait_for=f"css:{table_selector}", # Ensure the table loads
|
||||
)
|
||||
|
||||
if result.success:
|
||||
# Parse the extracted HTML with BeautifulSoup
|
||||
soup = BeautifulSoup(result.cleaned_html, 'html.parser')
|
||||
rows = soup.find_all('tr') # Find all table rows
|
||||
|
||||
# Process each row
|
||||
headers = []
|
||||
table_data = []
|
||||
|
||||
for i, row in enumerate(rows):
|
||||
cells = row.find_all(['td', 'th']) # Find table cells
|
||||
row_data = []
|
||||
|
||||
for j, cell in enumerate(cells):
|
||||
if j == len(cells) - 1: # Check if it's the last column
|
||||
# Extract link if available
|
||||
link = cell.find('a')
|
||||
if link and link.has_attr('href'):
|
||||
row_data.append({"text": cell.get_text(strip=True), "link": link['href']})
|
||||
else:
|
||||
row_data.append({"text": cell.get_text(strip=True), "link": None})
|
||||
else:
|
||||
# For other cells, just extract text
|
||||
row_data.append(cell.get_text(strip=True))
|
||||
|
||||
if i == 0: # Assume the first row is headers
|
||||
headers = [header if isinstance(header, str) else str(header) for header in row_data]
|
||||
else:
|
||||
table_data.append(row_data)
|
||||
|
||||
# Combine headers with data (last column handled as special case)
|
||||
structured_data = []
|
||||
for row in table_data:
|
||||
row_dict = {}
|
||||
for idx, header in enumerate(headers):
|
||||
header = header if isinstance(header, str) else str(header) # Ensure header is a string
|
||||
# Last column handling for text + link dictionaries
|
||||
if isinstance(row[idx], dict):
|
||||
row_dict[header] = row[idx]
|
||||
else:
|
||||
row_dict[header] = row[idx]
|
||||
structured_data.append(row_dict)
|
||||
|
||||
# Convert to JSON
|
||||
table_json = json.dumps(structured_data, indent=2)
|
||||
print("Extracted JSON with Links:")
|
||||
print(table_json)
|
||||
else:
|
||||
print(f"Failed to scrape the page: {result.error_message}")
|
||||
|
||||
def parse_tools(html):
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
tools = []
|
||||
|
||||
for tool_div in soup.find_all("div", recursive=False):
|
||||
try:
|
||||
# Nazwa narzędzia
|
||||
name = tool_div.find("p").text.strip()
|
||||
|
||||
# Link do narzędzia
|
||||
link = tool_div.find("a", href=True)["href"]
|
||||
|
||||
# Opis
|
||||
description = tool_div.find("p").find_next_sibling("p").text.strip()
|
||||
|
||||
# Ocena
|
||||
rating_span = tool_div.find("span", text=lambda t: t and "Rated" in t)
|
||||
rating = rating_span.text.strip() if rating_span else "No rating"
|
||||
|
||||
# Kategoria
|
||||
tags = [tag.text.strip() for tag in tool_div.find_all("a") if tag.text.startswith("#")]
|
||||
|
||||
# Model płatności
|
||||
payment_span = tool_div.find("span", text=lambda t: t and ("Free" in t or "Paid" in t))
|
||||
payment_model = payment_span.text.strip() if payment_span else "Unknown"
|
||||
|
||||
tools.append({
|
||||
"name": name,
|
||||
"link": link,
|
||||
"description": description,
|
||||
"rating": rating,
|
||||
"categories": tags,
|
||||
"payment_model": payment_model,
|
||||
})
|
||||
except AttributeError:
|
||||
continue # Pomijaj elementy bez wymaganych danych
|
||||
|
||||
return tools
|
||||
|
||||
|
||||
async def fetch_ai_agents():
|
||||
url = "https://www.futurepedia.io/ai-tools/ai-agents?sort=popular&page=1"
|
||||
|
||||
|
||||
# Convert XPath to CSS Selector using a utility or manually
|
||||
css_selector = ".grid-rows-3"
|
||||
|
||||
async with AsyncWebCrawler(
|
||||
browser_type="chromium", # Możesz użyć także "firefox" lub "webkit"
|
||||
headless=True, # Uruchom w trybie headless (bez GUI)
|
||||
verbose=True # Aktywuj szczegółowe logi
|
||||
) as crawler:
|
||||
result = await crawler.arun(
|
||||
url=url,
|
||||
css_selector=css_selector, # Ekstrakcja danych tylko z określonego elementu
|
||||
bypass_cache=True, # Zawsze omijaj cache dla świeżych danych
|
||||
magic=True # Aktywuj wszystkie funkcje antydetekcji
|
||||
)
|
||||
if result.success:
|
||||
|
||||
print("Zawartość: ", result.cleaned_html) # Wyczyść dane HTML
|
||||
else:
|
||||
print("Błąd podczas pobierania danych:", result.error_message)
|
||||
|
||||
# Uruchom asynchronicznie
|
||||
asyncio.run(fetch_ai_agents())
|
||||
|
||||
Reference in New Issue
Block a user