金融门户使用 Cloudflare Turnstile 和 reCAPTCHA 保护股票报价、收益数据和分析师报告。验证码在快速符号查找、历史数据下载和筛选器查询期间触发。以下是如何维护跨金融网站的可靠数据收集。
金融门户上的验证码模式
| 数据类型 | 门户示例 | 验证码类型 | 扳机 |
|---|---|---|---|
| 实时报价 | 财经门户网站 | Cloudflare Turnstile | 快速符号查找 |
| 历史价格 | 数据提供者 | reCAPTCHA v2 | 批量 CSV 下载 |
| 财务报表 | SEC 备案网站 | 图片验证码 | 重复 EDGAR 查询 |
| 筛选结果 | 股票筛选器 | Cloudflare 验证流程 | 复杂的过滤查询 |
| 分析师评级 | 研究门户网站 | reCAPTCHA v3 | 多页面浏览量 |
股票数据收集器
import requests
import time
import re
from datetime import datetime, timedelta
class StockDataCollector:
def __init__(self, api_key):
self.api_key = api_key
self.session = requests.Session()
self.session.headers.update({
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
})
def get_quote(self, portal_url, symbol):
"""Get current stock quote, solving CAPTCHAs if needed."""
url = f"{portal_url}/quote/{symbol}"
response = self.session.get(url)
if self._is_captcha_page(response):
response = self._solve_and_retry(response, url)
return self._parse_quote(response.text, symbol)
def get_historical(self, portal_url, symbol, days=365):
"""Download historical price data."""
url = f"{portal_url}/history/{symbol}"
params = {
"period": f"{days}d",
"interval": "1d"
}
response = self.session.get(url, params=params)
if self._is_captcha_page(response):
response = self._solve_and_retry(response, url)
return self._parse_historical(response.text)
def scan_symbols(self, portal_url, symbols, delay=2):
"""Collect quotes for multiple symbols."""
results = {}
for symbol in symbols:
try:
results[symbol] = self.get_quote(portal_url, symbol)
time.sleep(delay)
except Exception as e:
results[symbol] = {"error": str(e)}
return results
def _is_captcha_page(self, response):
return (
response.status_code == 403 or
"cf-turnstile" in response.text or
"challenges.cloudflare.com" in response.text
)
def _solve_and_retry(self, response, url):
match = re.search(r'data-sitekey="(0x[^"]+)"', response.text)
if not match:
# Fall back to reCAPTCHA detection
match = re.search(r'data-sitekey="([^"]+)"', response.text)
if match:
return self._solve_recaptcha_and_retry(match.group(1), url)
raise ValueError("No CAPTCHA sitekey found")
resp = requests.post("https://ocr.captchaai.com/in.php", data={
"key": self.api_key,
"method": "turnstile",
"sitekey": match.group(1),
"pageurl": url,
"json": 1
})
task_id = resp.json()["request"]
for _ in range(60):
time.sleep(3)
result = requests.get("https://ocr.captchaai.com/res.php", params={
"key": self.api_key,
"action": "get",
"id": task_id,
"json": 1
})
data = result.json()
if data["status"] == 1:
return self.session.post(url, data={
"cf-turnstile-response": data["request"]
})
raise TimeoutError("CAPTCHA solve timed out")
def _solve_recaptcha_and_retry(self, site_key, url):
resp = requests.post("https://ocr.captchaai.com/in.php", data={
"key": self.api_key,
"method": "userrecaptcha",
"googlekey": site_key,
"pageurl": url,
"json": 1
})
task_id = resp.json()["request"]
for _ in range(60):
time.sleep(3)
result = requests.get("https://ocr.captchaai.com/res.php", params={
"key": self.api_key,
"action": "get",
"id": task_id,
"json": 1
})
data = result.json()
if data["status"] == 1:
return self.session.post(url, data={
"g-recaptcha-response": data["request"]
})
raise TimeoutError("reCAPTCHA solve timed out")
def _parse_quote(self, html, symbol):
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, "html.parser")
def text_or_none(node):
return node.text.strip() if node and node.text else None
return {
"symbol": symbol,
"price": text_or_none(soup.select_one("[data-field='regularMarketPrice'], .price")),
"change": text_or_none(soup.select_one("[data-field='regularMarketChange'], .change")),
"volume": text_or_none(soup.select_one("[data-field='regularMarketVolume'], .volume")),
"market_cap": text_or_none(soup.select_one("[data-field='marketCap'], .market-cap")),
"timestamp": datetime.now().isoformat()
}
def _parse_historical(self, html):
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, "html.parser")
rows = []
for row in soup.select("table tr")[1:]: # Skip header
cells = [td.text.strip() for td in row.select("td")]
if len(cells) >= 6:
rows.append({
"date": cells[0],
"open": cells[1],
"high": cells[2],
"low": cells[3],
"close": cells[4],
"volume": cells[5]
})
return rows
# Usage
collector = StockDataCollector("YOUR_API_KEY")
# Single quote
quote = collector.get_quote("https://finance.example.com", "AAPL")
print(f"AAPL: ${quote['price']} ({quote['change']})")
# Scan multiple symbols
portfolio = collector.scan_symbols(
"https://finance.example.com",
["AAPL", "GOOGL", "MSFT", "AMZN", "TSLA"]
)
具有验证码处理功能的市场筛选器 (JavaScript)
class MarketScreener {
constructor(apiKey) {
this.apiKey = apiKey;
}
async screenStocks(portalUrl, filters) {
const params = new URLSearchParams(filters);
const response = await fetch(`${portalUrl}/screener?${params}`);
const html = await response.text();
if (html.includes('cf-turnstile') || response.status === 403) {
return this.solveAndScreen(portalUrl, filters, html);
}
return this.parseScreenerResults(html);
}
async solveAndScreen(portalUrl, filters, html) {
const match = html.match(/data-sitekey="(0x[^"]+)"/);
if (!match) throw new Error('Turnstile sitekey not found');
const submitResp = await fetch('https://ocr.captchaai.com/in.php', {
method: 'POST',
body: new URLSearchParams({
key: this.apiKey,
method: 'turnstile',
sitekey: match[1],
pageurl: portalUrl,
json: '1'
})
});
const { request: taskId } = await submitResp.json();
for (let i = 0; i < 60; i++) {
await new Promise(r => setTimeout(r, 3000));
const result = await fetch(
`https://ocr.captchaai.com/res.php?key=${this.apiKey}&action=get&id=${taskId}&json=1`
);
const data = await result.json();
if (data.status === 1) {
const response = await fetch(`${portalUrl}/screener`, {
method: 'POST',
body: new URLSearchParams({
...filters,
'cf-turnstile-response': data.request
})
});
return this.parseScreenerResults(await response.text());
}
}
throw new Error('Turnstile solve timed out');
}
parseScreenerResults(html) {
const rows = [];
const tableMatch = html.match(/<table[^>]*>[\s\S]*?<\/table>/i);
if (!tableMatch) return rows;
const rowMatches = tableMatch[0].matchAll(/<tr[^>]*>([\s\S]*?)<\/tr>/gi);
for (const row of rowMatches) {
const cells = [...row[1].matchAll(/<td[^>]*>([\s\S]*?)<\/td>/gi)]
.map(m => m[1].replace(/<[^>]+>/g, '').trim());
if (cells.length >= 4) {
rows.push({
symbol: cells[0],
price: cells[1],
change: cells[2],
volume: cells[3]
});
}
}
return rows;
}
}
// Usage
const screener = new MarketScreener('YOUR_API_KEY');
const results = await screener.screenStocks('https://finance.example.com', {
sector: 'technology',
marketCap: 'large',
peRatio: '<25'
});
按数据类型划分的收集频率
| 数据类型 | 推荐间隔 | 验证码频率 |
|---|---|---|
| 实时报价 | 1-5分钟 | 高 – 使用 API(如果可用) |
| 收盘价格 | 每天收盘后一次 | 低的 |
| 财务报表 | 季刊 | 最小 |
| 筛选结果 | 日常的 | 缓和 |
| 分析师评级 | 每周 | 低的 |
故障排除
| 问题 | 原因 | 处理方式 |
|---|---|---|
| 根据每个要求旋转栅门 | 每次都有新会话 | 跨请求保留 cookie |
| 历史数据不完整 | CAPTCHA 后面的分页 | 每页求解,遵循分页链接 |
| 报价数据过时 | 已提供缓存响应 | 添加缓存清除查询参数 |
| 429 速率限制 | 请求过多 | 增加延迟、QA 测试会话 |
常问问题
金融 API 比使用验证码抓取更好吗?
免费 API(如某些提供商的基本层)涵盖常见数据,但有速率限制。使用 CaptchaAI 进行网络抓取可以访问通过 API 无法获得的数据 - 筛选器过滤器、分析师评论和利基财务数据。
如何立即处理实时报价验证码?
在市场交易时间内预先验证您的会话。 Cloudflare 的 qa_session_cookie cookie 持续 15-30 分钟,因此解决一次并在该窗口内发出多个请求。
我可以同时从多个金融门户收集数据吗?
是的。每个门户使用单独的会话以及独立的 CaptchaAI 任务提交。 CaptchaAI 处理跨不同站点的并发求解。
相关文章
下一步
可靠地收集市场数据 -获取您的 CaptchaAI API 密钥并自动处理金融门户验证码。