CoolBotDataSys/crawlers/yichens_spider.py

141 lines
4.9 KiB
Python
Raw Normal View History

"""一尘网爬虫 - 龙钞价格数据采集"""
import requests
from bs4 import BeautifulSoup
import re
import time
import logging
from datetime import datetime
from database import db
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
class YichensSpider:
"""一尘网爬虫"""
def __init__(self):
self.name = "一尘网"
self.source = "yichens"
self.base_url = "https://www.yichens.com"
self.headers = {
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
def parse_price(self, price_str: str) -> float:
"""解析价格字符串"""
if not price_str:
return 0.0
match = re.search(r"[\d.]+", price_str.replace(",", ""))
return float(match.group()) if match else 0.0
def crawl_longchao_prices(self) -> list:
"""
采集龙钞价格数据
Returns:
list: 采集到的价格数据列表
"""
logger.info("开始采集龙钞价格数据...")
items = []
try:
# TODO: 根据实际网站结构调整URL和解析逻辑
url = f"{self.base_url}/nbbs/list?category=longchao"
logger.info(f"请求URL: {url}")
response = requests.get(url, headers=self.headers, timeout=10)
if response.status_code == 200:
soup = BeautifulSoup(response.text, "lxml")
logger.info(f"页面获取成功,内容长度: {len(response.text)}")
# TODO: 根据实际网页结构解析价格数据
else:
logger.warning(f"HTTP状态码: {response.status_code}")
except requests.RequestException as e:
logger.error(f"网络请求失败: {e}")
return items
def save_to_db(self, items: list) -> int:
"""保存到数据库"""
if not items:
logger.info("无新数据需要保存")
return 0
saved = 0
with db.get_cursor() as cursor:
for item in items:
try:
cursor.execute("""
INSERT INTO collections (name, category, serial_number, cost_price, status)
VALUES (%s, %s, %s, %s, %s)
ON DUPLICATE KEY UPDATE cost_price = VALUES(cost_price), updated_at = NOW()
""", (
item.get("name"),
item.get("category"),
item.get("serial_number"),
item.get("price"),
"in_collection"
))
collection_id = cursor.lastrowid if cursor.lastrowid else 0
cursor.execute("""
INSERT INTO price_history (collection_id, source, price, price_unit, price_type, url)
VALUES (%s, %s, %s, %s, %s, %s)
""", (
collection_id,
self.source,
item.get("price"),
item.get("unit", "元/张"),
item.get("price_type", "挂牌价"),
item.get("url", "")
))
saved += 1
except Exception as e:
logger.error(f"保存失败: {e}")
logger.info(f"成功保存 {saved} 条记录")
return saved
def run(self) -> list:
"""执行爬虫"""
log_id = self._log_start()
try:
items = self.crawl_longchao_prices()
saved = self.save_to_db(items)
self._log_finish(log_id, "success", saved)
return items
except Exception as e:
logger.error(f"爬虫执行失败: {e}")
self._log_finish(log_id, "failed", 0, str(e))
return []
def _log_start(self) -> int:
"""记录爬虫开始"""
with db.get_cursor() as cursor:
cursor.execute("""
INSERT INTO crawl_logs (source, status, started_at)
VALUES (%s, %s, NOW())
""", (self.source, "running"))
return cursor.lastrowid
def _log_finish(self, log_id: int, status: str, items_count: int, error: str = ""):
"""记录爬虫结束"""
with db.get_cursor() as cursor:
cursor.execute("""
UPDATE crawl_logs
SET status = %s, items_count = %s, error_message = %s, finished_at = NOW()
WHERE id = %s
""", (status, items_count, error, log_id))
if __name__ == "__main__":
spider = YichensSpider()
spider.run()