From 9a6f0de41814ef12c7e340d9416f69116c63df27 Mon Sep 17 00:00:00 2001 From: caibotmini Date: Sun, 5 Apr 2026 09:45:30 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E6=B7=BB=E5=8A=A0=E4=B8=80=E5=B0=98?= =?UTF-8?q?=E7=BD=91=E7=88=AC=E8=99=ABv4=20-=20BeautifulSoup=E8=A7=A3?= =?UTF-8?q?=E6=9E=90=20+=20=E6=97=A5=E6=9C=9F=E6=A0=BC=E5=BC=8F=E4=BF=AE?= =?UTF-8?q?=E5=A4=8D?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 使用BeautifulSoup替代失效正则,修复列表页解析 - 修复日期格式 %Y/%m/%d -> %Y/%-m/%-d(匹配网站非补零格式) - 支持增量去重,正确识别当日新帖子 --- crawlers/yichens_spider_v4.py | 724 ++++++++++++++++++++++++++++++++++ 1 file changed, 724 insertions(+) create mode 100644 crawlers/yichens_spider_v4.py diff --git a/crawlers/yichens_spider_v4.py b/crawlers/yichens_spider_v4.py new file mode 100644 index 0000000..e4f99ab --- /dev/null +++ b/crawlers/yichens_spider_v4.py @@ -0,0 +1,724 @@ +"""一尘网爬虫 v4 - 完整版 +- 爬取帖子 + 所有楼层用户信息 +- 优化字段识别(category/special_types/number_features/price) +- 增量去重 + 日志记录 + 防封延迟 +""" +import requests +from bs4 import BeautifulSoup +import re +import time +import yaml +import random +import json +from datetime import datetime +from concurrent.futures import ThreadPoolExecutor, as_completed + +def load_config(): + with open('/root/coolbot-data/config.yaml', 'r') as f: + return yaml.safe_load(f) + +config = load_config() +DB_CONFIG = config['database'] + +BASE_URL = 'http://www4.pm001.net' +BOARD_ID = '151' + +HEADERS = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36', + 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8', + 'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8', +} + +# ============ 数据库操作 ============ + +def get_db_conn(): + import psycopg2 + return psycopg2.connect( + host=DB_CONFIG['host'], + port=DB_CONFIG.get('port', 5432), + user=DB_CONFIG['user'], + password=DB_CONFIG['password'], + database=DB_CONFIG['database'] + ) + +def get_existing_post_ids(): + try: + conn = get_db_conn() + cur = conn.cursor() + cur.execute('SELECT post_id FROM yichens_posts') + existing = set(row[0] for row in cur.fetchall()) + cur.close() + conn.close() + return existing + except Exception as e: + print(f'获取已有post_id失败: {e}') + return set() + +def log_crawl(spider_name, status, items_count, error=''): + try: + conn = get_db_conn() + cur = conn.cursor() + cur.execute(""" + INSERT INTO crawl_logs (spider_name, status, items_count, error_message, started_at, finished_at) + VALUES (%s, %s, %s, %s, %s, %s) + """, (spider_name, status, items_count, error, datetime.now(), datetime.now())) + conn.commit() + cur.close() + conn.close() + except Exception as e: + print(f'写入爬虫日志失败: {e}') + +def save_post(post): + try: + conn = get_db_conn() + cur = conn.cursor() + sql = """ + INSERT INTO yichens_posts ( + post_id, title, content, category, post_type, price, price_unit, + special_types, number_features, author_username, author_id, + contact, has_lifebuoy, reply_count, view_count, + post_time, crawled_at, url + ) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) + ON CONFLICT (post_id) DO UPDATE SET + title = EXCLUDED.title, + content = EXCLUDED.content, + author_username = EXCLUDED.author_username, + contact = EXCLUDED.contact, + reply_count = EXCLUDED.reply_count, + view_count = EXCLUDED.view_count, + price = EXCLUDED.price, + updated_at = CURRENT_TIMESTAMP + """ + cur.execute(sql, ( + post.get('post_id'), + post.get('title'), + post.get('content'), + post.get('category'), + post.get('post_type'), + post.get('price'), + post.get('price_unit', '元'), + post.get('special_types'), + post.get('number_features'), + post.get('author_username'), + post.get('author_id'), + post.get('contact'), + post.get('has_lifebuoy', False), + post.get('reply_count', 0), + post.get('view_count', 0), + post.get('post_time'), + datetime.now(), + post.get('url') + )) + conn.commit() + cur.close() + conn.close() + return True + except Exception as e: + print(f' DB error: {e}') + return False + +def save_member(member): + """保存会员信息到 yichens_members 表""" + if not member or not member.get('user_id'): + return False + try: + conn = get_db_conn() + cur = conn.cursor() + sql = """ + INSERT INTO yichens_members ( + user_id, username, transaction_level, credit_score, + rating_count, post_count, post_points, + has_business_license, real_name, + phone, address, bank_accounts, alipay, + registration_date, updated_at + ) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) + ON CONFLICT (user_id) DO UPDATE SET + username = EXCLUDED.username, + transaction_level = EXCLUDED.transaction_level, + credit_score = EXCLUDED.credit_score, + rating_count = EXCLUDED.rating_count, + post_count = EXCLUDED.post_count, + has_business_license = EXCLUDED.has_business_license, + phone = EXCLUDED.phone, + updated_at = CURRENT_TIMESTAMP + """ + bank_accounts_json = json.dumps(member.get('bank_accounts', []), ensure_ascii=False) + + cur.execute(sql, ( + member.get('user_id'), + member.get('username'), + member.get('transaction_level'), + member.get('credit_score'), + member.get('rating_count', 0), + member.get('post_count', 0), + member.get('post_points', 0), + member.get('has_business_license', False), + member.get('real_name'), + member.get('phone'), + member.get('address'), + bank_accounts_json, + member.get('alipay'), + member.get('registration_date'), + datetime.now() + )) + conn.commit() + cur.close() + conn.close() + return True + except Exception as e: + print(f' Save member error: {e}') + return False + +# ============ 字段解析 ============ + +def parse_category(text): + """识别品类:龙钞、蛇钞、马钞" + 规则:标题或正文中有 龙钞/小龙钞/马钞/蛇钞,或者单独的 龙/马/蛇 字(作为钱币品种) + """ + if not text: + return '其他' + text = str(text) + + # 优先精确匹配 + if any(k in text for k in ['龙钞', '小龙钞']): + return '龙钞' + if any(k in text for k in ['蛇钞']): + return '蛇钞' + if any(k in text for k in ['马钞', '马年']): + return '马钞' + + # 单字匹配(作为品种上下文) + if '龙' in text: + exclude = ['龙年版', '龙年纪念', '成龙', '恐龙', '龙凤', '天龙', '卧龙', '龙魂', '龙珠笔', '龙马', '龙马精'] + if not any(e in text for e in exclude): + return '龙钞' + if '马' in text: + exclude = ['马自达', '马德里', '马蹄', '马化腾', '马虎', '马蜂窝', '马马虎虎'] + if not any(e in text for e in exclude): + return '马钞' + if '蛇' in text: + exclude = ['眼镜蛇', '蟒蛇', '蛇毒', '捕蛇', '蛇皮'] + if not any(e in text for e in exclude): + return '蛇钞' + + return '其他' + +def parse_special_types(title): + """识别特殊规格:标百(刀/刀货)、标十、单张、捆" + """ + if not title: + return None + title = str(title) + specials = [] + + # 标十 + if '标十' in title: + specials.append('标十') + + # 标百(包含刀、刀货) + if '标百' in title: + specials.append('标百') + elif any(k in title for k in ['刀', '刀货']): + specials.append('刀') + + # 单张/散张 + if any(k in title for k in ['单张', '散张']): + specials.append('单张') + + # 捆 + if '捆' in title: + specials.append('捆') + + # 百连/千连 + if '千连' in title: + specials.append('千连') + elif '百连' in title: + specials.append('百连') + + # 救生圈 + if any(k in title for k in ['大救生圈', '救生圈']): + specials.append('救生圈') + + return '|'.join(specials) if specials else None + +def parse_number_features(title): + """识别号码特征:圆圆、倒置、如意、朦胧、金马、金山、天马、钻石、永恒、无4、无47、带4、无347、无247、豹子、狮子、老虎、大象" + """ + if not title: + return None + title = str(title) + features = [] + + feature_map = { + '倒置': ['倒置'], + '如意': ['如意'], + '朦胧': ['朦胧'], + '金马': ['金马'], + '金山': ['金山'], + '天马': ['天马'], + '钻石': ['钻石'], + '永恒': ['永恒'], + '圆圆': ['圆圆'], + '无4': ['无4', '无四'], + '无47': ['无47'], + '无247': ['无247'], + '无34': ['无34'], + '无347': ['无347'], + '带4': ['带4'], + '豹子': ['豹子'], + '狮子': ['狮子'], + '老虎': ['老虎'], + '大象': ['大象'], + '生日': ['生日'], + '满号': ['满号'], + '首日': ['首日'], + } + + for feature, keywords in feature_map.items(): + if any(k in title for k in keywords): + features.append(feature) + + return '|'.join(features) if features else None + +def parse_price(text): + """从文本解析价格(标准格式用正则,非标准用启发式) + 标准:XXX元/组、XXX/组、XXX元/张、XXX/刀 + 非标准:标题中的模糊价格需要结合上下文 + """ + if not text: + return None, None + + # 优先从标题精确匹配 + patterns = [ + # 格式:数字+元+单位 + (r'(\d{4,5})\s*元\s*/\s*[组张刀条]', '元/组'), + (r'(\d{4,5})\s*/\s*[组张刀条]', '元/组'), + (r'(\d{4,5})\s*元', '元'), + # 求购格式:收XXX元、收XXX + (r'收\s*(\d{3,5})\s*元?', '元'), + # 出售格式:出XXX元、售XXX + (r'[出售售]\s*(\d{3,5})\s*元?', '元'), + ] + + for pattern, unit in patterns: + match = re.search(pattern, text) + if match: + try: + price = float(match.group(1)) + # 合理性校验 + if 10 <= price <= 999999: + return price, unit + except: + pass + + return None, None + +def parse_price_from_content(title, content): + """从正文提取所有价格,返回最高价格和单位" + 用于多货品帖子,取最高出价/要价 + """ + if not content: + return parse_price(title) + + # 收集所有符合格式的价格 + prices = [] + price_pattern = r'(\d{3,5})\s*元|收\s*(\d{3,5})|(\d{4,5})\s*/\s*[组张刀]' + + for match in re.finditer(price_pattern, content): + for group in match.groups(): + if group: + try: + p = float(group) + if 10 <= p <= 999999: + prices.append(p) + except: + pass + + if prices: + max_price = max(prices) + # 确定单位 + unit = '元' + if '元/组' in content or '/组' in content: + unit = '元/组' + elif '元/张' in content or '/张' in content: + unit = '元/张' + elif '元/刀' in content or '/刀' in content: + unit = '元/刀' + return max_price, unit + + return parse_price(title) + +def parse_post_type(title): + """识别交易类型:deal(出售)/want(求购)/normal" + """ + if not title: + return 'normal' + title = str(title) + + # 出售关键词 + deal_keywords = ['出售', '转让', '卖', '低价', '低出', '快出', '亏出', '吐血', '清仓', '出 ', '批出', '批售', '甩卖', '特价'] + if any(k in title for k in deal_keywords): + return 'deal' + + # 求购关键词 + want_keywords = ['求购', '收购', '收 ', '要 ', '收', '求', '要'] + if any(k in title for k in want_keywords): + return 'want' + + return 'normal' + +# ============ 页面爬取 ============ + +def crawl_detail(post_id, url): + """爬取帖子详情页,提取:内容、用户信息、回复信息""" + try: + resp = requests.get(url, headers=HEADERS, timeout=15) + resp.encoding = 'gb2312' + html = resp.text + soup = BeautifulSoup(html, 'html.parser') + + result = { + 'content': None, + 'author_username': None, + 'author_id': None, + 'author_info': {}, + 'members': [], + 'reply_count': 0, + 'view_count': 0, + 'post_time': None, + 'contact': None, + 'has_lifebuoy': False + } + + # 1. 提取浏览数 + view_match = re.search(r'您是本帖的第\s*(\d+)\s*个阅读者', html) + if view_match: + result['view_count'] = int(view_match.group(1)) + + # 2. 提取回复数(数 postbottom1/2 数量) + postbottom_count = len(soup.find_all('div', class_=lambda x: x and 'postbottom' in str(x))) + # 减去1因为第一个是主帖 + result['reply_count'] = max(0, postbottom_count - 1) + + # 3. 解析每个楼层 + page_text = soup.get_text(separator='\n', strip=True) + + # 找所有 postuserinfo 块 + userinfo_blocks = soup.find_all('div', class_='postuserinfo') + + for i, userinfo in enumerate(userinfo_blocks): + user_data = parse_userinfo_block(userinfo) + if user_data and user_data.get('user_id'): + result['members'].append(user_data) + + # 第一条是楼主 + if i == 0: + result['author_username'] = user_data.get('username') + result['author_id'] = user_data.get('user_id') + result['author_info'] = user_data + + # 4. 提取帖子内容 + post_div = soup.find('div', class_='post') + if post_div: + text = post_div.get_text(separator='\n', strip=True) + result['content'] = text[:8000] + if '救生圈' in text: + result['has_lifebuoy'] = True + + # 5. 提取联系方式 + phones = re.findall(r'1[3-9]\d{9}', page_text) + if phones: + result['contact'] = ','.join(dict.fromkeys(phones[:3])) + + # 6. 提取发帖时间(从第一个postbottom) + time_div = soup.find('div', class_='postbottom1') + if time_div: + time_match = re.search(r'(\d{4})/(\d{1,2})/(\d{1,2})\s+(\d{1,2}):(\d{2})', time_div.get_text()) + if time_match: + result['post_time'] = f"{time_match.group(1)}-{int(time_match.group(2)):02d}-{int(time_match.group(3)):02d} {time_match.group(4)}:{time_match.group(5)}:00" + + return result + except Exception as e: + print(f' Crawl error: {e}') + return {} + +def parse_userinfo_block(userinfo_div): + """解析单个用户的 userinfo div,提取完整用户信息""" + try: + text = userinfo_div.get_text(separator='\n', strip=True) + + # user_id - from showyyzz() JS call or j_gbook links + user_id = None + userinfo_html = str(userinfo_div) + js_match = re.search(r'showyyzz\((\d+)\)', userinfo_html) + if js_match: + user_id = js_match.group(1) + else: + link_match = re.search(r'j_gbook_add\.asp\?id=(\d+)', userinfo_html) + if link_match: + user_id = link_match.group(1) + else: + dispuser_match = re.search(r'dispuser\.asp\?id=(\d+)', userinfo_html) + if dispuser_match: + user_id = dispuser_match.group(1) + + # username + username = None + if user_link: + username = user_link.get_text(strip=True) + + # transaction_level + level_match = re.search(r'交易等级[::]\s*([^\n]+)', text) + transaction_level = level_match.group(1).strip() if level_match else None + + # credit_score + credit_match = re.search(r'信用积分[::]\s*(\d+)', text) + credit_score = int(credit_match.group(1)) if credit_match else None + + # rating_count + rating_match = re.search(r'评分次数[::]\s*(\d+)', text) + rating_count = int(rating_match.group(1)) if rating_match else 0 + + # post_count + post_match = re.search(r'发贴次数[::]\s*(\d+)', text) + post_count = int(post_match.group(1)) if post_match else 0 + + # post_points + points_match = re.search(r'发帖积分[::]\s*(\d+)', text) + post_points = int(points_match.group(1)) if points_match else 0 + + # registration_date + reg_match = re.search(r'注册日期[::]\s*(\d{4})年(\d{1,2})月(\d{1,2})日', text) + registration_date = None + if reg_match: + try: + registration_date = f"{reg_match.group(1)}-{int(reg_match.group(2)):02d}-{int(reg_match.group(3)):02d}" + except: + pass + + # business license + has_business_license = '点击查看' in text or '已认证' in text + + # real_name (from 认证员注) + real_name = None + name_match = re.search(r'姓名[::]([^\s\n]+)', text) + if name_match: + real_name = name_match.group(1) + + # phone (from userinfo or nearby content) + phone_match = re.search(r'电话[::]\s*([^\s\n]+)', text) + phone = phone_match.group(1).strip() if phone_match else None + + # address + addr_match = re.search(r'地址[::]\s*([^\n]+)', text) + address = addr_match.group(1).strip() if addr_match else None + + # bank_accounts + bank_accounts = [] + bank_types = ['农行', '工行', '建行', '中行', '交行', '招行', '兴业', '民生', '光大', '中信'] + for bank in bank_types: + if bank in text: + # 匹配账号和户名 + matches = re.findall(rf'{bank}[::]\s*(\d+)\s*([^\s\n]{{2,10}})', text) + for acc, name in matches: + bank_accounts.append({'bank': bank, 'account': acc.strip(), 'name': name.strip()}) + + # alipay + alipay = None + + return { + 'user_id': user_id, + 'username': username, + 'transaction_level': transaction_level, + 'credit_score': credit_score, + 'rating_count': rating_count, + 'post_count': post_count, + 'post_points': post_points, + 'registration_date': registration_date, + 'has_business_license': has_business_license, + 'real_name': real_name, + 'phone': phone, + 'address': address, + 'bank_accounts': bank_accounts, + 'alipay': alipay + } + except Exception as e: + print(f' Parse userinfo error: {e}') + return {} + +def crawl_list_page(page): + """爬取列表页,提取当天所有帖子(使用BeautifulSoup解析)""" + time.sleep(random.uniform(0.3, 1.5)) # 防封延迟 + + url = f'{BASE_URL}/index.asp?boardid={BOARD_ID}&page={page}' + try: + resp = requests.get(url, headers=HEADERS, timeout=15) + resp.encoding = 'gb2312' + html = resp.text + soup = BeautifulSoup(html, 'html.parser') + + posts_data = [] + today = datetime.now().strftime('%Y/%-m/%-d') + + # 找到所有包含 boardID=151 链接的
+ listtitle_divs = soup.find_all('div', class_='listtitle') + + for div in listtitle_divs: + # 在 listtitle div 内找 标签 + link = div.find('a', href=re.compile(r'dispbbs\.asp\?boardID=151&ID=\d+')) + if not link: + continue + + href = link.get('href', '') + id_match = re.search(r'ID=(\d+)', href) + if not id_match: + continue + post_id = id_match.group(1) + + # 从 title 属性提取信息:格式《标题》\n作者:xxx\n发表于:2026/4/5 9:35:00 + title_attr = link.get('title', '') + if not title_attr: + continue + + # 提取日期 + date_match = re.search(r'(\d{4}/\d{1,2}/\d{1,2})', title_attr) + if not date_match: + continue + post_date = date_match.group(1) + + # 只爬当天 + if post_date != today: + continue + + # 提取标题 + title_match = re.search(r'《([^》]+)》', title_attr) + title = title_match.group(1).strip() if title_match else '' + + if not title or len(title) < 5: + continue + + skip_titles = ['栏目交易规范', '固', '精华', '_TOP', '锁', '查看', '页面加载', '投资资讯', '版块主题', '固顶主题'] + if any(t in title for t in skip_titles): + continue + + full_url = f'{BASE_URL}/dispbbs.asp?boardid={BOARD_ID}&id={post_id}&page=1' + + post = { + 'post_id': post_id, + 'title': title, + 'url': full_url, + 'category': parse_category(title), + 'post_type': parse_post_type(title), + 'special_types': parse_special_types(title), + 'number_features': parse_number_features(title), + } + + # 从标题提取价格 + price, unit = parse_price(title) + post['price'] = price + post['price_unit'] = unit + post['has_lifebuoy'] = '救生圈' in title + + posts_data.append(post) + + return posts_data + except Exception as e: + print(f'Crawl page {page} error: {e}') + return [] + +def process_post(post): + """处理单个帖子:爬详情 + 提取价格 + 保存会员 + 入库""" + detail = crawl_detail(post['post_id'], post['url']) + post.update(detail) + + # 更新作者信息 + if detail.get('author_info'): + author = detail['author_info'] + post['author_username'] = author.get('username') + post['author_id'] = author.get('user_id') + + # 保存所有会员 + for member in detail.get('members', []): + if member.get('user_id'): + save_member(member) + + # 从正文提取价格(如果标题没有) + if not post.get('price') and detail.get('content'): + price, unit = parse_price_from_content(post.get('title', ''), detail.get('content', '')) + if price: + post['price'] = price + post['price_unit'] = unit + + # 合并 content + if detail.get('content'): + post['content'] = detail['content'] + + if save_post(post): + return 1 + return 0 + +def run(max_pages=5, workers=10): + spider_name = 'yichens_spider_v4' + start_time = datetime.now() + + print('=' * 60) + print(f'一尘网爬虫 v4 [{spider_name}]') + print(f'时间: {start_time.strftime("%Y-%m-%d %H:%M:%S")}') + print(f'并发: {workers} 线程') + print('=' * 60) + + # 1. 已有post_id(增量去重) + existing_ids = get_existing_post_ids() + print(f'数据库已有帖子: {len(existing_ids)} 条') + + # 2. 爬列表页 + all_posts = [] + for page in range(1, max_pages + 1): + print(f'扫描第 {page} 页...') + posts = crawl_list_page(page) + if not posts: + if page > 2: + print(f'第 {page} 页无新数据,停止') + break + else: + all_posts.extend(posts) + print(f' 第 {page} 页: {len(posts)} 条') + + # 3. 增量过滤 + new_posts = [p for p in all_posts if p['post_id'] not in existing_ids] + print(f'\n共找到 {len(all_posts)} 条今日帖子,新增 {len(new_posts)} 条') + + if not new_posts: + print('没有新帖子,退出') + log_crawl(spider_name, 'success', 0, '') + return 0 + + # 4. 并发处理 + saved = 0 + member_count = 0 + print(f'开始并发爬取 ({workers} 线程)...') + + with ThreadPoolExecutor(max_workers=workers) as executor: + future_to_post = {executor.submit(process_post, post): post for post in new_posts} + + for i, future in enumerate(as_completed(future_to_post)): + post = future_to_post[future] + try: + if future.result(): + saved += 1 + # 统计会员数(从 post 对象获取) + if post.get('author_id'): + member_count += 1 + if (i + 1) % 20 == 0: + print(f'进度: {i+1}/{len(new_posts)}, 已保存: {saved}') + except Exception as e: + print(f'处理 {post["post_id"]} 异常: {e}') + + elapsed = (datetime.now() - start_time).total_seconds() + print(f'\n完成! 新增帖子 {saved} 条, 会员 {member_count} 人, 耗时 {elapsed:.1f}秒') + + log_crawl(spider_name, 'success', saved, '') + return saved + +if __name__ == '__main__': + run(max_pages=5, workers=10)